@a11ign/screenreader-fleet 0.0.0-reserved.0 → 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (147) hide show
  1. package/LICENSE +661 -0
  2. package/README.md +94 -2
  3. package/dist/capture-client.d.mts +49 -0
  4. package/dist/capture-client.d.mts.map +1 -0
  5. package/dist/capture-client.mjs +352 -0
  6. package/dist/capture-client.mjs.map +1 -0
  7. package/dist/check-worker-code.d.mts +34 -0
  8. package/dist/check-worker-code.d.mts.map +1 -0
  9. package/dist/check-worker-code.mjs +173 -0
  10. package/dist/check-worker-code.mjs.map +1 -0
  11. package/dist/cli-flags.d.mts +71 -0
  12. package/dist/cli-flags.d.mts.map +1 -0
  13. package/dist/cli-flags.mjs +207 -0
  14. package/dist/cli-flags.mjs.map +1 -0
  15. package/dist/code-drift.d.mts +140 -0
  16. package/dist/code-drift.d.mts.map +1 -0
  17. package/dist/code-drift.mjs +284 -0
  18. package/dist/code-drift.mjs.map +1 -0
  19. package/dist/command-line-census.d.mts +33 -0
  20. package/dist/command-line-census.d.mts.map +1 -0
  21. package/dist/command-line-census.mjs +96 -0
  22. package/dist/command-line-census.mjs.map +1 -0
  23. package/dist/compare-workers.d.mts +3 -0
  24. package/dist/compare-workers.d.mts.map +1 -0
  25. package/dist/compare-workers.mjs +332 -0
  26. package/dist/compare-workers.mjs.map +1 -0
  27. package/dist/control-plane-isolation.d.mts +45 -0
  28. package/dist/control-plane-isolation.d.mts.map +1 -0
  29. package/dist/control-plane-isolation.mjs +67 -0
  30. package/dist/control-plane-isolation.mjs.map +1 -0
  31. package/dist/deploy-worker.d.mts +3 -0
  32. package/dist/deploy-worker.d.mts.map +1 -0
  33. package/dist/deploy-worker.mjs +333 -0
  34. package/dist/deploy-worker.mjs.map +1 -0
  35. package/dist/doctor.d.mts +216 -0
  36. package/dist/doctor.d.mts.map +1 -0
  37. package/dist/doctor.mjs +962 -0
  38. package/dist/doctor.mjs.map +1 -0
  39. package/dist/fleet-consistency.d.mts +235 -0
  40. package/dist/fleet-consistency.d.mts.map +1 -0
  41. package/dist/fleet-consistency.mjs +436 -0
  42. package/dist/fleet-consistency.mjs.map +1 -0
  43. package/dist/fleet-env.d.mts +228 -0
  44. package/dist/fleet-env.d.mts.map +1 -0
  45. package/dist/fleet-env.mjs +509 -0
  46. package/dist/fleet-env.mjs.map +1 -0
  47. package/dist/fleet-scripts.d.mts +11 -0
  48. package/dist/fleet-scripts.d.mts.map +1 -0
  49. package/dist/fleet-scripts.mjs +41 -0
  50. package/dist/fleet-scripts.mjs.map +1 -0
  51. package/dist/git-safe-env.d.mts +10 -0
  52. package/dist/git-safe-env.d.mts.map +1 -0
  53. package/dist/git-safe-env.mjs +44 -0
  54. package/dist/git-safe-env.mjs.map +1 -0
  55. package/dist/guest-run.d.mts +26 -0
  56. package/dist/guest-run.d.mts.map +1 -0
  57. package/dist/guest-run.mjs +164 -0
  58. package/dist/guest-run.mjs.map +1 -0
  59. package/dist/host-address.d.mts +33 -0
  60. package/dist/host-address.d.mts.map +1 -0
  61. package/dist/host-address.mjs +105 -0
  62. package/dist/host-address.mjs.map +1 -0
  63. package/dist/host-capacity.d.mts +64 -0
  64. package/dist/host-capacity.d.mts.map +1 -0
  65. package/dist/host-capacity.mjs +152 -0
  66. package/dist/host-capacity.mjs.map +1 -0
  67. package/dist/host-metrics.d.mts +116 -0
  68. package/dist/host-metrics.d.mts.map +1 -0
  69. package/dist/host-metrics.mjs +201 -0
  70. package/dist/host-metrics.mjs.map +1 -0
  71. package/dist/index.d.ts +23 -0
  72. package/dist/index.d.ts.map +1 -0
  73. package/dist/index.js +25 -0
  74. package/dist/index.js.map +1 -0
  75. package/dist/local-vm.d.ts +125 -0
  76. package/dist/local-vm.d.ts.map +1 -0
  77. package/dist/local-vm.js +360 -0
  78. package/dist/local-vm.js.map +1 -0
  79. package/dist/measure-guard.d.mts +34 -0
  80. package/dist/measure-guard.d.mts.map +1 -0
  81. package/dist/measure-guard.mjs +73 -0
  82. package/dist/measure-guard.mjs.map +1 -0
  83. package/dist/normalise-fleet.d.mts +2 -0
  84. package/dist/normalise-fleet.d.mts.map +1 -0
  85. package/dist/normalise-fleet.mjs +76 -0
  86. package/dist/normalise-fleet.mjs.map +1 -0
  87. package/dist/npm-cli-executable.d.mts +42 -0
  88. package/dist/npm-cli-executable.d.mts.map +1 -0
  89. package/dist/npm-cli-executable.mjs +159 -0
  90. package/dist/npm-cli-executable.mjs.map +1 -0
  91. package/dist/probe-outcome.d.mts +89 -0
  92. package/dist/probe-outcome.d.mts.map +1 -0
  93. package/dist/probe-outcome.mjs +104 -0
  94. package/dist/probe-outcome.mjs.map +1 -0
  95. package/dist/protocol-guard.d.mts +34 -0
  96. package/dist/protocol-guard.d.mts.map +1 -0
  97. package/dist/protocol-guard.mjs +121 -0
  98. package/dist/protocol-guard.mjs.map +1 -0
  99. package/dist/source-walk.d.mts +12 -0
  100. package/dist/source-walk.d.mts.map +1 -0
  101. package/dist/source-walk.mjs +56 -0
  102. package/dist/source-walk.mjs.map +1 -0
  103. package/dist/transient-fault.d.mts +6 -0
  104. package/dist/transient-fault.d.mts.map +1 -0
  105. package/dist/transient-fault.mjs +86 -0
  106. package/dist/transient-fault.mjs.map +1 -0
  107. package/dist/utm-deprecated.d.mts +6 -0
  108. package/dist/utm-deprecated.d.mts.map +1 -0
  109. package/dist/utm-deprecated.mjs +23 -0
  110. package/dist/utm-deprecated.mjs.map +1 -0
  111. package/dist/worker-code-check.d.mts +29 -0
  112. package/dist/worker-code-check.d.mts.map +1 -0
  113. package/dist/worker-code-check.mjs +78 -0
  114. package/dist/worker-code-check.mjs.map +1 -0
  115. package/dist/worker-health.d.mts +56 -0
  116. package/dist/worker-health.d.mts.map +1 -0
  117. package/dist/worker-health.mjs +73 -0
  118. package/dist/worker-health.mjs.map +1 -0
  119. package/dist/worker-http.d.mts +103 -0
  120. package/dist/worker-http.d.mts.map +1 -0
  121. package/dist/worker-http.mjs +277 -0
  122. package/dist/worker-http.mjs.map +1 -0
  123. package/dist/worker-stats.d.mts +66 -0
  124. package/dist/worker-stats.d.mts.map +1 -0
  125. package/dist/worker-stats.mjs +143 -0
  126. package/dist/worker-stats.mjs.map +1 -0
  127. package/package.json +96 -4
  128. package/src/local-worker/autounattend.xml +280 -0
  129. package/src/local-worker/build-vm.sh +218 -0
  130. package/src/local-worker/clone-worker.sh +141 -0
  131. package/src/local-worker/create-utm-vm.sh +202 -0
  132. package/src/local-worker/fetch-windows-iso.sh +238 -0
  133. package/src/local-worker/first-boot.cmd +58 -0
  134. package/src/local-worker/worker-ctl.sh +442 -0
  135. package/src/provisioning/README.md +28 -0
  136. package/src/provisioning/apply-foreground-lock-timeout.ps1 +71 -0
  137. package/src/provisioning/bare-metal/README.md +213 -0
  138. package/src/provisioning/bare-metal/a11y-bootstrap.service +58 -0
  139. package/src/provisioning/bare-metal/autounattend.xml +428 -0
  140. package/src/provisioning/bare-metal/serve-bootstrap.sh +86 -0
  141. package/src/provisioning/bootstrap-control-plane.sh +463 -0
  142. package/src/provisioning/bootstrap-windows-worker.ps1 +649 -0
  143. package/src/provisioning/build-lean-worker-image.ps1 +275 -0
  144. package/src/provisioning/diagnose-nvda-worker.ps1 +174 -0
  145. package/src/provisioning/provision-nvda-worker.ps1 +827 -0
  146. package/src/provisioning/set-display-mode.ps1 +411 -0
  147. package/src/provisioning/stamp-provision-revision.ps1 +184 -0
@@ -0,0 +1,442 @@
1
+ #!/usr/bin/env bash
2
+ # Start, pause, stop and check the local NVDA worker VM, so it is not burning resources
3
+ # while you are not capturing.
4
+ #
5
+ # npm run worker:ctl -- up # make it ready (start or resume), wait for /health
6
+ # ./packages/worker-fleet/src/local-worker/worker-ctl.sh pause # freeze it: ~0.6% CPU, instant resume, RAM not guaranteed
7
+ # ./packages/worker-fleet/src/local-worker/worker-ctl.sh stop # shut it down: nothing held, ~15 s to come back
8
+ # ./packages/worker-fleet/src/local-worker/worker-ctl.sh status # state, resource use, health
9
+ # ./packages/worker-fleet/src/local-worker/worker-ctl.sh json # the same, machine-readable (used by the CLI)
10
+ # ./packages/worker-fleet/src/local-worker/worker-ctl.sh pool # every a11y-worker* VM, as JSON
11
+ # ./packages/worker-fleet/src/local-worker/worker-ctl.sh pool-up # start them all, wait for health
12
+ # ./packages/worker-fleet/src/local-worker/worker-ctl.sh pool-stop # shut the whole pool down
13
+ # ./packages/worker-fleet/src/local-worker/worker-ctl.sh pool-pause # freeze the whole pool
14
+ #
15
+ # One VM serves one capture at a time, so throughput comes from more VMs. `pool` reports the
16
+ # lot; add one with clone-worker.sh (which handles the duplicate-MAC trap).
17
+ # Operate on a single named VM with A11Y_VM_NAME=a11y-worker-2.
18
+ # ./packages/worker-fleet/src/local-worker/worker-ctl.sh idle-pause 15 # watch, then pause after 15 idle minutes
19
+ # ./packages/worker-fleet/src/local-worker/worker-ctl.sh idle-stop 30 # same but shut down instead
20
+ #
21
+ # Measured on an M4 Max, 4 vCPU / 8 GB guest. Every number here was observed on this
22
+ # machine; none is an estimate:
23
+ #
24
+ # | state | host CPU | host RSS | back to /health |
25
+ # |--------------|-------------------|-----------------------|-----------------|
26
+ # | running idle | 2-86%, spiky | ~5 GB | - |
27
+ # | paused | ~0.6% | 0.8 GB *or* 4.5 GB | under 1 s |
28
+ # | stopped | none (no process) | none | 12-15 s, once 81 s |
29
+ #
30
+ # Read the caveats before trusting the table:
31
+ # - "running idle" is not idle. Windows keeps working in the background (Defender,
32
+ # Update, the search indexer), so a single sample means nothing: consecutive readings
33
+ # 20 s apart gave 1.8%, 25% and 86%. That is why `pause` is worth having at all.
34
+ # - `pause` reliably buys back CPU. It does NOT reliably give back memory. One paused run
35
+ # settled to ~0.8 GB within 40 s; the next held ~4.5 GB for three minutes straight with
36
+ # 66% of host memory free. Whether the host reclaims a suspended guest's pages is not
37
+ # ours to decide, so do not count on it.
38
+ #
39
+ # So: `pause` for a short gap between captures -- near-zero CPU and the guest never
40
+ # rebooted, so resume is instant. `stop` when you actually want the memory back, because it
41
+ # is the only one that guarantees it, and it is cheap: cold start reached /health in 12 s,
42
+ # 12 s and 15 s across three runs (`up` returns a few seconds later once it has the IP), and
43
+ # a capture immediately after a cold start was verified working, disclosure state change
44
+ # included. Do not read 12-15 s as a guarantee, though: a later run took 81 s, on a busier
45
+ # host with Windows doing its own post-boot work. `up` waits for /health rather than a fixed
46
+ # delay for exactly that reason. It comes back unattended because auto-logon plus the
47
+ # at-logon trigger restart the worker -- see docs/local-worker-vm.md.
48
+ set -euo pipefail
49
+
50
+ # architecture-audit.md §8: this manages a local UTM worker VM, which is deprecated -- "The UTM is
51
+ # deprecated, that was a testing thing." (repository owner, 2026-09-05). Capture on the bare-metal fleet
52
+ # instead: npm run fleet:status, npm run fleet:deploy. See CLAUDE.md's "Working on a Mac" section.
53
+ echo "DEPRECATED: worker-ctl.sh manages a local UTM worker VM. UTM was a testing path and is not the fleet." >&2
54
+ echo "Capture on the bare-metal fleet instead: npm run fleet:status, npm run fleet:deploy." >&2
55
+
56
+ # #636: a warning on a path nobody watches is a warning nobody reads -- so this REFUSES, rather than
57
+ # merely printing the two lines above and continuing. The measurements and reasoning behind the local-VM
58
+ # work stay in docs/local-worker-vm.md as history; only the path that runs is fenced.
59
+ if [ "${A11Y_LOCAL_VM:-}" != "1" ]; then
60
+ echo "refusing: set A11Y_LOCAL_VM=1 to run this deprecated local-VM script anyway." >&2
61
+ exit 1
62
+ fi
63
+
64
+ VM_NAME="${A11Y_VM_NAME:-a11y-worker}"
65
+ PORT="${A11Y_PORT:-8765}"
66
+
67
+ # Accept `--vm=<name>` as well as A11Y_VM_NAME, and accept it in ANY position.
68
+ #
69
+ # `worker:deploy` has always taken `--vm=`, so anyone who has used that reaches for it here too — and this
70
+ # script silently ignored it, then reported a DIFFERENT VM's state under the name you asked for. Silently,
71
+ # because a stray argument was simply never read. Two tools in one fleet disagreeing about how to name a
72
+ # machine is the kind of paper cut that gets diagnosed as "the guest is broken".
73
+ #
74
+ # Stripped from the positional arguments before CMD/ARG are taken, so `up --vm=x` and `--vm=x up` both work.
75
+ ARGS=()
76
+ for a in "$@"; do
77
+ case "$a" in
78
+ --vm=*) VM_NAME="${a#--vm=}" ;;
79
+ *) ARGS+=("$a") ;;
80
+ esac
81
+ done
82
+ set -- "${ARGS[@]+"${ARGS[@]}"}"
83
+
84
+ CMD="${1:-status}"
85
+ SHUTDOWN_GRACE_S=120 # how long to let Windows shut down cleanly before forcing
86
+ RECLAIM_SETTLE_S=45 # give the host a chance to reclaim pages before reporting usage
87
+ HEALTH_TIMEOUT_S=5 # per probe
88
+ HEALTH_TRIES=3 # before calling a worker unreachable
89
+ HEALTH_GAP_S=2 # between probes
90
+ ARG="${2:-}"
91
+
92
+ die() { echo "error: $*" >&2; exit 1; }
93
+ command -v utmctl >/dev/null || die "utmctl not found (brew install --cask utm)"
94
+
95
+ # utmctl is a client for the UTM APP, not a standalone daemon. With UTM not running it cannot
96
+ # answer, and the symptoms are misleading: a VM reports its state as `unknown` (or the command
97
+ # fails outright) even though the bundle is present and intact. Seen for real -- quitting UTM
98
+ # to edit its preferences made every utmctl call useless until the app was relaunched.
99
+ #
100
+ # So launch it and wait, rather than reporting a healthy VM as unknown.
101
+ ensure_utm_running() {
102
+ pgrep -x UTM >/dev/null && return
103
+ echo "UTM is not running (utmctl needs the app); launching it ..."
104
+ open -a UTM || die "could not launch UTM"
105
+ for _ in $(seq 1 15); do
106
+ sleep 2
107
+ utmctl list >/dev/null 2>&1 && { echo " UTM is up"; return; }
108
+ done
109
+ die "UTM did not become responsive. Open it once by hand and check it starts cleanly."
110
+ }
111
+ ensure_utm_running
112
+
113
+ # Resolve by UUID, never by name. Two registrations can share a name, and then `utmctl
114
+ # start <name>` silently picks the wrong one -- worse, `utmctl delete <name>` removes the
115
+ # shared bundle and takes the other VM's disk with it.
116
+ resolve_uuid() {
117
+ [ -n "${A11Y_VM_UUID:-}" ] && { echo "$A11Y_VM_UUID"; return; }
118
+ local matches
119
+ matches="$(utmctl list | awk -v n="$VM_NAME" '$3 == n { print $1 }')"
120
+ [ -n "$matches" ] || die "no VM named '$VM_NAME' (create one: packages/worker-fleet/src/local-worker/create-utm-vm.sh)"
121
+ if [ "$(echo "$matches" | wc -l | tr -d ' ')" -gt 1 ]; then
122
+ # Do NOT guess, and do NOT suggest deleting one. Duplicate registrations under the same
123
+ # name point at the SAME <name>.utm bundle, so `utmctl delete` on either removes that
124
+ # directory and destroys the other VM's disk and UEFI vars. That has already happened
125
+ # here once; the aftermath is a start that fails with
126
+ # 'The file "edk2-arm-vars.fd" doesn't exist'.
127
+ die "several VMs are registered as '$VM_NAME':
128
+ $matches
129
+ Pick one explicitly: A11Y_VM_UUID=<uuid> $0 $CMD
130
+
131
+ WARNING: do not 'utmctl delete' either of them. They share one bundle directory, so
132
+ deleting either destroys the other's disk. Resolve it by renaming one VM in the UTM UI
133
+ (which moves its bundle) before deleting anything."
134
+ fi
135
+ echo "$matches"
136
+ }
137
+
138
+ vm_state() { utmctl status "$1" 2>/dev/null || echo unknown; }
139
+
140
+ # `unknown` means utmctl could not answer, not that the VM is broken. Almost always UTM was
141
+ # not running (handled above) or the UUID is stale -- so say which, instead of leaving a
142
+ # caller to conclude the guest is dead.
143
+ explain_unknown() {
144
+ echo " state 'unknown' means utmctl could not answer for this VM, NOT that the guest is broken." >&2
145
+ echo " UTM app running: $(pgrep -x UTM >/dev/null && echo yes || echo NO)" >&2
146
+ echo " guest process: $(pgrep -f QEMULauncher >/dev/null && echo running || echo none)" >&2
147
+ echo " registered VMs:" >&2; utmctl list 2>&1 | sed 's/^/ /' >&2
148
+ }
149
+
150
+ guest_ip() {
151
+ utmctl ip-address "$1" 2>/dev/null \
152
+ | grep -oE '^[0-9]+\.[0-9]+\.[0-9]+\.[0-9]+' | grep -v '^127' | head -1
153
+ }
154
+
155
+ # One probe. Needs the guest agent to report an IP first, which only happens once the guest
156
+ # tools are installed.
157
+ health_once() {
158
+ local ip; ip="$(guest_ip "$1")"
159
+ [ -n "$ip" ] || return 1
160
+ curl -s -m "$HEALTH_TIMEOUT_S" "http://$ip:$PORT/health" 2>/dev/null
161
+ }
162
+
163
+ # A verdict, not a probe: retry before declaring a worker unreachable.
164
+ #
165
+ # One timed-out curl is not evidence of a dead worker, and treating it as such is how a
166
+ # healthy VM gets diagnosed as broken. The guest is busiest exactly when you most want to know
167
+ # it is alive -- Edge launching, and NVDA cold-starting every 25 captures -- and a worker
168
+ # restart takes /health down for 5-10s entirely legitimately.
169
+ #
170
+ # Measured during a live capture: 30/30 direct probes succeeded, none over a second, so this
171
+ # is not papering over a known flake. It is refusing to report a one-off as a fact.
172
+ health() {
173
+ local body
174
+ for attempt in $(seq 1 "$HEALTH_TRIES"); do
175
+ body="$(health_once "$1")" && [ -n "$body" ] && { echo "$body"; return 0; }
176
+ [ "$attempt" -lt "$HEALTH_TRIES" ] && sleep "$HEALTH_GAP_S"
177
+ done
178
+ return 1
179
+ }
180
+
181
+ # Waits for READY, not merely for an answer.
182
+ #
183
+ # This used to return the moment /health responded, which is precisely when the first capture
184
+ # failed: the port answers well before NVDA can start, so `up` reported success and the run
185
+ # immediately lost its first case to `nvda.start failed: Timed out waiting for NVDA to be
186
+ # running`. The worker now warms NVDA at boot and reports `ready:false` until it is answering,
187
+ # so waiting for that makes `up` mean what it says.
188
+ #
189
+ # A worker predating the field returns no `ready`, and is accepted as before.
190
+ wait_healthy() {
191
+ local uuid="$1" limit="${2:-180}" waited=0 body
192
+ while [ "$waited" -lt "$limit" ]; do
193
+ body="$(health_once "$uuid" || true)"
194
+ if [ -n "$body" ]; then
195
+ if ! echo "$body" | grep -q '"ready":false'; then
196
+ echo " ready after ${waited}s: $body"
197
+ return 0
198
+ fi
199
+ # Answering but still warming up. Say so, because silence here looks like a hang.
200
+ [ $((waited % 15)) -eq 0 ] && echo " answering, NVDA still warming up (${waited}s)"
201
+ fi
202
+ sleep 3; waited=$((waited + 3))
203
+ done
204
+ echo " NOT ready after ${limit}s" >&2
205
+ return 1
206
+ }
207
+
208
+ # This VM's own process, matched on its UUID in the qemu command line. Reporting whichever
209
+ # qemu process came first described a different worker as soon as there was a pool.
210
+ qemu_usage() {
211
+ local pid; pid="$(pgrep -f "uuid $UUID" | head -1 || true)"
212
+ if [ -z "$pid" ]; then echo "not running (RAM released)"; return; fi
213
+ ps -o pcpu=,rss= -p "$pid" | awk '{printf "cpu=%s%% rss=%.1fGB", $1, $2/1048576}'
214
+ }
215
+
216
+ UUID="$(resolve_uuid)"
217
+
218
+ case "$CMD" in
219
+ up)
220
+ state="$(vm_state "$UUID")"
221
+ case "$state" in
222
+ started) echo "already started";;
223
+ paused) echo "resuming from pause"; utmctl start "$UUID" >/dev/null;;
224
+ *) echo "cold starting (boot + auto-logon + worker task; waiting for /health)"; utmctl start "$UUID" >/dev/null;;
225
+ esac
226
+ if ! wait_healthy "$UUID"; then
227
+ [ "$state" = started ] && echo " it was already 'started', so it is either mid-shutdown or the worker task died: try '$0 stop && $0 up'" >&2
228
+ exit 1
229
+ fi
230
+ ip="$(guest_ip "$UUID")"
231
+ echo
232
+ echo " A11Y_WORKER=http://$ip:$PORT"
233
+ ;;
234
+
235
+ pause)
236
+ [ "$(vm_state "$UUID")" = "started" ] || { echo "not running (state: $(vm_state "$UUID"))"; exit 0; }
237
+ utmctl suspend "$UUID" >/dev/null
238
+ # Give the host a chance to reclaim the suspended guest's pages before reporting, but
239
+ # do not promise it will: observed both ~0.8 GB and ~4.5 GB while paused. Say which one
240
+ # happened, so a still-large footprint is visible rather than assumed away.
241
+ sleep "$RECLAIM_SETTLE_S"
242
+ echo "paused ($(qemu_usage)) -- resume with: $0 up (under a second; the guest never rebooted)"
243
+ echo "note: CPU is back either way; memory may or may not be. Use '$0 stop' to be sure of it."
244
+ ;;
245
+
246
+ stop)
247
+ state="$(vm_state "$UUID")"
248
+ [ "$state" = "stopped" ] && { echo "already stopped"; exit 0; }
249
+ # A paused VM cannot be shut down gracefully -- resume it first so Windows can flush
250
+ # and exit cleanly, rather than yanking the power from a frozen guest.
251
+ if [ "$state" = "paused" ]; then
252
+ echo "resuming briefly so the guest can shut down cleanly"
253
+ utmctl start "$UUID" >/dev/null; sleep 5
254
+ fi
255
+ # Ask Windows directly through the guest agent, rather than relying on ACPI.
256
+ #
257
+ # `utmctl stop --request` sends an ACPI power-button event, and this guest ignores it: every
258
+ # stop sat out the full 120s grace and then force-stopped, so every "clean" shutdown was
259
+ # actually a power cut. Setting AutoEndTasks and the kill timeouts did not help, which
260
+ # points at the guest's power-button action rather than at apps refusing to close.
261
+ #
262
+ # `shutdown /s /t 0` over the guest agent needs no network and no ACPI, and Windows flushes
263
+ # properly. ACPI stays as the fallback, and force as the last resort.
264
+ if utmctl exec "$UUID" --cmd shutdown.exe /s /t 0 >/dev/null 2>&1; then
265
+ echo " asked Windows to shut down (guest agent)"
266
+ else
267
+ echo " guest agent unavailable; falling back to ACPI"
268
+ utmctl stop "$UUID" --request >/dev/null 2>&1 || true
269
+ fi
270
+ # Wait on THIS VM's state, not on the absence of qemu processes.
271
+ #
272
+ # The old loop broke only when no QEMULauncher process existed anywhere, which is true with
273
+ # one VM and false with a pool: the other workers' processes kept it spinning for the full
274
+ # grace period, so every stop reported "guest ignored ACPI shutdown" and force-stopped a VM
275
+ # that had already shut down cleanly -- the giveaway being the force-stop then failing with
276
+ # "The virtual machine is not running". Two and a half minutes per stop, and every "clean"
277
+ # shutdown recorded as a power cut, all from a check that could not tell our VM from anyone
278
+ # else's.
279
+ # Poll THIS VM's own qemu process, by UUID. Two wrong turns got here:
280
+ #
281
+ # `pgrep -f QEMULauncher` breaks only when NO vm is running anywhere, so with a pool the
282
+ # other workers kept it spinning the full grace period and every
283
+ # clean shutdown was recorded as ignored, then force-stopped.
284
+ # `utmctl status` per-VM and correct, but slow enough that forty iterations took
285
+ # 464s of wall clock while `waited` only counted the sleeps.
286
+ #
287
+ # pgrep on the UUID is both: specific to this VM, and cheap enough to poll.
288
+ waited=0
289
+ while [ "$waited" -lt "$SHUTDOWN_GRACE_S" ]; do
290
+ pgrep -f "uuid $UUID" >/dev/null || break
291
+ sleep 3; waited=$((waited + 3))
292
+ done
293
+ if pgrep -f "uuid $UUID" >/dev/null; then
294
+ echo " guest ignored ACPI shutdown after ${waited}s -- forcing"
295
+ utmctl stop "$UUID" >/dev/null
296
+ for _ in $(seq 1 10); do pgrep -f "uuid $UUID" >/dev/null || break; sleep 3; done
297
+ fi
298
+ echo "stopped after ${waited}s ($(qemu_usage))"
299
+ ;;
300
+
301
+ status)
302
+ echo "vm: $VM_NAME ($UUID)"
303
+ state="$(vm_state "$UUID")"
304
+ echo "state: $state"
305
+ [ "$state" = unknown ] && explain_unknown
306
+ echo "host: $(qemu_usage)"
307
+ ip="$(guest_ip "$UUID" || true)"
308
+ echo "guest: ${ip:-no ip (agent not reporting)}"
309
+ body="$(health "$UUID" || true)"
310
+ if [ -n "$body" ]; then
311
+ echo "health: $body"
312
+ echo "$body" | grep -q '"busy":true' && echo " (busy is NORMAL: one capture at a time by design)"
313
+ else
314
+ echo "health: unreachable after $HEALTH_TRIES probes"
315
+ if [ -n "${ip:-}" ]; then
316
+ echo " the guest is up and has an IP, so this is the WORKER, not the VM." >&2
317
+ echo " it is down for 5-10s during a restart; if it persists:" >&2
318
+ echo " utmctl exec <uuid> --cmd powershell.exe -NoProfile -Command 'Start-ScheduledTask -TaskName a11ysrv'" >&2
319
+ else
320
+ # `$0` is this file's path, which after M6 is `packages/worker-fleet/src/local-worker/worker-ctl.sh` — true,
321
+ # and not what anyone wants to type. The npm alias is the stable way to say it, and it is what the docs use.
322
+ echo " no guest IP either, so the VM itself is not ready. Try 'npm run worker:ctl -- up'." >&2
323
+ fi
324
+ fi
325
+ ;;
326
+
327
+ json)
328
+ # One line of JSON for the control plane (src/capture/local-vm.ts). Keeping every UTM
329
+ # detail behind this command means the TS side never parses human-readable output and
330
+ # never learns about utmctl, bundles or bookmarks.
331
+ ip="$(guest_ip "$UUID" || true)"
332
+ body="$(health "$UUID" || true)"
333
+ healthy=false; [ -n "$body" ] && healthy=true
334
+ busy=false; if echo "$body" | grep -q '"busy":true'; then busy=true; fi
335
+ # A worker can answer /health while NVDA cannot start. `ready:false` says so explicitly;
336
+ # a worker predating the field reports nothing, and is treated as ready.
337
+ # `ready` is only meaningful when something ANSWERED. This defaulted to true and was lowered only if
338
+ # the body said `"ready":false` — so a stopped or unreachable VM, whose body is empty, reported
339
+ # `ready: true`. That is "we could not ask" rendered as "yes", on the one field CLAUDE.md says to
340
+ # dispatch on: a run reading this JSON could pick a stopped guest.
341
+ ready=false
342
+ if [ "$healthy" = true ] && ! echo "$body" | grep -q '"ready":false'; then ready=true; fi
343
+ state="$(vm_state "$UUID")"
344
+
345
+ # `means` carries no information the other fields lack; it exists because they were being
346
+ # read wrong. `healthy:false` says "not answering right now", but three of those in a row
347
+ # looks like a broken pool -- and since a run starts its own workers, stopped is the normal
348
+ # resting state. An agent read exactly this output, concluded the environment was down, and
349
+ # went hunting for a worker that had been decommissioned. So the JSON now says what it means.
350
+ if [ "$healthy" = true ] && [ "$ready" = false ]; then
351
+ means="answering but NOT ready -- NVDA still warming up, or it failed to start"
352
+ elif [ "$healthy" = true ]; then
353
+ means="ready"; [ "$busy" = true ] && means="ready, busy with a capture"
354
+ elif [ "$state" = "started" ]; then
355
+ means="running but not answering /health -- this one IS a fault"
356
+ else
357
+ means="$state -- normal at rest; a run starts it and stops it again"
358
+ fi
359
+ printf '{"uuid":"%s","name":"%s","state":"%s","ip":"%s","port":%s,"healthy":%s,"ready":%s,"busy":%s,"means":"%s"}\n' \
360
+ "$UUID" "$VM_NAME" "$state" "${ip:-}" "$PORT" "$healthy" "$ready" "$busy" "$means"
361
+ ;;
362
+
363
+ pool-stop|pool-pause)
364
+ # Stop or pause every worker in one go, for when a run has finished and you want the
365
+ # resources back. `idle-stop`/`idle-pause` do this automatically for a single VM; this is
366
+ # the manual, whole-pool version.
367
+ action="${CMD#pool-}"
368
+ for n in $(utmctl list | awk -v n="$VM_NAME" '$3 ~ "^"n { print $3 }' | sort -u); do
369
+ echo "--- $n ---"
370
+ A11Y_VM_NAME="$n" "$0" "$action" || echo " '$n' did not $action"
371
+ done
372
+ ;;
373
+
374
+ pool|pool-up)
375
+ # Every VM whose name starts with the base name: a11y-worker, a11y-worker-2, ...
376
+ # Emitted as one JSON array so a dispatcher can consume it without parsing prose.
377
+ names="$(utmctl list | awk -v n="$VM_NAME" '$3 ~ "^"n { print $3 }' | sort -u)"
378
+ [ -n "$names" ] || die "no VM whose name starts with '$VM_NAME'"
379
+ if [ "$CMD" = "pool-up" ]; then
380
+ # Sequentially, not in parallel: two Windows guests booting at once contend badly for
381
+ # disk, and one that is already up costs nothing to skip.
382
+ for n in $names; do
383
+ echo "--- $n ---" >&2
384
+ A11Y_VM_NAME="$n" "$0" up >&2 || echo " '$n' did not come up" >&2
385
+ done
386
+ fi
387
+ printf '['
388
+ first=1
389
+ for n in $names; do
390
+ entry="$(A11Y_VM_NAME="$n" "$0" json 2>/dev/null || true)"
391
+ [ -n "$entry" ] || continue
392
+ [ "$first" -eq 1 ] || printf ','
393
+ printf '%s' "$entry"
394
+ first=0
395
+ done
396
+ printf ']\n'
397
+ ;;
398
+
399
+ idle-pause|idle-stop)
400
+ mins="${ARG:-15}"
401
+ action=pause; [ "$CMD" = "idle-stop" ] && action=stop
402
+ echo "watching $VM_NAME; will $action after $mins idle minutes (Ctrl-C to stop watching)"
403
+ idle=0
404
+ # #635: `health()` already retries internally, but its budget is tuned for "is the box up right
405
+ # now" -- a few tries over seconds -- not for an hours-long watch. One round of THAT budget
406
+ # exhausting used to end the whole watch on the spot, so a single transient blip (this fleet has
407
+ # measured EHOSTUNREACH for 48 straight requests before a worker recovered) silently abandoned
408
+ # hours of monitoring. Require several consecutive OUTER-LOOP misses, at this loop's own cadence,
409
+ # before concluding the worker is actually gone.
410
+ misses=0
411
+ max_misses=5
412
+ while :; do
413
+ body="$(health "$UUID" || true)"
414
+ if [ -z "$body" ]; then
415
+ misses=$((misses + 1))
416
+ echo "worker unreachable (state: $(vm_state "$UUID"), miss $misses/$max_misses)"
417
+ if [ "$misses" -ge "$max_misses" ]; then
418
+ echo "worker unreachable for $max_misses consecutive checks -- giving up the watch"
419
+ exit 0
420
+ fi
421
+ sleep 60
422
+ continue
423
+ fi
424
+ misses=0
425
+ # `busy` is true only while a capture is in flight, so it is the honest activity
426
+ # signal. Any capture resets the clock.
427
+ if echo "$body" | grep -q '"busy":true'; then
428
+ [ "$idle" -gt 0 ] && echo " capture in flight, resetting idle clock"
429
+ idle=0
430
+ else
431
+ idle=$((idle + 1))
432
+ fi
433
+ if [ "$idle" -ge "$mins" ]; then
434
+ echo "idle $mins min -> $action"
435
+ exec "$0" "$action"
436
+ fi
437
+ sleep 60
438
+ done
439
+ ;;
440
+
441
+ *) die "unknown command '$CMD' (up | pause | stop | status | json | pool | pool-up | pool-stop | pool-pause | idle-pause [min] | idle-stop [min])";;
442
+ esac
@@ -0,0 +1,28 @@
1
+ # Provisioning a capture machine — who each path is for
2
+
3
+ This directory has provisioning for the **fleet** (machines this project already owns, with
4
+ infrastructure this project already runs) and provisioning for a **bare Windows machine someone else
5
+ owns**. They look similar — both end with a box answering `/health` — and are easy to mistake for the
6
+ same problem. They are not: only one of the paths below is something an outside contributor can follow
7
+ start to finish.
8
+
9
+ | here | who it is for | what it assumes |
10
+ |---|---|---|
11
+ | `bare-metal/` (PXE, zero-touch) | this project's own fleet operator | Proxmox CT 110 `iventoy-pxe`, the fleet-control container serving the bootstrap payload, and `inventory.yml`/`ansible.cfg` wiring — infrastructure that lives outside this checkout and that only this project runs |
12
+ | `provision-nvda-worker.ps1` | this project's own fleet operator, repairing or re-provisioning a machine that is *already reachable* (an enrolled fleet worker, or a box `bootstrap-windows-worker.ps1` below has already prepared) | an existing checkout on the target **and** SSH access already set up — its usual invocation is `scp` + `ssh` from a machine that already has both |
13
+ | `bootstrap-windows-worker.ps1` | **an outside contributor with a spare Windows machine** | nothing but that machine. Self-contained: one elevated PowerShell line (`irm ... \| iex`), no prior checkout, no existing SSH access. It installs prerequisites, makes the box reachable, then hands off to `provision-nvda-worker.ps1` above for the NVDA/OS steps. Documented as **Route B** ("you already have a Windows machine") in [`docs/getting-started.md`](../../../../docs/getting-started.md). Its own header says it has not yet been run end-to-end on a fresh install — expect to babysit the first run. |
14
+ | the GitHub Action (`.github/workflows/capture-regression.yml`) | **an outside contributor with no machine to spare at all** | nothing — real NVDA runs on a GitHub-hosted Windows runner. Documented as **Route C** in `docs/getting-started.md`. |
15
+
16
+ The Ansible role (`packages/control/ansible/provision-role.yml`) is a second, in-progress way to do what
17
+ `provision-nvda-worker.ps1` does — same audience, same assumptions, coexisting with the script pending a
18
+ parity check its own header spells out. That is a fleet-internal question about which of two tools this
19
+ project uses on its own boxes; it does not change any row in the table above.
20
+
21
+ ## Can an outside stranger provision a capture machine?
22
+
23
+ Not through `bare-metal/` or through running `provision-nvda-worker.ps1` on its own — both assume
24
+ infrastructure or access a stranger does not have, and naming that plainly is this file's job. The
25
+ supported starting points for someone with nothing already set up are `bootstrap-windows-worker.ps1`
26
+ (their own Windows machine) and the GitHub Action (no machine at all), both linked above. Self-hosting a
27
+ *fleet-joined* worker the zero-touch way is not something this project supports for an outside
28
+ contributor yet.
@@ -0,0 +1,71 @@
1
+ # Apply ForegroundLockTimeout = 0 to the CURRENT session, immediately.
2
+ #
3
+ # Why this exists: writing HKCU\Control Panel\Desktop\ForegroundLockTimeout is not enough.
4
+ # The value is cached per session, so a registry write does nothing until the next logon --
5
+ # and worse, Windows does not reliably consume that value at logon either, so even a reboot
6
+ # is not a guarantee. The supported way to change it live is SystemParametersInfo with
7
+ # SPI_SETFOREGROUNDLOCKTIMEOUT, which is what this does.
8
+ #
9
+ # Why it matters: with a non-zero timeout, Windows refuses to let Edge be forced into the
10
+ # foreground, so NVDA has nothing to read. The capture then returns 0 phrases with NO
11
+ # error anywhere -- nvda.start() succeeds, windowsActivate reports ok, and every read comes
12
+ # back empty. It is the single most misleading failure in the pipeline, and
13
+ # packages/nvda-worker/README.md calls it the #1 flakiness fix.
14
+ #
15
+ # MUST run in the interactive desktop session. Per the API docs, "the calling thread must
16
+ # be able to change the foreground window, otherwise the call fails" -- so running this as
17
+ # SYSTEM (e.g. via a guest agent, or a scheduled task without LogonType Interactive) will
18
+ # fail even though the registry write appears to succeed.
19
+ #
20
+ # Because the setting does not reliably survive a logon, this is called from BOTH
21
+ # provision-nvda-worker.ps1 (so a freshly provisioned box can capture without a reboot)
22
+ # and run-server.cmd (so every worker start re-applies it for that session).
23
+ #
24
+ # Style note: `#` line comments and no param() block, matching the other scripts here --
25
+ # see packages/worker-fleet/src/provisioning/diagnose-nvda-worker.ps1 for why.
26
+
27
+ $ErrorActionPreference = 'Stop'
28
+
29
+ if (-not ('A11y.Spi' -as [type])) {
30
+ Add-Type -Namespace 'A11y' -Name 'Spi' -MemberDefinition @'
31
+ // For SPI_SETFOREGROUNDLOCKTIMEOUT the new value is passed AS pvParam (the DWORD cast
32
+ // to a pointer-sized value), not as a pointer to it. Getting this backwards silently
33
+ // sets a garbage timeout.
34
+ [DllImport("user32.dll", SetLastError = true)]
35
+ public static extern bool SystemParametersInfo(uint uiAction, uint uiParam, UIntPtr pvParam, uint fWinIni);
36
+
37
+ // For the GET action pvParam DOES point at a DWORD that receives the value.
38
+ [DllImport("user32.dll", SetLastError = true, EntryPoint = "SystemParametersInfoW")]
39
+ public static extern bool SystemParametersInfoGet(uint uiAction, uint uiParam, ref uint pvParam, uint fWinIni);
40
+ '@
41
+ }
42
+
43
+ $SPI_GETFOREGROUNDLOCKTIMEOUT = 0x2000
44
+ $SPI_SETFOREGROUNDLOCKTIMEOUT = 0x2001
45
+ $SPIF_UPDATEINIFILE = 0x01 # persist into the user profile
46
+ $SPIF_SENDCHANGE = 0x02 # broadcast WM_SETTINGCHANGE
47
+
48
+ $before = 0
49
+ [void][A11y.Spi]::SystemParametersInfoGet($SPI_GETFOREGROUNDLOCKTIMEOUT, 0, [ref] $before, 0)
50
+
51
+ $ok = [A11y.Spi]::SystemParametersInfo(
52
+ $SPI_SETFOREGROUNDLOCKTIMEOUT, 0, [UIntPtr]::Zero,
53
+ ($SPIF_UPDATEINIFILE -bor $SPIF_SENDCHANGE))
54
+ $err = [Runtime.InteropServices.Marshal]::GetLastWin32Error()
55
+
56
+ $after = 0
57
+ [void][A11y.Spi]::SystemParametersInfoGet($SPI_GETFOREGROUNDLOCKTIMEOUT, 0, [ref] $after, 0)
58
+
59
+ # Belt and braces: also write the registry value so it is at least declared for future
60
+ # sessions. The API call above is what makes THIS session work.
61
+ Set-ItemProperty 'HKCU:\Control Panel\Desktop' -Name ForegroundLockTimeout -Value 0 -Type DWord -ErrorAction SilentlyContinue
62
+
63
+ if ($after -eq 0) {
64
+ Write-Output "ForegroundLockTimeout: $before -> $after (applied to this session)"
65
+ exit 0
66
+ }
67
+
68
+ Write-Output "ForegroundLockTimeout: still $after (SystemParametersInfo ok=$ok lastError=$err)"
69
+ Write-Output 'Most likely cause: not running in the interactive desktop session -- the API'
70
+ Write-Output 'requires a thread that is allowed to change the foreground window.'
71
+ exit 1