@a11ign/screenreader-fleet 0.1.0 → 0.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (129) hide show
  1. package/dist/capture-client.d.mts +0 -1
  2. package/dist/capture-client.mjs +144 -306
  3. package/dist/check-worker-code.d.mts +0 -1
  4. package/dist/check-worker-code.mjs +42 -141
  5. package/dist/cli-flags.d.mts +0 -1
  6. package/dist/cli-flags.mjs +33 -179
  7. package/dist/code-drift.d.mts +0 -1
  8. package/dist/command-line-census.d.mts +0 -1
  9. package/dist/compare-workers.d.mts +0 -1
  10. package/dist/compare-workers.mjs +383 -255
  11. package/dist/control-plane-isolation.d.mts +0 -1
  12. package/dist/deploy-worker.d.mts +0 -1
  13. package/dist/deploy-worker.mjs +101 -242
  14. package/dist/doctor.d.mts +0 -1
  15. package/dist/doctor.mjs +361 -809
  16. package/dist/fleet-consistency.d.mts +0 -1
  17. package/dist/fleet-consistency.mjs +155 -379
  18. package/dist/fleet-env.d.mts +0 -1
  19. package/dist/fleet-env.mjs +148 -438
  20. package/dist/fleet-scripts.d.mts +0 -1
  21. package/dist/git-safe-env.d.mts +0 -1
  22. package/dist/guest-run.d.mts +0 -1
  23. package/dist/host-address.d.mts +0 -1
  24. package/dist/host-address.mjs +19 -90
  25. package/dist/host-capacity.d.mts +0 -1
  26. package/dist/host-capacity.mjs +22 -136
  27. package/dist/host-metrics.d.mts +0 -1
  28. package/dist/index.d.ts +0 -1
  29. package/dist/index.mjs +231 -0
  30. package/dist/local-vm.d.ts +0 -1
  31. package/dist/measure-guard.d.mts +0 -1
  32. package/dist/normalise-fleet.d.mts +0 -1
  33. package/dist/npm-cli-executable.d.mts +0 -1
  34. package/dist/probe-outcome.d.mts +0 -1
  35. package/dist/probe-outcome.mjs +50 -96
  36. package/dist/protocol-guard.d.mts +0 -1
  37. package/dist/source-walk.d.mts +0 -1
  38. package/dist/src_fleet-scripts_mjs.mjs +16 -0
  39. package/dist/src_git-safe-env_mjs.mjs +9 -0
  40. package/dist/src_utm-deprecated_mjs.mjs +4 -0
  41. package/dist/transient-fault.d.mts +0 -1
  42. package/dist/transient-fault.mjs +21 -81
  43. package/dist/utm-deprecated.d.mts +0 -1
  44. package/dist/worker-code-check.d.mts +0 -1
  45. package/dist/worker-code-check.mjs +127 -76
  46. package/dist/worker-health.d.mts +0 -1
  47. package/dist/worker-health.mjs +16 -61
  48. package/dist/worker-http.d.mts +0 -1
  49. package/dist/worker-http.mjs +42 -234
  50. package/dist/worker-stats.d.mts +0 -1
  51. package/package.json +12 -5
  52. package/dist/capture-client.d.mts.map +0 -1
  53. package/dist/capture-client.mjs.map +0 -1
  54. package/dist/check-worker-code.d.mts.map +0 -1
  55. package/dist/check-worker-code.mjs.map +0 -1
  56. package/dist/cli-flags.d.mts.map +0 -1
  57. package/dist/cli-flags.mjs.map +0 -1
  58. package/dist/code-drift.d.mts.map +0 -1
  59. package/dist/code-drift.mjs +0 -284
  60. package/dist/code-drift.mjs.map +0 -1
  61. package/dist/command-line-census.d.mts.map +0 -1
  62. package/dist/command-line-census.mjs +0 -96
  63. package/dist/command-line-census.mjs.map +0 -1
  64. package/dist/compare-workers.d.mts.map +0 -1
  65. package/dist/compare-workers.mjs.map +0 -1
  66. package/dist/control-plane-isolation.d.mts.map +0 -1
  67. package/dist/control-plane-isolation.mjs +0 -67
  68. package/dist/control-plane-isolation.mjs.map +0 -1
  69. package/dist/deploy-worker.d.mts.map +0 -1
  70. package/dist/deploy-worker.mjs.map +0 -1
  71. package/dist/doctor.d.mts.map +0 -1
  72. package/dist/doctor.mjs.map +0 -1
  73. package/dist/fleet-consistency.d.mts.map +0 -1
  74. package/dist/fleet-consistency.mjs.map +0 -1
  75. package/dist/fleet-env.d.mts.map +0 -1
  76. package/dist/fleet-env.mjs.map +0 -1
  77. package/dist/fleet-scripts.d.mts.map +0 -1
  78. package/dist/fleet-scripts.mjs +0 -41
  79. package/dist/fleet-scripts.mjs.map +0 -1
  80. package/dist/git-safe-env.d.mts.map +0 -1
  81. package/dist/git-safe-env.mjs +0 -44
  82. package/dist/git-safe-env.mjs.map +0 -1
  83. package/dist/guest-run.d.mts.map +0 -1
  84. package/dist/guest-run.mjs +0 -164
  85. package/dist/guest-run.mjs.map +0 -1
  86. package/dist/host-address.d.mts.map +0 -1
  87. package/dist/host-address.mjs.map +0 -1
  88. package/dist/host-capacity.d.mts.map +0 -1
  89. package/dist/host-capacity.mjs.map +0 -1
  90. package/dist/host-metrics.d.mts.map +0 -1
  91. package/dist/host-metrics.mjs +0 -201
  92. package/dist/host-metrics.mjs.map +0 -1
  93. package/dist/index.d.ts.map +0 -1
  94. package/dist/index.js +0 -25
  95. package/dist/index.js.map +0 -1
  96. package/dist/local-vm.d.ts.map +0 -1
  97. package/dist/local-vm.js +0 -360
  98. package/dist/local-vm.js.map +0 -1
  99. package/dist/measure-guard.d.mts.map +0 -1
  100. package/dist/measure-guard.mjs +0 -73
  101. package/dist/measure-guard.mjs.map +0 -1
  102. package/dist/normalise-fleet.d.mts.map +0 -1
  103. package/dist/normalise-fleet.mjs +0 -76
  104. package/dist/normalise-fleet.mjs.map +0 -1
  105. package/dist/npm-cli-executable.d.mts.map +0 -1
  106. package/dist/npm-cli-executable.mjs +0 -159
  107. package/dist/npm-cli-executable.mjs.map +0 -1
  108. package/dist/probe-outcome.d.mts.map +0 -1
  109. package/dist/probe-outcome.mjs.map +0 -1
  110. package/dist/protocol-guard.d.mts.map +0 -1
  111. package/dist/protocol-guard.mjs +0 -121
  112. package/dist/protocol-guard.mjs.map +0 -1
  113. package/dist/source-walk.d.mts.map +0 -1
  114. package/dist/source-walk.mjs +0 -56
  115. package/dist/source-walk.mjs.map +0 -1
  116. package/dist/transient-fault.d.mts.map +0 -1
  117. package/dist/transient-fault.mjs.map +0 -1
  118. package/dist/utm-deprecated.d.mts.map +0 -1
  119. package/dist/utm-deprecated.mjs +0 -23
  120. package/dist/utm-deprecated.mjs.map +0 -1
  121. package/dist/worker-code-check.d.mts.map +0 -1
  122. package/dist/worker-code-check.mjs.map +0 -1
  123. package/dist/worker-health.d.mts.map +0 -1
  124. package/dist/worker-health.mjs.map +0 -1
  125. package/dist/worker-http.d.mts.map +0 -1
  126. package/dist/worker-http.mjs.map +0 -1
  127. package/dist/worker-stats.d.mts.map +0 -1
  128. package/dist/worker-stats.mjs +0 -143
  129. package/dist/worker-stats.mjs.map +0 -1
@@ -1,104 +1,58 @@
1
- // @ts-check
2
- /**
3
- * WHAT ONE `/health` PROBE CAN SAY, for the entries that reach a worker from the PUBLISHED side and by hand
4
- * (`witness`, `worker:compare`, `auth:leak-check`, #2683 of #2655).
5
- *
6
- * Imports `worker-http` and nothing else: no `control` (this package is published and `@a11ign/control` never
7
- * is, `worker-fleet-does-not-read-control.test.ts`) and no corpus reader, so a test may import it without
8
- * pulling the corpus closure in. These entries do NOT wake a box (ADR 0012, product-manager 2026-09-26): they
9
- * say it did not answer, and name the command that wakes one. `packages/control/src/fleet-wake.mjs` keeps its
10
- * own `probeWorker` with the same five outcomes because it is the unpublished side of that line.
11
- *
12
- * The outcomes, and the one distinction that matters (an unanswered probe is UNKNOWN, never "down"):
13
- *
14
- * ready the box's own report, `ready: true`
15
- * busy the box's own report, `busy: true`: a capture is running. Up, and not free
16
- * not-ready it answered and its own `ready:false` says why, or it answered a non-2xx status. UP
17
- * refused the connection was refused, so something answered the TCP handshake with a reset: the BOX IS UP
18
- * and the worker is not listening. UP
19
- * no-answer nothing came back inside the timeout (`timedOut`), or the transport failed some other way
20
- * (unreachable, reset). One silent probe cannot separate "off" from "slow" from "the path dropped
21
- * it", so it is never called down
22
- *
23
- * `health` is the body the box sent, on the three outcomes where one arrived, because "usable" is a different
24
- * question in different entries (`workerIsUsable` counts a worker predating the `ready` field as usable; the
25
- * measurement guard wants `ready: true`) and this module does not choose for them.
26
- *
27
- * @typedef {{ outcome: "ready", health: any } | { outcome: "busy", health: any }
28
- * | { outcome: "not-ready", reason: string, health: any }
29
- * | { outcome: "refused", message: string }
30
- * | { outcome: "no-answer", message: string, timedOut: boolean }} Probe
31
- */
32
1
  import { requestJson } from "./worker-http.mjs";
33
- /** @typedef {typeof requestJson} ProbeRequest */
34
- /**
35
- * THE PER-PROBE TIMEOUT, WITH ITS READING. A probe that outlives it is UNKNOWN, never "down", so this number
36
- * decides how much slowness a healthy box is allowed.
37
- *
38
- * All of it READ by others on the real fleet (`orchestrator`, #2671) and none measured by this row, whose
39
- * engineer is barred from probing it:
40
- * - the slowest HEALTHY box: 2.80 to 3.09 s on a11y-worker-13, -14 and -16, twelve others 0.53 to 0.76 s;
41
- * - that is the FIRST answer after the box has been quiet more than 5 s, because the worker rebuilds its
42
- * environment block with two synchronous `powershell.exe` calls when it is older than that, so a probe made
43
- * by hand is nearly always the slow case;
44
- * - a LOADED box (one that has just stopped a capture) can take up to about 10 s: each of the two calls is
45
- * bounded at 5 s and they stop the worker's event loop for the whole time.
46
- * 12 s is that loaded ceiling plus 2 s, the number `fleet-wake.mjs` `HEALTH_TIMEOUT_MS` states for the same
47
- * reading. Before #2683 these entries used 5 s (`witness`, 1.91 s over the 3.09 s box and none over a loaded
48
- * one), 8 s (`auth:leak-check`) and 10 s (`worker:compare`'s busy guard), read at `d119fb0f2`. Cost of the
49
- * generosity: a box that really is off costs one 12 s wait before the message. A refusal costs nothing, it comes
50
- * back at once.
51
- */
52
- export const WORKER_PROBE_TIMEOUT_MS = 12_000;
53
- /**
54
- * @param {string} worker the worker's base URL
55
- * @param {{ timeoutMs?: number, request?: ProbeRequest }} [options]
56
- * @returns {Promise<Probe>}
57
- */
58
- export async function probeHealth(worker, { timeoutMs = WORKER_PROBE_TIMEOUT_MS, request = requestJson } = {}) {
2
+ const WORKER_PROBE_TIMEOUT_MS = 12000;
3
+ async function probeHealth(worker, { timeoutMs = WORKER_PROBE_TIMEOUT_MS, request = requestJson } = {}) {
59
4
  let response;
60
5
  try {
61
- response = await request(`${worker.replace(/\/$/, "")}/health`, { timeoutMs });
62
- }
63
- catch (error) {
64
- const { code, message } = /** @type {NodeJS.ErrnoException} */ (error);
65
- // `message` is EMPTY for a raw ECONNREFUSED on this Node version, so it falls back to the code.
66
- if (code === "ECONNREFUSED")
67
- return { outcome: "refused", message: message || code };
68
- return { outcome: "no-answer", timedOut: code === "ETIMEDOUT",
69
- message: `${code ? `${code}: ` : ""}${message}` };
6
+ response = await request(`${worker.replace(/\/$/, "")}/health`, {
7
+ timeoutMs
8
+ });
9
+ } catch (error) {
10
+ const { code, message } = error;
11
+ if ("ECONNREFUSED" === code) return {
12
+ outcome: "refused",
13
+ message: message || code
14
+ };
15
+ return {
16
+ outcome: "no-answer",
17
+ timedOut: "ETIMEDOUT" === code,
18
+ message: `${code ? `${code}: ` : ""}${message}`
19
+ };
70
20
  }
71
21
  const health = response.json;
72
- if (!response.ok)
73
- return { outcome: "not-ready", reason: `/health answered HTTP ${response.status}`, health };
74
- if (health?.busy === true)
75
- return { outcome: "busy", health };
76
- if (health?.ready === true)
77
- return { outcome: "ready", health };
78
- return { outcome: "not-ready", health,
79
- reason: response.json?.reason ?? "/health answered without `ready: true` and without a reason" };
22
+ if (!response.ok) return {
23
+ outcome: "not-ready",
24
+ reason: `/health answered HTTP ${response.status}`,
25
+ health
26
+ };
27
+ if (health?.busy === true) return {
28
+ outcome: "busy",
29
+ health
30
+ };
31
+ if (health?.ready === true) return {
32
+ outcome: "ready",
33
+ health
34
+ };
35
+ return {
36
+ outcome: "not-ready",
37
+ health,
38
+ reason: response.json?.reason ?? "/health answered without `ready: true` and without a reason"
39
+ };
80
40
  }
81
- /** How to wake a box from a checkout of this repo. The published entries do not do it themselves. */
82
- export const WAKE_HINT = "If it is a fleet box that has gone to sleep, wake it from a checkout of the a11ign repo with "
83
- + "`npm run fleet:wake -- <name>` (<name> is its entry in inventory.yml).";
84
- /**
85
- * One sentence per outcome, in the words of what was OBSERVED. Only `refused`, `not-ready` and `busy` say the box is
86
- * up, because only those are things the box itself said; a silent probe says "did not answer" and never "down".
87
- *
88
- * @param {Probe} probe
89
- * @param {{ worker: string, timeoutMs?: number }} about
90
- * @returns {string}
91
- */
92
- export function describeProbe(probe, { worker, timeoutMs = WORKER_PROBE_TIMEOUT_MS }) {
93
- switch (probe.outcome) {
94
- case "ready": return `${worker} answered and is ready.`;
95
- case "busy": return `${worker} is up and busy with a capture.`;
96
- case "not-ready": return `${worker} is up and answered, but it says it is not ready: ${probe.reason}.`;
97
- case "refused": return `${worker} refused the connection (${probe.message}): the machine is up and the worker `
98
- + "is not listening. Waking it will not help; start the worker on it.";
99
- case "no-answer": return `${worker} did not answer${probe.timedOut ? ` within ${timeoutMs / 1000} s` : ""} `
100
- + `(${probe.message}). That does not say it is off: it may be asleep, slow, or not reachable from here. ${WAKE_HINT}`;
101
- default: throw new Error(`unknown probe outcome: ${JSON.stringify(probe)}`);
41
+ const WAKE_HINT = "If it is a fleet box that has gone to sleep, wake it from a checkout of the a11ign repo with `npm run fleet:wake -- <name>` (<name> is its entry in inventory.yml).";
42
+ function describeProbe(probe, { worker, timeoutMs = WORKER_PROBE_TIMEOUT_MS }) {
43
+ switch(probe.outcome){
44
+ case "ready":
45
+ return `${worker} answered and is ready.`;
46
+ case "busy":
47
+ return `${worker} is up and busy with a capture.`;
48
+ case "not-ready":
49
+ return `${worker} is up and answered, but it says it is not ready: ${probe.reason}.`;
50
+ case "refused":
51
+ return `${worker} refused the connection (${probe.message}): the machine is up and the worker is not listening. Waking it will not help; start the worker on it.`;
52
+ case "no-answer":
53
+ return `${worker} did not answer${probe.timedOut ? ` within ${timeoutMs / 1000} s` : ""} (${probe.message}). That does not say it is off: it may be asleep, slow, or not reachable from here. ${WAKE_HINT}`;
54
+ default:
55
+ throw new Error(`unknown probe outcome: ${JSON.stringify(probe)}`);
102
56
  }
103
57
  }
104
- //# sourceMappingURL=probe-outcome.mjs.map
58
+ export { WAKE_HINT, WORKER_PROBE_TIMEOUT_MS, describeProbe, probeHealth };
@@ -31,4 +31,3 @@ export function servedProtocols(urls: string[]): Promise<{
31
31
  protocol: number | string | null;
32
32
  }[]>;
33
33
  export const RECAPTURE_COST: string;
34
- //# sourceMappingURL=protocol-guard.d.mts.map
@@ -9,4 +9,3 @@ export function sourceFiles({ root }?: {
9
9
  }): Array<[string, string]>;
10
10
  /** The `packages/` directory, resolved from this module rather than from the caller's cwd. */
11
11
  export const PACKAGES: string;
12
- //# sourceMappingURL=source-walk.d.mts.map
@@ -0,0 +1,16 @@
1
+ import { fileURLToPath } from "node:url";
2
+ import { join } from "node:path";
3
+ const assetDir = ()=>fileURLToPath(new URL("../src/local-worker/", import.meta.url));
4
+ function fleetScriptPaths() {
5
+ const dir = assetDir();
6
+ return {
7
+ dir,
8
+ workerCtl: join(dir, "worker-ctl.sh"),
9
+ buildVm: join(dir, "build-vm.sh"),
10
+ cloneWorker: join(dir, "clone-worker.sh"),
11
+ createUtmVm: join(dir, "create-utm-vm.sh"),
12
+ fetchWindowsIso: join(dir, "fetch-windows-iso.sh"),
13
+ provisioning: fileURLToPath(new URL("../src/provisioning/", import.meta.url))
14
+ };
15
+ }
16
+ export { fleetScriptPaths };
@@ -0,0 +1,9 @@
1
+ function sandboxGitEnv(extra = {}) {
2
+ const scrubbed = {};
3
+ for (const [key, value] of Object.entries(process.env))if (!key.startsWith("GIT_")) scrubbed[key] = value;
4
+ return {
5
+ ...scrubbed,
6
+ ...extra
7
+ };
8
+ }
9
+ export { sandboxGitEnv };
@@ -0,0 +1,4 @@
1
+ function warnUtmDeprecated(what) {
2
+ process.stderr.write(`DEPRECATED: ${what} manages a local UTM worker VM. UTM was a testing path and is not the fleet.\nCapture on the bare-metal fleet instead: npm run fleet:status, npm run fleet:deploy. See CLAUDE.md's\n"Working on a Mac" section.\n`);
3
+ }
4
+ export { warnUtmDeprecated };
@@ -3,4 +3,3 @@
3
3
  * @returns {boolean}
4
4
  */
5
5
  export function isTransient(error: unknown): boolean;
6
- //# sourceMappingURL=transient-fault.d.mts.map
@@ -1,86 +1,26 @@
1
- // @ts-check
2
- /**
3
- * Is a capture-worker failure recoverable, or the end of this case?
4
- *
5
- * MOVED HERE from `packages/lab/src/training/capture-decisions.mjs` — architecture-audit.md §5, item 3:
6
- * "the protocol version and fault codes are reached by scraping" because no shared, dependency-free home
7
- * existed for classification that both the lab AND anything else speaking to a worker over HTTP need. This
8
- * package already exists for exactly that ("host-side lifecycle, health and capacity for a fleet of
9
- * Windows NVDA capture workers"), and `capture-client.mjs` — which needs this to decide whether a lost
10
- * response is worth reconciling rather than failing outright — moved here alongside it for the same
11
- * reason: `packages/cli` can depend on `@a11ign/screenreader-fleet` (it already does, for `requestJson`)
12
- * but must never depend on `@a11ign/lab`, which is private and never published.
13
- *
14
- * `capture-decisions.mjs` re-exports `isTransient` from here so every existing lab-side importer is
15
- * unchanged.
16
- */
17
- // BY CODE, from the module that defines them — architecture-audit.md §5, item 3 and item 4: "fault codes
18
- // are copied as string literals... because no ./capture-faults subpath is exported". `capture-faults.mjs`
19
- // has no imports of its own, so it was always safe to expose; the subpath just did not exist. Reading the
20
- // actual codes here means a renamed fault cannot silently stop being recognised as recoverable.
21
1
  import { FAULT } from "@a11ign/screenreader-worker/capture-faults";
22
- /**
23
- * Recoverable, or the end of this case?
24
- *
25
- * Everything here heals on its own, which is why waiting beats failing. The connection errors are
26
- * here because the first full dataset run lost its last four cases to one guest bugchecking — it came
27
- * back by itself, but the run had already recorded four permanent failures.
28
- *
29
- * `running but not speaking` and `hard timeout` are the subtle ones: both make the worker STOP its
30
- * screen reader, so the next capture cold-starts a fresh one. They are self-healing by construction,
31
- * and classifying them fatal cost a case in the run that proved it.
32
- */
33
- const TRANSIENT = new RegExp([
34
- "fetch failed", "ECONNREFUSED", "ECONNRESET", "socket hang up", "timed out", "aborted",
35
- "HTTP 429.*capture is already in progress",
36
- "running but not speaking",
37
- "hard timeout",
38
- ].join("|"), "i");
39
- /**
40
- * Faults the WORKER named for us, which never need matching against prose.
41
- *
42
- * Both self-heal: the worker stops NVDA on any failed capture, so the next attempt cold-starts a clean
43
- * one. The worker now retries these itself before answering, so seeing one here means even its retry
44
- * did not clear it — still worth reissuing the case rather than recording a permanent failure.
45
- */
46
- const TRANSIENT_FAULTS = new Set([FAULT.SCREEN_READER_MUTE, FAULT.SCREEN_READER_START_FAILED]);
47
- /**
48
- * Network failures that heal on their own, by CODE rather than by wording.
49
- *
50
- * These became visible when the capture clients moved off `fetch` to `node:http` (see
51
- * `worker-fleet/src/worker-http.mjs` for why they had to). `fetch` collapsed every network failure into
52
- * `TypeError: fetch failed`, which the regex above matched — so the whole class was transient by accident,
53
- * through a wrapper's wording rather than through anything we had decided.
54
- *
55
- * `EHOSTUNREACH` is the one that would have bitten. It is how a bare-metal worker presents while its NIC
56
- * wakes from selective suspend, recorded in provision-nvda-worker.ps1: 48 instant failures in one
57
- * evidence-check run, and the box answered a curl thirty seconds later. Under the real code, and without
58
- * this set, that would now be classified FATAL and fail 48 cases permanently.
59
- *
60
- * `ETIMEDOUT` covers both a dead peer and our own deadline in `requestJson`, which is deliberate: a
61
- * capture that outran its budget is exactly the case the worker recovers from by cold-starting NVDA.
62
- */
2
+ const TRANSIENT = new RegExp("fetch failed|ECONNREFUSED|ECONNRESET|socket hang up|timed out|aborted|HTTP 429.*capture is already in progress|running but not speaking|hard timeout", "i");
3
+ const TRANSIENT_FAULTS = new Set([
4
+ FAULT.SCREEN_READER_MUTE,
5
+ FAULT.SCREEN_READER_START_FAILED
6
+ ]);
63
7
  const TRANSIENT_NETWORK_CODES = new Set([
64
- "ECONNREFUSED", "ECONNRESET", "EHOSTUNREACH", "ENETUNREACH", "ENETDOWN",
65
- "EPIPE", "ETIMEDOUT", "EAI_AGAIN", "UND_ERR_HEADERS_TIMEOUT", "UND_ERR_BODY_TIMEOUT",
8
+ "ECONNREFUSED",
9
+ "ECONNRESET",
10
+ "EHOSTUNREACH",
11
+ "ENETUNREACH",
12
+ "ENETDOWN",
13
+ "EPIPE",
14
+ "ETIMEDOUT",
15
+ "EAI_AGAIN",
16
+ "UND_ERR_HEADERS_TIMEOUT",
17
+ "UND_ERR_BODY_TIMEOUT"
66
18
  ]);
67
- /**
68
- * @param {unknown} error anything a failed request threw — a node:http Error, an undici one, a string
69
- * @returns {boolean}
70
- */
71
- export function isTransient(error) {
72
- const failure = /** @type {{ code?: string, cause?: { code?: string }, message?: string }} */ (error);
73
- // Prefer the code. The regex below is the fallback for older workers and for host-side failures
74
- // (a dropped socket has no fault code), but a message is prose and prose gets reworded — see
75
- // packages/nvda-worker/src/capture-faults.mjs for what that cost.
76
- if (TRANSIENT_FAULTS.has(failure?.code ?? ""))
77
- return true;
78
- if (TRANSIENT_NETWORK_CODES.has(failure?.code ?? ""))
79
- return true;
80
- // A node:http error carries its code on the error itself; an undici one hides it on `cause`. Checking
81
- // both means the classification does not depend on which client the caller happened to use.
82
- if (TRANSIENT_NETWORK_CODES.has(failure?.cause?.code ?? ""))
83
- return true;
19
+ function isTransient(error) {
20
+ const failure = error;
21
+ if (TRANSIENT_FAULTS.has(failure?.code ?? "")) return true;
22
+ if (TRANSIENT_NETWORK_CODES.has(failure?.code ?? "")) return true;
23
+ if (TRANSIENT_NETWORK_CODES.has(failure?.cause?.code ?? "")) return true;
84
24
  return TRANSIENT.test(String(failure?.message ?? error ?? ""));
85
25
  }
86
- //# sourceMappingURL=transient-fault.mjs.map
26
+ export { isTransient };
@@ -3,4 +3,3 @@
3
3
  * specific to what actually fired, not a generic banner every UTM-adjacent file prints identically.
4
4
  */
5
5
  export function warnUtmDeprecated(what: string): void;
6
- //# sourceMappingURL=utm-deprecated.d.mts.map
@@ -26,4 +26,3 @@ import { readWorkerCode } from "./code-drift.mjs";
26
26
  import { remedyLines } from "./code-drift.mjs";
27
27
  import { workerSourceDirty } from "./code-drift.mjs";
28
28
  export { codeDrift, describeCodeDrift, describeEmptyPool, readWorkerCode, remedyLines, workerSourceDirty };
29
- //# sourceMappingURL=worker-code-check.d.mts.map
@@ -1,78 +1,129 @@
1
- // @ts-check
2
- /**
3
- * Is the fleet running the code this checkout expects — asked BEFORE a capture run, not after it.
4
- *
5
- * ## The hole this closes
6
- *
7
- * `run-job.yml` refuses to run at a commit other than the one asked for, and the comment above that
8
- * refusal says why: *"a job that quietly runs four commits behind reports success for code you did not
9
- * ask for."* That guard covers the LAB. It says nothing about the twelve machines that actually take the
10
- * captures, and those are a second checkout, deployed by a separate command nobody is forced to run.
11
- *
12
- * So a capture run could be dispatched at the right commit, on a lab that proved it was at the right
13
- * commit, and still capture with the PREVIOUS release of `capture-core.mjs`. Measured on 2026-08-25: after
14
- * `MAX_TAB_STOPS` went 12 -> 150 and `collectByType` started recording `prevCount`, the real-page corpus
15
- * held both populations at once, and the only way to read it was to bucket captures by whether they
16
- * carried the new diagnostic mark at all. The evidence was mixed, the run reported success, and the
17
- * separation had to be done by hand afterwards.
18
- *
19
- * `npm run worker:code` has answered this question correctly the whole time. It is a separate command a
20
- * human must remember, which is this repo's own definition of a check that does not happen — and it was
21
- * remembered by hand four times in one day before this existed.
22
- *
23
- * ## Why a REFUSAL, and why on any difference at all
24
- *
25
- * `workerCode` is deliberately outside the capture cache key ("it changes when a comment changes, and
26
- * invalidating the WHOLE corpus over a reworded comment is how a cache becomes something people turn
27
- * off") and deliberately outside
28
- * `fleet-consistency.mjs`'s `MUST_MATCH` for the same reason. Both of those are the right call for
29
- * the questions they answer — *is this evidence still valid* and *are these guests interchangeable*.
30
- *
31
- * This is a third question with a different answer: *am I about to capture with the code I asked for*. A
32
- * comment-only drift is a false alarm here and it costs one `fleet:deploy`; a real drift costs a corpus and
33
- * is invisible, because nothing downstream keys on `workerCode`. That asymmetry is the whole argument.
34
- *
35
- * It is a PRECONDITION and never a key: nothing here invalidates a cached capture.
36
- *
37
- * ## The comparison itself lives in `code-drift.mjs`, and this file is the reason for the split
38
- *
39
- * `expectedWorkerCode` below needs `codeVersion`/`workerSourceDir`, reached through a SUBPATH export
40
- * (`@a11ign/screenreader-worker/code-version`) rather than a relative path — a relative one drags
41
- * `nvda-worker`'s `.mjs` files into this package's own tsc project and the build dies with TS5055 ("would
42
- * overwrite input file"). That subpath resolves through `node_modules`, which is exactly what
43
- * `packages/control` does not have (ADR 0012) — so when `lab-job.mjs` needed this same comparison BEFORE
44
- * dispatching to the lab, it could not import this file. `code-drift.mjs` is the part of this file with no
45
- * opinion about what "expected" means: it takes the hash as a parameter, imports nothing but
46
- * `node:child_process`, and is safe from both places. This file supplies the one thing only it can compute.
47
- */
48
- import { codeDrift, describeCodeDrift, describeEmptyPool, readWorkerCode, remedyLines, workerSourceDirty, assertWorkersServe } from "./code-drift.mjs";
49
- // A SUBPATH export, not a deep relative path: `../../nvda-worker/src/...` drags those .mjs files into
50
- // worker-fleet's tsc project and the build dies with TS5055 "would overwrite input file". The subpath is
51
- // also the shape already in use for the same reason -- `@a11ign/screenreader-fleet/worker-http`.
52
- // `code-version.mjs` imports nothing but node stdlib and `worker-files.mjs`, which is why it is safe and
53
- // why it is its own module. Still the ONE hasher: the subpath is the same function.
1
+ import { execFileSync } from "node:child_process";
54
2
  import { codeVersion, workerSourceDir } from "@a11ign/screenreader-worker/code-version";
55
- /** The hash this checkout expects every worker to be serving. One hasher, shared with the guest. */
56
- export const expectedWorkerCode = () => codeVersion(workerSourceDir());
57
- // Re-exported rather than duplicated: existing callers (`capture-real-pages.mjs`,
58
- // `capture-screenreader-dataset.mjs`, and this module's own test) import these from here, and moving their
59
- // implementation to `code-drift.mjs` must not become a second place either has to be found.
60
- export { codeDrift, describeCodeDrift, describeEmptyPool, readWorkerCode, remedyLines, workerSourceDirty };
61
- /**
62
- * Refuse to capture with a fleet that is not running this checkout.
63
- *
64
- * Called at the boundary of every capture entry point, for the reason `assertWorkerUrl` is: the
65
- * alternative is discovering it in the evidence weeks later, where a stale worker looks like a page that
66
- * changed. **Both entry points, not one** — a remedy that reaches one of several paths is the shape this
67
- * repo has paid for three times over (`anchorToTop`, `ensureSpeechChannel`, `waitForAnnouncement`), and
68
- * `capture-preflight.test.ts` pins that both call it.
69
- *
70
- * A thin wrapper over `assertWorkersServe`, supplying the one thing only this file can compute: the hash.
71
- *
72
- * @param {string[]} workers
73
- * @param {{when?: string, allow?: boolean, read?: (url: string) => Promise<string|null>, bareMetalUrls?: string[]}} options
74
- */
75
- export async function assertFleetRunsThisCheckout(workers, options = {}) {
76
- return assertWorkersServe(expectedWorkerCode(), workers, { ...options, sourceDir: workerSourceDir() });
3
+ import { requestJson } from "./worker-http.mjs";
4
+ import { sandboxGitEnv } from "./src_git-safe-env_mjs.mjs";
5
+ const HEALTH_TIMEOUT_MS = 15000;
6
+ function codeDrift(expected, readings) {
7
+ const stale = [];
8
+ const unreachable = [];
9
+ let answered = 0;
10
+ for (const reading of readings ?? []){
11
+ const code = reading?.code;
12
+ if (null == code) {
13
+ unreachable.push(reading?.worker);
14
+ continue;
15
+ }
16
+ answered += 1;
17
+ if (code !== expected) stale.push({
18
+ worker: reading.worker,
19
+ serving: String(code)
20
+ });
21
+ }
22
+ return {
23
+ expected,
24
+ stale,
25
+ unreachable,
26
+ answered
27
+ };
77
28
  }
78
- //# sourceMappingURL=worker-code-check.mjs.map
29
+ function remedyLines(staleUrls, bareMetalUrls) {
30
+ const bareMetal = new Set(bareMetalUrls);
31
+ const physical = staleUrls.filter((u)=>bareMetal.has(u));
32
+ const vms = staleUrls.filter((u)=>!bareMetal.has(u));
33
+ const lines = [
34
+ `\n${staleUrls.length} stale worker(s).`
35
+ ];
36
+ if (physical.length) lines.push(`\n ${physical.length} in inventory.yml — bare metal, so they deploy by PULLING:`, " npm run fleet:deploy", " `npm run worker:deploy` cannot reach these: it is utmctl, keyed on a VM UUID.");
37
+ if (vms.length) lines.push(`\n ${vms.length} not in inventory.yml — local VM(s). A restart via \`utmctl exec\``, " silently does nothing on some guests; rebooting always picks up a pushed file:", " npm run worker:deploy");
38
+ return lines;
39
+ }
40
+ function describeCodeDrift(drift, { when = "before the run", bareMetalUrls = [], sourceDirty = "" } = {}) {
41
+ if (drift?.answered === 0 && drift?.unreachable?.length) return [
42
+ "",
43
+ `REFUSING to vouch for the fleet ${when}: ${drift.unreachable.length} worker(s) were asked and NONE`,
44
+ `answered, so nothing was compared against this checkout (${drift.expected}).`,
45
+ ...drift.unreachable.map((w)=>` ${w}`),
46
+ "",
47
+ "This is not a clean fleet, it is an unexamined one. A capture is about to dispatch to these boxes,",
48
+ "so silence here is a broken invocation rather than a pass. Common causes: the fleet is powered",
49
+ "down (npm run fleet:status), or a deploy just rebooted it and nothing waited.",
50
+ "",
51
+ "Or pass --allow-stale-workers if you know something this check does not.",
52
+ ""
53
+ ].join("\n");
54
+ if (!drift?.stale?.length) return null;
55
+ const lines = [
56
+ `\nFLEET IS NOT RUNNING THIS CHECKOUT ${when}.`,
57
+ `This checkout expects worker code ${drift.expected}; ${drift.stale.length} worker(s) serve something else:`,
58
+ ...drift.stale.map(({ worker, serving })=>` ${worker} ${serving}`)
59
+ ];
60
+ if (drift.unreachable.length) lines.push(` (${drift.unreachable.length} worker(s) did not answer and were not judged: ${drift.unreachable.join(", ")})`);
61
+ lines.push("", "A capture stamps the commit the LAB is at, and nothing downstream keys on the worker's code —", "`workerCode` is outside the cache key on purpose — so evidence taken by a stale worker is", "indistinguishable from current evidence for ever after.");
62
+ if (sourceDirty) lines.push("", "But the drift is on THIS side: the worker source in this checkout is modified against HEAD —", ` ${sourceDirty}`, "so the fleet may be perfectly current and deploying would ship uncommitted work. Commit or revert", "first, then re-check. Do not reach for the remedy below until this is clean.");
63
+ lines.push(...remedyLines(drift.stale.map((s)=>s.worker), bareMetalUrls));
64
+ lines.push("", "Or pass --allow-stale-workers if you know something this check does not; it will say so in", "the output rather than passing quietly.");
65
+ return `${lines.join("\n")}\n`;
66
+ }
67
+ async function readWorkerCode(url) {
68
+ try {
69
+ const response = await requestJson(`${String(url).replace(/\/$/, "")}/health`, {
70
+ timeoutMs: HEALTH_TIMEOUT_MS
71
+ });
72
+ if (void 0 === response.json) throw new Error(`invalid JSON from ${url}`);
73
+ return response.json.code ?? "absent";
74
+ } catch {
75
+ return null;
76
+ }
77
+ }
78
+ function workerSourceDirty(sourceDir) {
79
+ if ("string" != typeof sourceDir || !sourceDir) throw new TypeError("workerSourceDirty needs the worker source directory to read; there is no default");
80
+ try {
81
+ return execFileSync("git", [
82
+ "-C",
83
+ sourceDir,
84
+ "status",
85
+ "--porcelain",
86
+ "--",
87
+ "."
88
+ ], {
89
+ encoding: "utf8",
90
+ env: sandboxGitEnv()
91
+ }).trim().split("\n").filter(Boolean).join("; ");
92
+ } catch {
93
+ return "";
94
+ }
95
+ }
96
+ function describeEmptyPool(workers, expected) {
97
+ if (workers?.length) return null;
98
+ return `REFUSING to vouch for the fleet: no workers were given, so nothing was compared against this checkout (${expected}). A capture dispatches to workers, so an empty pool is a broken invocation rather than a clean fleet — check A11Y_WORKER(S), the local pool, or inventory.yml.\n`;
99
+ }
100
+ async function assertWorkersServe(expected, workers, options) {
101
+ const { when = "before the run", allow = false, read = readWorkerCode, bareMetalUrls = [], sourceDir } = options;
102
+ if (allow) return void process.stdout.write("--allow-stale-workers: NOT checking that the fleet runs this checkout.\n");
103
+ const empty = describeEmptyPool(workers, expected);
104
+ if (empty) {
105
+ process.stderr.write(empty);
106
+ process.exit(3);
107
+ }
108
+ const readings = await Promise.all(workers.map(async (worker)=>({
109
+ worker,
110
+ code: await read(worker)
111
+ })));
112
+ const drift = codeDrift(expected, readings);
113
+ const refusal = describeCodeDrift(drift, {
114
+ when,
115
+ bareMetalUrls,
116
+ sourceDirty: workerSourceDirty(sourceDir)
117
+ });
118
+ if (!refusal) return void process.stdout.write(`Fleet runs this checkout (worker code ${expected}, ${drift.unreachable.length ? `${readings.length - drift.unreachable.length} of ` : ""}${readings.length} worker(s) checked).\n`);
119
+ process.stderr.write(refusal);
120
+ process.exit(3);
121
+ }
122
+ const expectedWorkerCode = ()=>codeVersion();
123
+ async function assertFleetRunsThisCheckout(workers, options = {}) {
124
+ return assertWorkersServe(expectedWorkerCode(), workers, {
125
+ ...options,
126
+ sourceDir: workerSourceDir()
127
+ });
128
+ }
129
+ export { assertFleetRunsThisCheckout, codeDrift, describeCodeDrift, describeEmptyPool, expectedWorkerCode, readWorkerCode, remedyLines, workerSourceDirty };
@@ -53,4 +53,3 @@ export function assessWorker(vitals: {
53
53
  reason: string | null;
54
54
  recoveryShare: number | null;
55
55
  };
56
- //# sourceMappingURL=worker-health.d.mts.map
@@ -1,73 +1,28 @@
1
- // @ts-check
2
- /**
3
- * Is a worker healthy, degraded, or unusable — from the vitals it reports.
4
- *
5
- * This exists because a guest whose NVDA was broken on **every single capture** sat in the pool at four
6
- * times the cost of its neighbours and nothing noticed. Measured, same page, same code, same moment:
7
- *
8
- * worker 1 4 captures, 4 recoveries, 0 failures nvdaStart 19.1s/capture WALL 122.9s
9
- * worker 2 9 captures, 0 recoveries, 0 failures nvdaStart 0.0s/capture WALL 40.6s
10
- *
11
- * Two things conspired to hide it. The worker's own retry absorbed every fault, so `failures` stayed 0
12
- * and the run's eviction rule — three consecutive FAILURES — could never fire. And wall-clock time only
13
- * said "slower", which I twice misattributed to Edge.
14
- *
15
- * So degradation is defined on the recovery RATE, not on failures. `recoveries` counts faults the worker
16
- * papered over for the caller, which makes it the one number that rises while everything still appears
17
- * to work.
18
- *
19
- * **Degraded workers keep taking work.** They are slow, not broken, and pulling one from a three-VM pool
20
- * costs more throughput than it saves. This mirrors the standard health-check split — a degraded service
21
- * returns 200 and is *surfaced* rather than restarted, because declaring degraded things unhealthy is how
22
- * you end up with nothing left to serve (Distributed Systems with Node.js, ch. 4).
23
- */
24
- /**
25
- * Can this worker take a capture right now? ONE definition, because there were four that disagreed.
26
- *
27
- * `ready` is about the ENVIRONMENT — Edge resolvable, ForegroundLockTimeout 0, the worker free — and a
28
- * worker reports `ready: false` while NVDA warms up after a boot, which is normal and self-correcting.
29
- *
30
- * The subtlety, and the reason this is `!== false` rather than `=== true`: **a worker predating the
31
- * field reports neither.** Treating absent as ready keeps an un-redeployed guest working instead of
32
- * stalling a run against it forever; staleness has its own detector in `npm run worker:code`. The
33
- * dataset runner got this right and said so. `repeat-capture.mjs` tested `health.ready` for truthiness
34
- * and `capture-real-pages.mjs` tested `=== true`, so both would have waited out their whole readiness
35
- * budget against a perfectly good older worker and then blamed the page.
36
- *
37
- * @param {{ busy?: boolean, ready?: boolean } | null | undefined} health
38
- * @returns {boolean}
39
- */
40
- export function workerIsUsable(health) {
41
- if (!health)
42
- return false;
43
- return !health.busy && health.ready !== false;
1
+ function workerIsUsable(health) {
2
+ if (!health) return false;
3
+ return !health.busy && false !== health.ready;
44
4
  }
45
- /** Below this many captures the rate is noise: one recovery out of one capture is not a pattern. */
46
5
  const MIN_CAPTURES_TO_JUDGE = 4;
47
- /** Above this share of captures needing a recovery, the guest is not merely unlucky. */
48
6
  const DEGRADED_RECOVERY_SHARE = 0.5;
49
- /**
50
- * @param {{ captures?: number, recoveries?: number, failures?: number } | null | undefined} vitals
51
- * @returns {{ degraded: boolean, reason: string | null, recoveryShare: number | null }}
52
- */
53
- export function assessWorker(vitals) {
7
+ function assessWorker(vitals) {
54
8
  const captures = vitals?.captures ?? 0;
55
9
  const recoveries = vitals?.recoveries ?? 0;
56
- // Recoveries are counted per capture served, so the share can exceed nothing sensible above 1.
57
10
  const attempted = captures + (vitals?.failures ?? 0);
58
- if (attempted < MIN_CAPTURES_TO_JUDGE) {
59
- return { degraded: false, reason: null, recoveryShare: null };
60
- }
11
+ if (attempted < MIN_CAPTURES_TO_JUDGE) return {
12
+ degraded: false,
13
+ reason: null,
14
+ recoveryShare: null
15
+ };
61
16
  const recoveryShare = recoveries / attempted;
62
- if (recoveryShare <= DEGRADED_RECOVERY_SHARE) {
63
- return { degraded: false, reason: null, recoveryShare };
64
- }
17
+ if (recoveryShare <= DEGRADED_RECOVERY_SHARE) return {
18
+ degraded: false,
19
+ reason: null,
20
+ recoveryShare
21
+ };
65
22
  return {
66
23
  degraded: true,
67
24
  recoveryShare,
68
- reason: `${recoveries} of ${attempted} captures needed a screen-reader recovery ` +
69
- `(${Math.round(recoveryShare * 100)}%) — this guest's NVDA is failing and every capture pays for it. ` +
70
- "Reinstall NVDA or re-provision it (docs/nvda-worker-runbook.md); it is still serving, just slowly.",
25
+ reason: `${recoveries} of ${attempted} captures needed a screen-reader recovery (${Math.round(100 * recoveryShare)}%) — this guest's NVDA is failing and every capture pays for it. Reinstall NVDA or re-provision it (docs/nvda-worker-runbook.md); it is still serving, just slowly.`
71
26
  };
72
27
  }
73
- //# sourceMappingURL=worker-health.mjs.map
28
+ export { assessWorker, workerIsUsable };
@@ -100,4 +100,3 @@ export const CAPTURE_CLIENT_TIMEOUT_MS: 620000;
100
100
  * the server's call and passed with this hook DELETED — found by mutation, not by reading.
101
101
  */
102
102
  export const KEEPALIVE_DELAY_MS: 15000;
103
- //# sourceMappingURL=worker-http.d.mts.map