@a11ign/screenreader-fleet 0.1.0 → 0.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/capture-client.d.mts +0 -1
- package/dist/capture-client.mjs +144 -306
- package/dist/check-worker-code.d.mts +0 -1
- package/dist/check-worker-code.mjs +42 -141
- package/dist/cli-flags.d.mts +0 -1
- package/dist/cli-flags.mjs +33 -179
- package/dist/code-drift.d.mts +0 -1
- package/dist/command-line-census.d.mts +0 -1
- package/dist/compare-workers.d.mts +0 -1
- package/dist/compare-workers.mjs +383 -255
- package/dist/control-plane-isolation.d.mts +0 -1
- package/dist/deploy-worker.d.mts +0 -1
- package/dist/deploy-worker.mjs +101 -242
- package/dist/doctor.d.mts +0 -1
- package/dist/doctor.mjs +361 -809
- package/dist/fleet-consistency.d.mts +0 -1
- package/dist/fleet-consistency.mjs +155 -379
- package/dist/fleet-env.d.mts +0 -1
- package/dist/fleet-env.mjs +148 -438
- package/dist/fleet-scripts.d.mts +0 -1
- package/dist/git-safe-env.d.mts +0 -1
- package/dist/guest-run.d.mts +0 -1
- package/dist/host-address.d.mts +0 -1
- package/dist/host-address.mjs +19 -90
- package/dist/host-capacity.d.mts +0 -1
- package/dist/host-capacity.mjs +22 -136
- package/dist/host-metrics.d.mts +0 -1
- package/dist/index.d.ts +0 -1
- package/dist/index.mjs +231 -0
- package/dist/local-vm.d.ts +0 -1
- package/dist/measure-guard.d.mts +0 -1
- package/dist/normalise-fleet.d.mts +0 -1
- package/dist/npm-cli-executable.d.mts +0 -1
- package/dist/probe-outcome.d.mts +0 -1
- package/dist/probe-outcome.mjs +50 -96
- package/dist/protocol-guard.d.mts +0 -1
- package/dist/source-walk.d.mts +0 -1
- package/dist/src_fleet-scripts_mjs.mjs +16 -0
- package/dist/src_git-safe-env_mjs.mjs +9 -0
- package/dist/src_utm-deprecated_mjs.mjs +4 -0
- package/dist/transient-fault.d.mts +0 -1
- package/dist/transient-fault.mjs +21 -81
- package/dist/utm-deprecated.d.mts +0 -1
- package/dist/worker-code-check.d.mts +0 -1
- package/dist/worker-code-check.mjs +127 -76
- package/dist/worker-health.d.mts +0 -1
- package/dist/worker-health.mjs +16 -61
- package/dist/worker-http.d.mts +0 -1
- package/dist/worker-http.mjs +42 -234
- package/dist/worker-stats.d.mts +0 -1
- package/package.json +12 -5
- package/dist/capture-client.d.mts.map +0 -1
- package/dist/capture-client.mjs.map +0 -1
- package/dist/check-worker-code.d.mts.map +0 -1
- package/dist/check-worker-code.mjs.map +0 -1
- package/dist/cli-flags.d.mts.map +0 -1
- package/dist/cli-flags.mjs.map +0 -1
- package/dist/code-drift.d.mts.map +0 -1
- package/dist/code-drift.mjs +0 -284
- package/dist/code-drift.mjs.map +0 -1
- package/dist/command-line-census.d.mts.map +0 -1
- package/dist/command-line-census.mjs +0 -96
- package/dist/command-line-census.mjs.map +0 -1
- package/dist/compare-workers.d.mts.map +0 -1
- package/dist/compare-workers.mjs.map +0 -1
- package/dist/control-plane-isolation.d.mts.map +0 -1
- package/dist/control-plane-isolation.mjs +0 -67
- package/dist/control-plane-isolation.mjs.map +0 -1
- package/dist/deploy-worker.d.mts.map +0 -1
- package/dist/deploy-worker.mjs.map +0 -1
- package/dist/doctor.d.mts.map +0 -1
- package/dist/doctor.mjs.map +0 -1
- package/dist/fleet-consistency.d.mts.map +0 -1
- package/dist/fleet-consistency.mjs.map +0 -1
- package/dist/fleet-env.d.mts.map +0 -1
- package/dist/fleet-env.mjs.map +0 -1
- package/dist/fleet-scripts.d.mts.map +0 -1
- package/dist/fleet-scripts.mjs +0 -41
- package/dist/fleet-scripts.mjs.map +0 -1
- package/dist/git-safe-env.d.mts.map +0 -1
- package/dist/git-safe-env.mjs +0 -44
- package/dist/git-safe-env.mjs.map +0 -1
- package/dist/guest-run.d.mts.map +0 -1
- package/dist/guest-run.mjs +0 -164
- package/dist/guest-run.mjs.map +0 -1
- package/dist/host-address.d.mts.map +0 -1
- package/dist/host-address.mjs.map +0 -1
- package/dist/host-capacity.d.mts.map +0 -1
- package/dist/host-capacity.mjs.map +0 -1
- package/dist/host-metrics.d.mts.map +0 -1
- package/dist/host-metrics.mjs +0 -201
- package/dist/host-metrics.mjs.map +0 -1
- package/dist/index.d.ts.map +0 -1
- package/dist/index.js +0 -25
- package/dist/index.js.map +0 -1
- package/dist/local-vm.d.ts.map +0 -1
- package/dist/local-vm.js +0 -360
- package/dist/local-vm.js.map +0 -1
- package/dist/measure-guard.d.mts.map +0 -1
- package/dist/measure-guard.mjs +0 -73
- package/dist/measure-guard.mjs.map +0 -1
- package/dist/normalise-fleet.d.mts.map +0 -1
- package/dist/normalise-fleet.mjs +0 -76
- package/dist/normalise-fleet.mjs.map +0 -1
- package/dist/npm-cli-executable.d.mts.map +0 -1
- package/dist/npm-cli-executable.mjs +0 -159
- package/dist/npm-cli-executable.mjs.map +0 -1
- package/dist/probe-outcome.d.mts.map +0 -1
- package/dist/probe-outcome.mjs.map +0 -1
- package/dist/protocol-guard.d.mts.map +0 -1
- package/dist/protocol-guard.mjs +0 -121
- package/dist/protocol-guard.mjs.map +0 -1
- package/dist/source-walk.d.mts.map +0 -1
- package/dist/source-walk.mjs +0 -56
- package/dist/source-walk.mjs.map +0 -1
- package/dist/transient-fault.d.mts.map +0 -1
- package/dist/transient-fault.mjs.map +0 -1
- package/dist/utm-deprecated.d.mts.map +0 -1
- package/dist/utm-deprecated.mjs +0 -23
- package/dist/utm-deprecated.mjs.map +0 -1
- package/dist/worker-code-check.d.mts.map +0 -1
- package/dist/worker-code-check.mjs.map +0 -1
- package/dist/worker-health.d.mts.map +0 -1
- package/dist/worker-health.mjs.map +0 -1
- package/dist/worker-http.d.mts.map +0 -1
- package/dist/worker-http.mjs.map +0 -1
- package/dist/worker-stats.d.mts.map +0 -1
- package/dist/worker-stats.mjs +0 -143
- package/dist/worker-stats.mjs.map +0 -1
package/dist/probe-outcome.mjs
CHANGED
|
@@ -1,104 +1,58 @@
|
|
|
1
|
-
// @ts-check
|
|
2
|
-
/**
|
|
3
|
-
* WHAT ONE `/health` PROBE CAN SAY, for the entries that reach a worker from the PUBLISHED side and by hand
|
|
4
|
-
* (`witness`, `worker:compare`, `auth:leak-check`, #2683 of #2655).
|
|
5
|
-
*
|
|
6
|
-
* Imports `worker-http` and nothing else: no `control` (this package is published and `@a11ign/control` never
|
|
7
|
-
* is, `worker-fleet-does-not-read-control.test.ts`) and no corpus reader, so a test may import it without
|
|
8
|
-
* pulling the corpus closure in. These entries do NOT wake a box (ADR 0012, product-manager 2026-09-26): they
|
|
9
|
-
* say it did not answer, and name the command that wakes one. `packages/control/src/fleet-wake.mjs` keeps its
|
|
10
|
-
* own `probeWorker` with the same five outcomes because it is the unpublished side of that line.
|
|
11
|
-
*
|
|
12
|
-
* The outcomes, and the one distinction that matters (an unanswered probe is UNKNOWN, never "down"):
|
|
13
|
-
*
|
|
14
|
-
* ready the box's own report, `ready: true`
|
|
15
|
-
* busy the box's own report, `busy: true`: a capture is running. Up, and not free
|
|
16
|
-
* not-ready it answered and its own `ready:false` says why, or it answered a non-2xx status. UP
|
|
17
|
-
* refused the connection was refused, so something answered the TCP handshake with a reset: the BOX IS UP
|
|
18
|
-
* and the worker is not listening. UP
|
|
19
|
-
* no-answer nothing came back inside the timeout (`timedOut`), or the transport failed some other way
|
|
20
|
-
* (unreachable, reset). One silent probe cannot separate "off" from "slow" from "the path dropped
|
|
21
|
-
* it", so it is never called down
|
|
22
|
-
*
|
|
23
|
-
* `health` is the body the box sent, on the three outcomes where one arrived, because "usable" is a different
|
|
24
|
-
* question in different entries (`workerIsUsable` counts a worker predating the `ready` field as usable; the
|
|
25
|
-
* measurement guard wants `ready: true`) and this module does not choose for them.
|
|
26
|
-
*
|
|
27
|
-
* @typedef {{ outcome: "ready", health: any } | { outcome: "busy", health: any }
|
|
28
|
-
* | { outcome: "not-ready", reason: string, health: any }
|
|
29
|
-
* | { outcome: "refused", message: string }
|
|
30
|
-
* | { outcome: "no-answer", message: string, timedOut: boolean }} Probe
|
|
31
|
-
*/
|
|
32
1
|
import { requestJson } from "./worker-http.mjs";
|
|
33
|
-
|
|
34
|
-
|
|
35
|
-
* THE PER-PROBE TIMEOUT, WITH ITS READING. A probe that outlives it is UNKNOWN, never "down", so this number
|
|
36
|
-
* decides how much slowness a healthy box is allowed.
|
|
37
|
-
*
|
|
38
|
-
* All of it READ by others on the real fleet (`orchestrator`, #2671) and none measured by this row, whose
|
|
39
|
-
* engineer is barred from probing it:
|
|
40
|
-
* - the slowest HEALTHY box: 2.80 to 3.09 s on a11y-worker-13, -14 and -16, twelve others 0.53 to 0.76 s;
|
|
41
|
-
* - that is the FIRST answer after the box has been quiet more than 5 s, because the worker rebuilds its
|
|
42
|
-
* environment block with two synchronous `powershell.exe` calls when it is older than that, so a probe made
|
|
43
|
-
* by hand is nearly always the slow case;
|
|
44
|
-
* - a LOADED box (one that has just stopped a capture) can take up to about 10 s: each of the two calls is
|
|
45
|
-
* bounded at 5 s and they stop the worker's event loop for the whole time.
|
|
46
|
-
* 12 s is that loaded ceiling plus 2 s, the number `fleet-wake.mjs` `HEALTH_TIMEOUT_MS` states for the same
|
|
47
|
-
* reading. Before #2683 these entries used 5 s (`witness`, 1.91 s over the 3.09 s box and none over a loaded
|
|
48
|
-
* one), 8 s (`auth:leak-check`) and 10 s (`worker:compare`'s busy guard), read at `d119fb0f2`. Cost of the
|
|
49
|
-
* generosity: a box that really is off costs one 12 s wait before the message. A refusal costs nothing, it comes
|
|
50
|
-
* back at once.
|
|
51
|
-
*/
|
|
52
|
-
export const WORKER_PROBE_TIMEOUT_MS = 12_000;
|
|
53
|
-
/**
|
|
54
|
-
* @param {string} worker the worker's base URL
|
|
55
|
-
* @param {{ timeoutMs?: number, request?: ProbeRequest }} [options]
|
|
56
|
-
* @returns {Promise<Probe>}
|
|
57
|
-
*/
|
|
58
|
-
export async function probeHealth(worker, { timeoutMs = WORKER_PROBE_TIMEOUT_MS, request = requestJson } = {}) {
|
|
2
|
+
const WORKER_PROBE_TIMEOUT_MS = 12000;
|
|
3
|
+
async function probeHealth(worker, { timeoutMs = WORKER_PROBE_TIMEOUT_MS, request = requestJson } = {}) {
|
|
59
4
|
let response;
|
|
60
5
|
try {
|
|
61
|
-
response = await request(`${worker.replace(/\/$/, "")}/health`, {
|
|
62
|
-
|
|
63
|
-
|
|
64
|
-
|
|
65
|
-
|
|
66
|
-
if (
|
|
67
|
-
|
|
68
|
-
|
|
69
|
-
|
|
6
|
+
response = await request(`${worker.replace(/\/$/, "")}/health`, {
|
|
7
|
+
timeoutMs
|
|
8
|
+
});
|
|
9
|
+
} catch (error) {
|
|
10
|
+
const { code, message } = error;
|
|
11
|
+
if ("ECONNREFUSED" === code) return {
|
|
12
|
+
outcome: "refused",
|
|
13
|
+
message: message || code
|
|
14
|
+
};
|
|
15
|
+
return {
|
|
16
|
+
outcome: "no-answer",
|
|
17
|
+
timedOut: "ETIMEDOUT" === code,
|
|
18
|
+
message: `${code ? `${code}: ` : ""}${message}`
|
|
19
|
+
};
|
|
70
20
|
}
|
|
71
21
|
const health = response.json;
|
|
72
|
-
if (!response.ok)
|
|
73
|
-
|
|
74
|
-
|
|
75
|
-
|
|
76
|
-
|
|
77
|
-
|
|
78
|
-
|
|
79
|
-
|
|
22
|
+
if (!response.ok) return {
|
|
23
|
+
outcome: "not-ready",
|
|
24
|
+
reason: `/health answered HTTP ${response.status}`,
|
|
25
|
+
health
|
|
26
|
+
};
|
|
27
|
+
if (health?.busy === true) return {
|
|
28
|
+
outcome: "busy",
|
|
29
|
+
health
|
|
30
|
+
};
|
|
31
|
+
if (health?.ready === true) return {
|
|
32
|
+
outcome: "ready",
|
|
33
|
+
health
|
|
34
|
+
};
|
|
35
|
+
return {
|
|
36
|
+
outcome: "not-ready",
|
|
37
|
+
health,
|
|
38
|
+
reason: response.json?.reason ?? "/health answered without `ready: true` and without a reason"
|
|
39
|
+
};
|
|
80
40
|
}
|
|
81
|
-
|
|
82
|
-
|
|
83
|
-
|
|
84
|
-
|
|
85
|
-
|
|
86
|
-
|
|
87
|
-
|
|
88
|
-
|
|
89
|
-
|
|
90
|
-
|
|
91
|
-
|
|
92
|
-
|
|
93
|
-
|
|
94
|
-
|
|
95
|
-
|
|
96
|
-
case "not-ready": return `${worker} is up and answered, but it says it is not ready: ${probe.reason}.`;
|
|
97
|
-
case "refused": return `${worker} refused the connection (${probe.message}): the machine is up and the worker `
|
|
98
|
-
+ "is not listening. Waking it will not help; start the worker on it.";
|
|
99
|
-
case "no-answer": return `${worker} did not answer${probe.timedOut ? ` within ${timeoutMs / 1000} s` : ""} `
|
|
100
|
-
+ `(${probe.message}). That does not say it is off: it may be asleep, slow, or not reachable from here. ${WAKE_HINT}`;
|
|
101
|
-
default: throw new Error(`unknown probe outcome: ${JSON.stringify(probe)}`);
|
|
41
|
+
const WAKE_HINT = "If it is a fleet box that has gone to sleep, wake it from a checkout of the a11ign repo with `npm run fleet:wake -- <name>` (<name> is its entry in inventory.yml).";
|
|
42
|
+
function describeProbe(probe, { worker, timeoutMs = WORKER_PROBE_TIMEOUT_MS }) {
|
|
43
|
+
switch(probe.outcome){
|
|
44
|
+
case "ready":
|
|
45
|
+
return `${worker} answered and is ready.`;
|
|
46
|
+
case "busy":
|
|
47
|
+
return `${worker} is up and busy with a capture.`;
|
|
48
|
+
case "not-ready":
|
|
49
|
+
return `${worker} is up and answered, but it says it is not ready: ${probe.reason}.`;
|
|
50
|
+
case "refused":
|
|
51
|
+
return `${worker} refused the connection (${probe.message}): the machine is up and the worker is not listening. Waking it will not help; start the worker on it.`;
|
|
52
|
+
case "no-answer":
|
|
53
|
+
return `${worker} did not answer${probe.timedOut ? ` within ${timeoutMs / 1000} s` : ""} (${probe.message}). That does not say it is off: it may be asleep, slow, or not reachable from here. ${WAKE_HINT}`;
|
|
54
|
+
default:
|
|
55
|
+
throw new Error(`unknown probe outcome: ${JSON.stringify(probe)}`);
|
|
102
56
|
}
|
|
103
57
|
}
|
|
104
|
-
|
|
58
|
+
export { WAKE_HINT, WORKER_PROBE_TIMEOUT_MS, describeProbe, probeHealth };
|
package/dist/source-walk.d.mts
CHANGED
|
@@ -0,0 +1,16 @@
|
|
|
1
|
+
import { fileURLToPath } from "node:url";
|
|
2
|
+
import { join } from "node:path";
|
|
3
|
+
const assetDir = ()=>fileURLToPath(new URL("../src/local-worker/", import.meta.url));
|
|
4
|
+
function fleetScriptPaths() {
|
|
5
|
+
const dir = assetDir();
|
|
6
|
+
return {
|
|
7
|
+
dir,
|
|
8
|
+
workerCtl: join(dir, "worker-ctl.sh"),
|
|
9
|
+
buildVm: join(dir, "build-vm.sh"),
|
|
10
|
+
cloneWorker: join(dir, "clone-worker.sh"),
|
|
11
|
+
createUtmVm: join(dir, "create-utm-vm.sh"),
|
|
12
|
+
fetchWindowsIso: join(dir, "fetch-windows-iso.sh"),
|
|
13
|
+
provisioning: fileURLToPath(new URL("../src/provisioning/", import.meta.url))
|
|
14
|
+
};
|
|
15
|
+
}
|
|
16
|
+
export { fleetScriptPaths };
|
|
@@ -0,0 +1,4 @@
|
|
|
1
|
+
function warnUtmDeprecated(what) {
|
|
2
|
+
process.stderr.write(`DEPRECATED: ${what} manages a local UTM worker VM. UTM was a testing path and is not the fleet.\nCapture on the bare-metal fleet instead: npm run fleet:status, npm run fleet:deploy. See CLAUDE.md's\n"Working on a Mac" section.\n`);
|
|
3
|
+
}
|
|
4
|
+
export { warnUtmDeprecated };
|
package/dist/transient-fault.mjs
CHANGED
|
@@ -1,86 +1,26 @@
|
|
|
1
|
-
// @ts-check
|
|
2
|
-
/**
|
|
3
|
-
* Is a capture-worker failure recoverable, or the end of this case?
|
|
4
|
-
*
|
|
5
|
-
* MOVED HERE from `packages/lab/src/training/capture-decisions.mjs` — architecture-audit.md §5, item 3:
|
|
6
|
-
* "the protocol version and fault codes are reached by scraping" because no shared, dependency-free home
|
|
7
|
-
* existed for classification that both the lab AND anything else speaking to a worker over HTTP need. This
|
|
8
|
-
* package already exists for exactly that ("host-side lifecycle, health and capacity for a fleet of
|
|
9
|
-
* Windows NVDA capture workers"), and `capture-client.mjs` — which needs this to decide whether a lost
|
|
10
|
-
* response is worth reconciling rather than failing outright — moved here alongside it for the same
|
|
11
|
-
* reason: `packages/cli` can depend on `@a11ign/screenreader-fleet` (it already does, for `requestJson`)
|
|
12
|
-
* but must never depend on `@a11ign/lab`, which is private and never published.
|
|
13
|
-
*
|
|
14
|
-
* `capture-decisions.mjs` re-exports `isTransient` from here so every existing lab-side importer is
|
|
15
|
-
* unchanged.
|
|
16
|
-
*/
|
|
17
|
-
// BY CODE, from the module that defines them — architecture-audit.md §5, item 3 and item 4: "fault codes
|
|
18
|
-
// are copied as string literals... because no ./capture-faults subpath is exported". `capture-faults.mjs`
|
|
19
|
-
// has no imports of its own, so it was always safe to expose; the subpath just did not exist. Reading the
|
|
20
|
-
// actual codes here means a renamed fault cannot silently stop being recognised as recoverable.
|
|
21
1
|
import { FAULT } from "@a11ign/screenreader-worker/capture-faults";
|
|
22
|
-
|
|
23
|
-
|
|
24
|
-
|
|
25
|
-
|
|
26
|
-
|
|
27
|
-
* back by itself, but the run had already recorded four permanent failures.
|
|
28
|
-
*
|
|
29
|
-
* `running but not speaking` and `hard timeout` are the subtle ones: both make the worker STOP its
|
|
30
|
-
* screen reader, so the next capture cold-starts a fresh one. They are self-healing by construction,
|
|
31
|
-
* and classifying them fatal cost a case in the run that proved it.
|
|
32
|
-
*/
|
|
33
|
-
const TRANSIENT = new RegExp([
|
|
34
|
-
"fetch failed", "ECONNREFUSED", "ECONNRESET", "socket hang up", "timed out", "aborted",
|
|
35
|
-
"HTTP 429.*capture is already in progress",
|
|
36
|
-
"running but not speaking",
|
|
37
|
-
"hard timeout",
|
|
38
|
-
].join("|"), "i");
|
|
39
|
-
/**
|
|
40
|
-
* Faults the WORKER named for us, which never need matching against prose.
|
|
41
|
-
*
|
|
42
|
-
* Both self-heal: the worker stops NVDA on any failed capture, so the next attempt cold-starts a clean
|
|
43
|
-
* one. The worker now retries these itself before answering, so seeing one here means even its retry
|
|
44
|
-
* did not clear it — still worth reissuing the case rather than recording a permanent failure.
|
|
45
|
-
*/
|
|
46
|
-
const TRANSIENT_FAULTS = new Set([FAULT.SCREEN_READER_MUTE, FAULT.SCREEN_READER_START_FAILED]);
|
|
47
|
-
/**
|
|
48
|
-
* Network failures that heal on their own, by CODE rather than by wording.
|
|
49
|
-
*
|
|
50
|
-
* These became visible when the capture clients moved off `fetch` to `node:http` (see
|
|
51
|
-
* `worker-fleet/src/worker-http.mjs` for why they had to). `fetch` collapsed every network failure into
|
|
52
|
-
* `TypeError: fetch failed`, which the regex above matched — so the whole class was transient by accident,
|
|
53
|
-
* through a wrapper's wording rather than through anything we had decided.
|
|
54
|
-
*
|
|
55
|
-
* `EHOSTUNREACH` is the one that would have bitten. It is how a bare-metal worker presents while its NIC
|
|
56
|
-
* wakes from selective suspend, recorded in provision-nvda-worker.ps1: 48 instant failures in one
|
|
57
|
-
* evidence-check run, and the box answered a curl thirty seconds later. Under the real code, and without
|
|
58
|
-
* this set, that would now be classified FATAL and fail 48 cases permanently.
|
|
59
|
-
*
|
|
60
|
-
* `ETIMEDOUT` covers both a dead peer and our own deadline in `requestJson`, which is deliberate: a
|
|
61
|
-
* capture that outran its budget is exactly the case the worker recovers from by cold-starting NVDA.
|
|
62
|
-
*/
|
|
2
|
+
const TRANSIENT = new RegExp("fetch failed|ECONNREFUSED|ECONNRESET|socket hang up|timed out|aborted|HTTP 429.*capture is already in progress|running but not speaking|hard timeout", "i");
|
|
3
|
+
const TRANSIENT_FAULTS = new Set([
|
|
4
|
+
FAULT.SCREEN_READER_MUTE,
|
|
5
|
+
FAULT.SCREEN_READER_START_FAILED
|
|
6
|
+
]);
|
|
63
7
|
const TRANSIENT_NETWORK_CODES = new Set([
|
|
64
|
-
"ECONNREFUSED",
|
|
65
|
-
"
|
|
8
|
+
"ECONNREFUSED",
|
|
9
|
+
"ECONNRESET",
|
|
10
|
+
"EHOSTUNREACH",
|
|
11
|
+
"ENETUNREACH",
|
|
12
|
+
"ENETDOWN",
|
|
13
|
+
"EPIPE",
|
|
14
|
+
"ETIMEDOUT",
|
|
15
|
+
"EAI_AGAIN",
|
|
16
|
+
"UND_ERR_HEADERS_TIMEOUT",
|
|
17
|
+
"UND_ERR_BODY_TIMEOUT"
|
|
66
18
|
]);
|
|
67
|
-
|
|
68
|
-
|
|
69
|
-
|
|
70
|
-
|
|
71
|
-
|
|
72
|
-
const failure = /** @type {{ code?: string, cause?: { code?: string }, message?: string }} */ (error);
|
|
73
|
-
// Prefer the code. The regex below is the fallback for older workers and for host-side failures
|
|
74
|
-
// (a dropped socket has no fault code), but a message is prose and prose gets reworded — see
|
|
75
|
-
// packages/nvda-worker/src/capture-faults.mjs for what that cost.
|
|
76
|
-
if (TRANSIENT_FAULTS.has(failure?.code ?? ""))
|
|
77
|
-
return true;
|
|
78
|
-
if (TRANSIENT_NETWORK_CODES.has(failure?.code ?? ""))
|
|
79
|
-
return true;
|
|
80
|
-
// A node:http error carries its code on the error itself; an undici one hides it on `cause`. Checking
|
|
81
|
-
// both means the classification does not depend on which client the caller happened to use.
|
|
82
|
-
if (TRANSIENT_NETWORK_CODES.has(failure?.cause?.code ?? ""))
|
|
83
|
-
return true;
|
|
19
|
+
function isTransient(error) {
|
|
20
|
+
const failure = error;
|
|
21
|
+
if (TRANSIENT_FAULTS.has(failure?.code ?? "")) return true;
|
|
22
|
+
if (TRANSIENT_NETWORK_CODES.has(failure?.code ?? "")) return true;
|
|
23
|
+
if (TRANSIENT_NETWORK_CODES.has(failure?.cause?.code ?? "")) return true;
|
|
84
24
|
return TRANSIENT.test(String(failure?.message ?? error ?? ""));
|
|
85
25
|
}
|
|
86
|
-
|
|
26
|
+
export { isTransient };
|
|
@@ -26,4 +26,3 @@ import { readWorkerCode } from "./code-drift.mjs";
|
|
|
26
26
|
import { remedyLines } from "./code-drift.mjs";
|
|
27
27
|
import { workerSourceDirty } from "./code-drift.mjs";
|
|
28
28
|
export { codeDrift, describeCodeDrift, describeEmptyPool, readWorkerCode, remedyLines, workerSourceDirty };
|
|
29
|
-
//# sourceMappingURL=worker-code-check.d.mts.map
|
|
@@ -1,78 +1,129 @@
|
|
|
1
|
-
|
|
2
|
-
/**
|
|
3
|
-
* Is the fleet running the code this checkout expects — asked BEFORE a capture run, not after it.
|
|
4
|
-
*
|
|
5
|
-
* ## The hole this closes
|
|
6
|
-
*
|
|
7
|
-
* `run-job.yml` refuses to run at a commit other than the one asked for, and the comment above that
|
|
8
|
-
* refusal says why: *"a job that quietly runs four commits behind reports success for code you did not
|
|
9
|
-
* ask for."* That guard covers the LAB. It says nothing about the twelve machines that actually take the
|
|
10
|
-
* captures, and those are a second checkout, deployed by a separate command nobody is forced to run.
|
|
11
|
-
*
|
|
12
|
-
* So a capture run could be dispatched at the right commit, on a lab that proved it was at the right
|
|
13
|
-
* commit, and still capture with the PREVIOUS release of `capture-core.mjs`. Measured on 2026-08-25: after
|
|
14
|
-
* `MAX_TAB_STOPS` went 12 -> 150 and `collectByType` started recording `prevCount`, the real-page corpus
|
|
15
|
-
* held both populations at once, and the only way to read it was to bucket captures by whether they
|
|
16
|
-
* carried the new diagnostic mark at all. The evidence was mixed, the run reported success, and the
|
|
17
|
-
* separation had to be done by hand afterwards.
|
|
18
|
-
*
|
|
19
|
-
* `npm run worker:code` has answered this question correctly the whole time. It is a separate command a
|
|
20
|
-
* human must remember, which is this repo's own definition of a check that does not happen — and it was
|
|
21
|
-
* remembered by hand four times in one day before this existed.
|
|
22
|
-
*
|
|
23
|
-
* ## Why a REFUSAL, and why on any difference at all
|
|
24
|
-
*
|
|
25
|
-
* `workerCode` is deliberately outside the capture cache key ("it changes when a comment changes, and
|
|
26
|
-
* invalidating the WHOLE corpus over a reworded comment is how a cache becomes something people turn
|
|
27
|
-
* off") and deliberately outside
|
|
28
|
-
* `fleet-consistency.mjs`'s `MUST_MATCH` for the same reason. Both of those are the right call for
|
|
29
|
-
* the questions they answer — *is this evidence still valid* and *are these guests interchangeable*.
|
|
30
|
-
*
|
|
31
|
-
* This is a third question with a different answer: *am I about to capture with the code I asked for*. A
|
|
32
|
-
* comment-only drift is a false alarm here and it costs one `fleet:deploy`; a real drift costs a corpus and
|
|
33
|
-
* is invisible, because nothing downstream keys on `workerCode`. That asymmetry is the whole argument.
|
|
34
|
-
*
|
|
35
|
-
* It is a PRECONDITION and never a key: nothing here invalidates a cached capture.
|
|
36
|
-
*
|
|
37
|
-
* ## The comparison itself lives in `code-drift.mjs`, and this file is the reason for the split
|
|
38
|
-
*
|
|
39
|
-
* `expectedWorkerCode` below needs `codeVersion`/`workerSourceDir`, reached through a SUBPATH export
|
|
40
|
-
* (`@a11ign/screenreader-worker/code-version`) rather than a relative path — a relative one drags
|
|
41
|
-
* `nvda-worker`'s `.mjs` files into this package's own tsc project and the build dies with TS5055 ("would
|
|
42
|
-
* overwrite input file"). That subpath resolves through `node_modules`, which is exactly what
|
|
43
|
-
* `packages/control` does not have (ADR 0012) — so when `lab-job.mjs` needed this same comparison BEFORE
|
|
44
|
-
* dispatching to the lab, it could not import this file. `code-drift.mjs` is the part of this file with no
|
|
45
|
-
* opinion about what "expected" means: it takes the hash as a parameter, imports nothing but
|
|
46
|
-
* `node:child_process`, and is safe from both places. This file supplies the one thing only it can compute.
|
|
47
|
-
*/
|
|
48
|
-
import { codeDrift, describeCodeDrift, describeEmptyPool, readWorkerCode, remedyLines, workerSourceDirty, assertWorkersServe } from "./code-drift.mjs";
|
|
49
|
-
// A SUBPATH export, not a deep relative path: `../../nvda-worker/src/...` drags those .mjs files into
|
|
50
|
-
// worker-fleet's tsc project and the build dies with TS5055 "would overwrite input file". The subpath is
|
|
51
|
-
// also the shape already in use for the same reason -- `@a11ign/screenreader-fleet/worker-http`.
|
|
52
|
-
// `code-version.mjs` imports nothing but node stdlib and `worker-files.mjs`, which is why it is safe and
|
|
53
|
-
// why it is its own module. Still the ONE hasher: the subpath is the same function.
|
|
1
|
+
import { execFileSync } from "node:child_process";
|
|
54
2
|
import { codeVersion, workerSourceDir } from "@a11ign/screenreader-worker/code-version";
|
|
55
|
-
|
|
56
|
-
|
|
57
|
-
|
|
58
|
-
|
|
59
|
-
|
|
60
|
-
|
|
61
|
-
|
|
62
|
-
|
|
63
|
-
|
|
64
|
-
|
|
65
|
-
|
|
66
|
-
|
|
67
|
-
|
|
68
|
-
|
|
69
|
-
|
|
70
|
-
|
|
71
|
-
|
|
72
|
-
|
|
73
|
-
|
|
74
|
-
|
|
75
|
-
|
|
76
|
-
|
|
3
|
+
import { requestJson } from "./worker-http.mjs";
|
|
4
|
+
import { sandboxGitEnv } from "./src_git-safe-env_mjs.mjs";
|
|
5
|
+
const HEALTH_TIMEOUT_MS = 15000;
|
|
6
|
+
function codeDrift(expected, readings) {
|
|
7
|
+
const stale = [];
|
|
8
|
+
const unreachable = [];
|
|
9
|
+
let answered = 0;
|
|
10
|
+
for (const reading of readings ?? []){
|
|
11
|
+
const code = reading?.code;
|
|
12
|
+
if (null == code) {
|
|
13
|
+
unreachable.push(reading?.worker);
|
|
14
|
+
continue;
|
|
15
|
+
}
|
|
16
|
+
answered += 1;
|
|
17
|
+
if (code !== expected) stale.push({
|
|
18
|
+
worker: reading.worker,
|
|
19
|
+
serving: String(code)
|
|
20
|
+
});
|
|
21
|
+
}
|
|
22
|
+
return {
|
|
23
|
+
expected,
|
|
24
|
+
stale,
|
|
25
|
+
unreachable,
|
|
26
|
+
answered
|
|
27
|
+
};
|
|
77
28
|
}
|
|
78
|
-
|
|
29
|
+
function remedyLines(staleUrls, bareMetalUrls) {
|
|
30
|
+
const bareMetal = new Set(bareMetalUrls);
|
|
31
|
+
const physical = staleUrls.filter((u)=>bareMetal.has(u));
|
|
32
|
+
const vms = staleUrls.filter((u)=>!bareMetal.has(u));
|
|
33
|
+
const lines = [
|
|
34
|
+
`\n${staleUrls.length} stale worker(s).`
|
|
35
|
+
];
|
|
36
|
+
if (physical.length) lines.push(`\n ${physical.length} in inventory.yml — bare metal, so they deploy by PULLING:`, " npm run fleet:deploy", " `npm run worker:deploy` cannot reach these: it is utmctl, keyed on a VM UUID.");
|
|
37
|
+
if (vms.length) lines.push(`\n ${vms.length} not in inventory.yml — local VM(s). A restart via \`utmctl exec\``, " silently does nothing on some guests; rebooting always picks up a pushed file:", " npm run worker:deploy");
|
|
38
|
+
return lines;
|
|
39
|
+
}
|
|
40
|
+
function describeCodeDrift(drift, { when = "before the run", bareMetalUrls = [], sourceDirty = "" } = {}) {
|
|
41
|
+
if (drift?.answered === 0 && drift?.unreachable?.length) return [
|
|
42
|
+
"",
|
|
43
|
+
`REFUSING to vouch for the fleet ${when}: ${drift.unreachable.length} worker(s) were asked and NONE`,
|
|
44
|
+
`answered, so nothing was compared against this checkout (${drift.expected}).`,
|
|
45
|
+
...drift.unreachable.map((w)=>` ${w}`),
|
|
46
|
+
"",
|
|
47
|
+
"This is not a clean fleet, it is an unexamined one. A capture is about to dispatch to these boxes,",
|
|
48
|
+
"so silence here is a broken invocation rather than a pass. Common causes: the fleet is powered",
|
|
49
|
+
"down (npm run fleet:status), or a deploy just rebooted it and nothing waited.",
|
|
50
|
+
"",
|
|
51
|
+
"Or pass --allow-stale-workers if you know something this check does not.",
|
|
52
|
+
""
|
|
53
|
+
].join("\n");
|
|
54
|
+
if (!drift?.stale?.length) return null;
|
|
55
|
+
const lines = [
|
|
56
|
+
`\nFLEET IS NOT RUNNING THIS CHECKOUT ${when}.`,
|
|
57
|
+
`This checkout expects worker code ${drift.expected}; ${drift.stale.length} worker(s) serve something else:`,
|
|
58
|
+
...drift.stale.map(({ worker, serving })=>` ${worker} ${serving}`)
|
|
59
|
+
];
|
|
60
|
+
if (drift.unreachable.length) lines.push(` (${drift.unreachable.length} worker(s) did not answer and were not judged: ${drift.unreachable.join(", ")})`);
|
|
61
|
+
lines.push("", "A capture stamps the commit the LAB is at, and nothing downstream keys on the worker's code —", "`workerCode` is outside the cache key on purpose — so evidence taken by a stale worker is", "indistinguishable from current evidence for ever after.");
|
|
62
|
+
if (sourceDirty) lines.push("", "But the drift is on THIS side: the worker source in this checkout is modified against HEAD —", ` ${sourceDirty}`, "so the fleet may be perfectly current and deploying would ship uncommitted work. Commit or revert", "first, then re-check. Do not reach for the remedy below until this is clean.");
|
|
63
|
+
lines.push(...remedyLines(drift.stale.map((s)=>s.worker), bareMetalUrls));
|
|
64
|
+
lines.push("", "Or pass --allow-stale-workers if you know something this check does not; it will say so in", "the output rather than passing quietly.");
|
|
65
|
+
return `${lines.join("\n")}\n`;
|
|
66
|
+
}
|
|
67
|
+
async function readWorkerCode(url) {
|
|
68
|
+
try {
|
|
69
|
+
const response = await requestJson(`${String(url).replace(/\/$/, "")}/health`, {
|
|
70
|
+
timeoutMs: HEALTH_TIMEOUT_MS
|
|
71
|
+
});
|
|
72
|
+
if (void 0 === response.json) throw new Error(`invalid JSON from ${url}`);
|
|
73
|
+
return response.json.code ?? "absent";
|
|
74
|
+
} catch {
|
|
75
|
+
return null;
|
|
76
|
+
}
|
|
77
|
+
}
|
|
78
|
+
function workerSourceDirty(sourceDir) {
|
|
79
|
+
if ("string" != typeof sourceDir || !sourceDir) throw new TypeError("workerSourceDirty needs the worker source directory to read; there is no default");
|
|
80
|
+
try {
|
|
81
|
+
return execFileSync("git", [
|
|
82
|
+
"-C",
|
|
83
|
+
sourceDir,
|
|
84
|
+
"status",
|
|
85
|
+
"--porcelain",
|
|
86
|
+
"--",
|
|
87
|
+
"."
|
|
88
|
+
], {
|
|
89
|
+
encoding: "utf8",
|
|
90
|
+
env: sandboxGitEnv()
|
|
91
|
+
}).trim().split("\n").filter(Boolean).join("; ");
|
|
92
|
+
} catch {
|
|
93
|
+
return "";
|
|
94
|
+
}
|
|
95
|
+
}
|
|
96
|
+
function describeEmptyPool(workers, expected) {
|
|
97
|
+
if (workers?.length) return null;
|
|
98
|
+
return `REFUSING to vouch for the fleet: no workers were given, so nothing was compared against this checkout (${expected}). A capture dispatches to workers, so an empty pool is a broken invocation rather than a clean fleet — check A11Y_WORKER(S), the local pool, or inventory.yml.\n`;
|
|
99
|
+
}
|
|
100
|
+
async function assertWorkersServe(expected, workers, options) {
|
|
101
|
+
const { when = "before the run", allow = false, read = readWorkerCode, bareMetalUrls = [], sourceDir } = options;
|
|
102
|
+
if (allow) return void process.stdout.write("--allow-stale-workers: NOT checking that the fleet runs this checkout.\n");
|
|
103
|
+
const empty = describeEmptyPool(workers, expected);
|
|
104
|
+
if (empty) {
|
|
105
|
+
process.stderr.write(empty);
|
|
106
|
+
process.exit(3);
|
|
107
|
+
}
|
|
108
|
+
const readings = await Promise.all(workers.map(async (worker)=>({
|
|
109
|
+
worker,
|
|
110
|
+
code: await read(worker)
|
|
111
|
+
})));
|
|
112
|
+
const drift = codeDrift(expected, readings);
|
|
113
|
+
const refusal = describeCodeDrift(drift, {
|
|
114
|
+
when,
|
|
115
|
+
bareMetalUrls,
|
|
116
|
+
sourceDirty: workerSourceDirty(sourceDir)
|
|
117
|
+
});
|
|
118
|
+
if (!refusal) return void process.stdout.write(`Fleet runs this checkout (worker code ${expected}, ${drift.unreachable.length ? `${readings.length - drift.unreachable.length} of ` : ""}${readings.length} worker(s) checked).\n`);
|
|
119
|
+
process.stderr.write(refusal);
|
|
120
|
+
process.exit(3);
|
|
121
|
+
}
|
|
122
|
+
const expectedWorkerCode = ()=>codeVersion();
|
|
123
|
+
async function assertFleetRunsThisCheckout(workers, options = {}) {
|
|
124
|
+
return assertWorkersServe(expectedWorkerCode(), workers, {
|
|
125
|
+
...options,
|
|
126
|
+
sourceDir: workerSourceDir()
|
|
127
|
+
});
|
|
128
|
+
}
|
|
129
|
+
export { assertFleetRunsThisCheckout, codeDrift, describeCodeDrift, describeEmptyPool, expectedWorkerCode, readWorkerCode, remedyLines, workerSourceDirty };
|
package/dist/worker-health.d.mts
CHANGED
package/dist/worker-health.mjs
CHANGED
|
@@ -1,73 +1,28 @@
|
|
|
1
|
-
|
|
2
|
-
|
|
3
|
-
|
|
4
|
-
*
|
|
5
|
-
* This exists because a guest whose NVDA was broken on **every single capture** sat in the pool at four
|
|
6
|
-
* times the cost of its neighbours and nothing noticed. Measured, same page, same code, same moment:
|
|
7
|
-
*
|
|
8
|
-
* worker 1 4 captures, 4 recoveries, 0 failures nvdaStart 19.1s/capture WALL 122.9s
|
|
9
|
-
* worker 2 9 captures, 0 recoveries, 0 failures nvdaStart 0.0s/capture WALL 40.6s
|
|
10
|
-
*
|
|
11
|
-
* Two things conspired to hide it. The worker's own retry absorbed every fault, so `failures` stayed 0
|
|
12
|
-
* and the run's eviction rule — three consecutive FAILURES — could never fire. And wall-clock time only
|
|
13
|
-
* said "slower", which I twice misattributed to Edge.
|
|
14
|
-
*
|
|
15
|
-
* So degradation is defined on the recovery RATE, not on failures. `recoveries` counts faults the worker
|
|
16
|
-
* papered over for the caller, which makes it the one number that rises while everything still appears
|
|
17
|
-
* to work.
|
|
18
|
-
*
|
|
19
|
-
* **Degraded workers keep taking work.** They are slow, not broken, and pulling one from a three-VM pool
|
|
20
|
-
* costs more throughput than it saves. This mirrors the standard health-check split — a degraded service
|
|
21
|
-
* returns 200 and is *surfaced* rather than restarted, because declaring degraded things unhealthy is how
|
|
22
|
-
* you end up with nothing left to serve (Distributed Systems with Node.js, ch. 4).
|
|
23
|
-
*/
|
|
24
|
-
/**
|
|
25
|
-
* Can this worker take a capture right now? ONE definition, because there were four that disagreed.
|
|
26
|
-
*
|
|
27
|
-
* `ready` is about the ENVIRONMENT — Edge resolvable, ForegroundLockTimeout 0, the worker free — and a
|
|
28
|
-
* worker reports `ready: false` while NVDA warms up after a boot, which is normal and self-correcting.
|
|
29
|
-
*
|
|
30
|
-
* The subtlety, and the reason this is `!== false` rather than `=== true`: **a worker predating the
|
|
31
|
-
* field reports neither.** Treating absent as ready keeps an un-redeployed guest working instead of
|
|
32
|
-
* stalling a run against it forever; staleness has its own detector in `npm run worker:code`. The
|
|
33
|
-
* dataset runner got this right and said so. `repeat-capture.mjs` tested `health.ready` for truthiness
|
|
34
|
-
* and `capture-real-pages.mjs` tested `=== true`, so both would have waited out their whole readiness
|
|
35
|
-
* budget against a perfectly good older worker and then blamed the page.
|
|
36
|
-
*
|
|
37
|
-
* @param {{ busy?: boolean, ready?: boolean } | null | undefined} health
|
|
38
|
-
* @returns {boolean}
|
|
39
|
-
*/
|
|
40
|
-
export function workerIsUsable(health) {
|
|
41
|
-
if (!health)
|
|
42
|
-
return false;
|
|
43
|
-
return !health.busy && health.ready !== false;
|
|
1
|
+
function workerIsUsable(health) {
|
|
2
|
+
if (!health) return false;
|
|
3
|
+
return !health.busy && false !== health.ready;
|
|
44
4
|
}
|
|
45
|
-
/** Below this many captures the rate is noise: one recovery out of one capture is not a pattern. */
|
|
46
5
|
const MIN_CAPTURES_TO_JUDGE = 4;
|
|
47
|
-
/** Above this share of captures needing a recovery, the guest is not merely unlucky. */
|
|
48
6
|
const DEGRADED_RECOVERY_SHARE = 0.5;
|
|
49
|
-
|
|
50
|
-
* @param {{ captures?: number, recoveries?: number, failures?: number } | null | undefined} vitals
|
|
51
|
-
* @returns {{ degraded: boolean, reason: string | null, recoveryShare: number | null }}
|
|
52
|
-
*/
|
|
53
|
-
export function assessWorker(vitals) {
|
|
7
|
+
function assessWorker(vitals) {
|
|
54
8
|
const captures = vitals?.captures ?? 0;
|
|
55
9
|
const recoveries = vitals?.recoveries ?? 0;
|
|
56
|
-
// Recoveries are counted per capture served, so the share can exceed nothing sensible above 1.
|
|
57
10
|
const attempted = captures + (vitals?.failures ?? 0);
|
|
58
|
-
if (attempted < MIN_CAPTURES_TO_JUDGE) {
|
|
59
|
-
|
|
60
|
-
|
|
11
|
+
if (attempted < MIN_CAPTURES_TO_JUDGE) return {
|
|
12
|
+
degraded: false,
|
|
13
|
+
reason: null,
|
|
14
|
+
recoveryShare: null
|
|
15
|
+
};
|
|
61
16
|
const recoveryShare = recoveries / attempted;
|
|
62
|
-
if (recoveryShare <= DEGRADED_RECOVERY_SHARE) {
|
|
63
|
-
|
|
64
|
-
|
|
17
|
+
if (recoveryShare <= DEGRADED_RECOVERY_SHARE) return {
|
|
18
|
+
degraded: false,
|
|
19
|
+
reason: null,
|
|
20
|
+
recoveryShare
|
|
21
|
+
};
|
|
65
22
|
return {
|
|
66
23
|
degraded: true,
|
|
67
24
|
recoveryShare,
|
|
68
|
-
reason: `${recoveries} of ${attempted} captures needed a screen-reader recovery
|
|
69
|
-
`(${Math.round(recoveryShare * 100)}%) — this guest's NVDA is failing and every capture pays for it. ` +
|
|
70
|
-
"Reinstall NVDA or re-provision it (docs/nvda-worker-runbook.md); it is still serving, just slowly.",
|
|
25
|
+
reason: `${recoveries} of ${attempted} captures needed a screen-reader recovery (${Math.round(100 * recoveryShare)}%) — this guest's NVDA is failing and every capture pays for it. Reinstall NVDA or re-provision it (docs/nvda-worker-runbook.md); it is still serving, just slowly.`
|
|
71
26
|
};
|
|
72
27
|
}
|
|
73
|
-
|
|
28
|
+
export { assessWorker, workerIsUsable };
|
package/dist/worker-http.d.mts
CHANGED