@a11ign/screenreader-fleet 0.0.0-reserved.0 → 0.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +661 -0
- package/README.md +94 -2
- package/dist/capture-client.d.mts +49 -0
- package/dist/capture-client.d.mts.map +1 -0
- package/dist/capture-client.mjs +352 -0
- package/dist/capture-client.mjs.map +1 -0
- package/dist/check-worker-code.d.mts +34 -0
- package/dist/check-worker-code.d.mts.map +1 -0
- package/dist/check-worker-code.mjs +142 -0
- package/dist/check-worker-code.mjs.map +1 -0
- package/dist/cli-flags.d.mts +71 -0
- package/dist/cli-flags.d.mts.map +1 -0
- package/dist/cli-flags.mjs +207 -0
- package/dist/cli-flags.mjs.map +1 -0
- package/dist/code-drift.d.mts +140 -0
- package/dist/code-drift.d.mts.map +1 -0
- package/dist/code-drift.mjs +284 -0
- package/dist/code-drift.mjs.map +1 -0
- package/dist/command-line-census.d.mts +33 -0
- package/dist/command-line-census.d.mts.map +1 -0
- package/dist/command-line-census.mjs +96 -0
- package/dist/command-line-census.mjs.map +1 -0
- package/dist/compare-workers.d.mts +3 -0
- package/dist/compare-workers.d.mts.map +1 -0
- package/dist/compare-workers.mjs +332 -0
- package/dist/compare-workers.mjs.map +1 -0
- package/dist/control-plane-isolation.d.mts +45 -0
- package/dist/control-plane-isolation.d.mts.map +1 -0
- package/dist/control-plane-isolation.mjs +67 -0
- package/dist/control-plane-isolation.mjs.map +1 -0
- package/dist/deploy-worker.d.mts +3 -0
- package/dist/deploy-worker.d.mts.map +1 -0
- package/dist/deploy-worker.mjs +333 -0
- package/dist/deploy-worker.mjs.map +1 -0
- package/dist/doctor.d.mts +216 -0
- package/dist/doctor.d.mts.map +1 -0
- package/dist/doctor.mjs +962 -0
- package/dist/doctor.mjs.map +1 -0
- package/dist/fleet-consistency.d.mts +235 -0
- package/dist/fleet-consistency.d.mts.map +1 -0
- package/dist/fleet-consistency.mjs +436 -0
- package/dist/fleet-consistency.mjs.map +1 -0
- package/dist/fleet-env.d.mts +228 -0
- package/dist/fleet-env.d.mts.map +1 -0
- package/dist/fleet-env.mjs +509 -0
- package/dist/fleet-env.mjs.map +1 -0
- package/dist/fleet-scripts.d.mts +11 -0
- package/dist/fleet-scripts.d.mts.map +1 -0
- package/dist/fleet-scripts.mjs +41 -0
- package/dist/fleet-scripts.mjs.map +1 -0
- package/dist/git-safe-env.d.mts +10 -0
- package/dist/git-safe-env.d.mts.map +1 -0
- package/dist/git-safe-env.mjs +44 -0
- package/dist/git-safe-env.mjs.map +1 -0
- package/dist/guest-run.d.mts +26 -0
- package/dist/guest-run.d.mts.map +1 -0
- package/dist/guest-run.mjs +164 -0
- package/dist/guest-run.mjs.map +1 -0
- package/dist/host-address.d.mts +33 -0
- package/dist/host-address.d.mts.map +1 -0
- package/dist/host-address.mjs +105 -0
- package/dist/host-address.mjs.map +1 -0
- package/dist/host-capacity.d.mts +64 -0
- package/dist/host-capacity.d.mts.map +1 -0
- package/dist/host-capacity.mjs +152 -0
- package/dist/host-capacity.mjs.map +1 -0
- package/dist/host-metrics.d.mts +116 -0
- package/dist/host-metrics.d.mts.map +1 -0
- package/dist/host-metrics.mjs +201 -0
- package/dist/host-metrics.mjs.map +1 -0
- package/dist/index.d.ts +23 -0
- package/dist/index.d.ts.map +1 -0
- package/dist/index.js +25 -0
- package/dist/index.js.map +1 -0
- package/dist/local-vm.d.ts +125 -0
- package/dist/local-vm.d.ts.map +1 -0
- package/dist/local-vm.js +360 -0
- package/dist/local-vm.js.map +1 -0
- package/dist/measure-guard.d.mts +34 -0
- package/dist/measure-guard.d.mts.map +1 -0
- package/dist/measure-guard.mjs +73 -0
- package/dist/measure-guard.mjs.map +1 -0
- package/dist/normalise-fleet.d.mts +2 -0
- package/dist/normalise-fleet.d.mts.map +1 -0
- package/dist/normalise-fleet.mjs +76 -0
- package/dist/normalise-fleet.mjs.map +1 -0
- package/dist/npm-cli-executable.d.mts +42 -0
- package/dist/npm-cli-executable.d.mts.map +1 -0
- package/dist/npm-cli-executable.mjs +159 -0
- package/dist/npm-cli-executable.mjs.map +1 -0
- package/dist/probe-outcome.d.mts +89 -0
- package/dist/probe-outcome.d.mts.map +1 -0
- package/dist/probe-outcome.mjs +104 -0
- package/dist/probe-outcome.mjs.map +1 -0
- package/dist/protocol-guard.d.mts +34 -0
- package/dist/protocol-guard.d.mts.map +1 -0
- package/dist/protocol-guard.mjs +121 -0
- package/dist/protocol-guard.mjs.map +1 -0
- package/dist/source-walk.d.mts +12 -0
- package/dist/source-walk.d.mts.map +1 -0
- package/dist/source-walk.mjs +56 -0
- package/dist/source-walk.mjs.map +1 -0
- package/dist/transient-fault.d.mts +6 -0
- package/dist/transient-fault.d.mts.map +1 -0
- package/dist/transient-fault.mjs +86 -0
- package/dist/transient-fault.mjs.map +1 -0
- package/dist/utm-deprecated.d.mts +6 -0
- package/dist/utm-deprecated.d.mts.map +1 -0
- package/dist/utm-deprecated.mjs +23 -0
- package/dist/utm-deprecated.mjs.map +1 -0
- package/dist/worker-code-check.d.mts +29 -0
- package/dist/worker-code-check.d.mts.map +1 -0
- package/dist/worker-code-check.mjs +85 -0
- package/dist/worker-code-check.mjs.map +1 -0
- package/dist/worker-health.d.mts +56 -0
- package/dist/worker-health.d.mts.map +1 -0
- package/dist/worker-health.mjs +73 -0
- package/dist/worker-health.mjs.map +1 -0
- package/dist/worker-http.d.mts +103 -0
- package/dist/worker-http.d.mts.map +1 -0
- package/dist/worker-http.mjs +277 -0
- package/dist/worker-http.mjs.map +1 -0
- package/dist/worker-stats.d.mts +66 -0
- package/dist/worker-stats.d.mts.map +1 -0
- package/dist/worker-stats.mjs +143 -0
- package/dist/worker-stats.mjs.map +1 -0
- package/package.json +96 -4
- package/src/local-worker/autounattend.xml +280 -0
- package/src/local-worker/build-vm.sh +218 -0
- package/src/local-worker/clone-worker.sh +141 -0
- package/src/local-worker/create-utm-vm.sh +202 -0
- package/src/local-worker/fetch-windows-iso.sh +238 -0
- package/src/local-worker/first-boot.cmd +58 -0
- package/src/local-worker/worker-ctl.sh +442 -0
- package/src/provisioning/README.md +28 -0
- package/src/provisioning/apply-foreground-lock-timeout.ps1 +71 -0
- package/src/provisioning/bare-metal/README.md +213 -0
- package/src/provisioning/bare-metal/a11y-bootstrap.service +58 -0
- package/src/provisioning/bare-metal/autounattend.xml +428 -0
- package/src/provisioning/bare-metal/serve-bootstrap.sh +86 -0
- package/src/provisioning/bootstrap-control-plane.sh +463 -0
- package/src/provisioning/bootstrap-windows-worker.ps1 +649 -0
- package/src/provisioning/build-lean-worker-image.ps1 +275 -0
- package/src/provisioning/diagnose-nvda-worker.ps1 +174 -0
- package/src/provisioning/provision-nvda-worker.ps1 +827 -0
- package/src/provisioning/set-display-mode.ps1 +411 -0
- package/src/provisioning/stamp-provision-revision.ps1 +184 -0
|
@@ -0,0 +1,86 @@
|
|
|
1
|
+
// @ts-check
|
|
2
|
+
/**
|
|
3
|
+
* Is a capture-worker failure recoverable, or the end of this case?
|
|
4
|
+
*
|
|
5
|
+
* MOVED HERE from `packages/lab/src/training/capture-decisions.mjs` — architecture-audit.md §5, item 3:
|
|
6
|
+
* "the protocol version and fault codes are reached by scraping" because no shared, dependency-free home
|
|
7
|
+
* existed for classification that both the lab AND anything else speaking to a worker over HTTP need. This
|
|
8
|
+
* package already exists for exactly that ("host-side lifecycle, health and capacity for a fleet of
|
|
9
|
+
* Windows NVDA capture workers"), and `capture-client.mjs` — which needs this to decide whether a lost
|
|
10
|
+
* response is worth reconciling rather than failing outright — moved here alongside it for the same
|
|
11
|
+
* reason: `packages/cli` can depend on `@a11ign/screenreader-fleet` (it already does, for `requestJson`)
|
|
12
|
+
* but must never depend on `@a11ign/lab`, which is private and never published.
|
|
13
|
+
*
|
|
14
|
+
* `capture-decisions.mjs` re-exports `isTransient` from here so every existing lab-side importer is
|
|
15
|
+
* unchanged.
|
|
16
|
+
*/
|
|
17
|
+
// BY CODE, from the module that defines them — architecture-audit.md §5, item 3 and item 4: "fault codes
|
|
18
|
+
// are copied as string literals... because no ./capture-faults subpath is exported". `capture-faults.mjs`
|
|
19
|
+
// has no imports of its own, so it was always safe to expose; the subpath just did not exist. Reading the
|
|
20
|
+
// actual codes here means a renamed fault cannot silently stop being recognised as recoverable.
|
|
21
|
+
import { FAULT } from "@a11ign/screenreader-worker/capture-faults";
|
|
22
|
+
/**
|
|
23
|
+
* Recoverable, or the end of this case?
|
|
24
|
+
*
|
|
25
|
+
* Everything here heals on its own, which is why waiting beats failing. The connection errors are
|
|
26
|
+
* here because the first full dataset run lost its last four cases to one guest bugchecking — it came
|
|
27
|
+
* back by itself, but the run had already recorded four permanent failures.
|
|
28
|
+
*
|
|
29
|
+
* `running but not speaking` and `hard timeout` are the subtle ones: both make the worker STOP its
|
|
30
|
+
* screen reader, so the next capture cold-starts a fresh one. They are self-healing by construction,
|
|
31
|
+
* and classifying them fatal cost a case in the run that proved it.
|
|
32
|
+
*/
|
|
33
|
+
const TRANSIENT = new RegExp([
|
|
34
|
+
"fetch failed", "ECONNREFUSED", "ECONNRESET", "socket hang up", "timed out", "aborted",
|
|
35
|
+
"HTTP 429.*capture is already in progress",
|
|
36
|
+
"running but not speaking",
|
|
37
|
+
"hard timeout",
|
|
38
|
+
].join("|"), "i");
|
|
39
|
+
/**
|
|
40
|
+
* Faults the WORKER named for us, which never need matching against prose.
|
|
41
|
+
*
|
|
42
|
+
* Both self-heal: the worker stops NVDA on any failed capture, so the next attempt cold-starts a clean
|
|
43
|
+
* one. The worker now retries these itself before answering, so seeing one here means even its retry
|
|
44
|
+
* did not clear it — still worth reissuing the case rather than recording a permanent failure.
|
|
45
|
+
*/
|
|
46
|
+
const TRANSIENT_FAULTS = new Set([FAULT.SCREEN_READER_MUTE, FAULT.SCREEN_READER_START_FAILED]);
|
|
47
|
+
/**
|
|
48
|
+
* Network failures that heal on their own, by CODE rather than by wording.
|
|
49
|
+
*
|
|
50
|
+
* These became visible when the capture clients moved off `fetch` to `node:http` (see
|
|
51
|
+
* `worker-fleet/src/worker-http.mjs` for why they had to). `fetch` collapsed every network failure into
|
|
52
|
+
* `TypeError: fetch failed`, which the regex above matched — so the whole class was transient by accident,
|
|
53
|
+
* through a wrapper's wording rather than through anything we had decided.
|
|
54
|
+
*
|
|
55
|
+
* `EHOSTUNREACH` is the one that would have bitten. It is how a bare-metal worker presents while its NIC
|
|
56
|
+
* wakes from selective suspend, recorded in provision-nvda-worker.ps1: 48 instant failures in one
|
|
57
|
+
* evidence-check run, and the box answered a curl thirty seconds later. Under the real code, and without
|
|
58
|
+
* this set, that would now be classified FATAL and fail 48 cases permanently.
|
|
59
|
+
*
|
|
60
|
+
* `ETIMEDOUT` covers both a dead peer and our own deadline in `requestJson`, which is deliberate: a
|
|
61
|
+
* capture that outran its budget is exactly the case the worker recovers from by cold-starting NVDA.
|
|
62
|
+
*/
|
|
63
|
+
const TRANSIENT_NETWORK_CODES = new Set([
|
|
64
|
+
"ECONNREFUSED", "ECONNRESET", "EHOSTUNREACH", "ENETUNREACH", "ENETDOWN",
|
|
65
|
+
"EPIPE", "ETIMEDOUT", "EAI_AGAIN", "UND_ERR_HEADERS_TIMEOUT", "UND_ERR_BODY_TIMEOUT",
|
|
66
|
+
]);
|
|
67
|
+
/**
|
|
68
|
+
* @param {unknown} error anything a failed request threw — a node:http Error, an undici one, a string
|
|
69
|
+
* @returns {boolean}
|
|
70
|
+
*/
|
|
71
|
+
export function isTransient(error) {
|
|
72
|
+
const failure = /** @type {{ code?: string, cause?: { code?: string }, message?: string }} */ (error);
|
|
73
|
+
// Prefer the code. The regex below is the fallback for older workers and for host-side failures
|
|
74
|
+
// (a dropped socket has no fault code), but a message is prose and prose gets reworded — see
|
|
75
|
+
// packages/nvda-worker/src/capture-faults.mjs for what that cost.
|
|
76
|
+
if (TRANSIENT_FAULTS.has(failure?.code ?? ""))
|
|
77
|
+
return true;
|
|
78
|
+
if (TRANSIENT_NETWORK_CODES.has(failure?.code ?? ""))
|
|
79
|
+
return true;
|
|
80
|
+
// A node:http error carries its code on the error itself; an undici one hides it on `cause`. Checking
|
|
81
|
+
// both means the classification does not depend on which client the caller happened to use.
|
|
82
|
+
if (TRANSIENT_NETWORK_CODES.has(failure?.cause?.code ?? ""))
|
|
83
|
+
return true;
|
|
84
|
+
return TRANSIENT.test(String(failure?.message ?? error ?? ""));
|
|
85
|
+
}
|
|
86
|
+
//# sourceMappingURL=transient-fault.mjs.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"transient-fault.mjs","sourceRoot":"","sources":["../src/transient-fault.mjs"],"names":[],"mappings":"AAAA,YAAY;AACZ;;;;;;;;;;;;;;GAcG;AACH,yGAAyG;AACzG,0GAA0G;AAC1G,0GAA0G;AAC1G,gGAAgG;AAChG,OAAO,EAAE,KAAK,EAAE,MAAM,4CAA4C,CAAC;AAEnE;;;;;;;;;;GAUG;AACH,MAAM,SAAS,GAAG,IAAI,MAAM,CAAC;IAC3B,cAAc,EAAE,cAAc,EAAE,YAAY,EAAE,gBAAgB,EAAE,WAAW,EAAE,SAAS;IACtF,0CAA0C;IAC1C,0BAA0B;IAC1B,cAAc;CACf,CAAC,IAAI,CAAC,GAAG,CAAC,EAAE,GAAG,CAAC,CAAC;AAElB;;;;;;GAMG;AACH,MAAM,gBAAgB,GAAG,IAAI,GAAG,CAAC,CAAC,KAAK,CAAC,kBAAkB,EAAE,KAAK,CAAC,0BAA0B,CAAC,CAAC,CAAC;AAE/F;;;;;;;;;;;;;;;GAeG;AACH,MAAM,uBAAuB,GAAG,IAAI,GAAG,CAAC;IACtC,cAAc,EAAE,YAAY,EAAE,cAAc,EAAE,aAAa,EAAE,UAAU;IACvE,OAAO,EAAE,WAAW,EAAE,WAAW,EAAE,yBAAyB,EAAE,sBAAsB;CACrF,CAAC,CAAC;AAEH;;;GAGG;AACH,MAAM,UAAU,WAAW,CAAC,KAAK;IAC/B,MAAM,OAAO,GAAG,6EAA6E,CAAC,CAAC,KAAK,CAAC,CAAC;IACtG,gGAAgG;IAChG,6FAA6F;IAC7F,kEAAkE;IAClE,IAAI,gBAAgB,CAAC,GAAG,CAAC,OAAO,EAAE,IAAI,IAAI,EAAE,CAAC;QAAE,OAAO,IAAI,CAAC;IAC3D,IAAI,uBAAuB,CAAC,GAAG,CAAC,OAAO,EAAE,IAAI,IAAI,EAAE,CAAC;QAAE,OAAO,IAAI,CAAC;IAClE,sGAAsG;IACtG,4FAA4F;IAC5F,IAAI,uBAAuB,CAAC,GAAG,CAAC,OAAO,EAAE,KAAK,EAAE,IAAI,IAAI,EAAE,CAAC;QAAE,OAAO,IAAI,CAAC;IACzE,OAAO,SAAS,CAAC,IAAI,CAAC,MAAM,CAAC,OAAO,EAAE,OAAO,IAAI,KAAK,IAAI,EAAE,CAAC,CAAC,CAAC;AACjE,CAAC"}
|
|
@@ -0,0 +1,6 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* @param {string} what The command or module the caller is about to run — named so the message is
|
|
3
|
+
* specific to what actually fired, not a generic banner every UTM-adjacent file prints identically.
|
|
4
|
+
*/
|
|
5
|
+
export function warnUtmDeprecated(what: string): void;
|
|
6
|
+
//# sourceMappingURL=utm-deprecated.d.mts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"utm-deprecated.d.mts","sourceRoot":"","sources":["../src/utm-deprecated.mjs"],"names":[],"mappings":"AAcA;;;GAGG;AACH,wCAHW,MAAM,QAShB"}
|
|
@@ -0,0 +1,23 @@
|
|
|
1
|
+
// @ts-check
|
|
2
|
+
// One deprecation notice, said the same way everywhere it applies.
|
|
3
|
+
//
|
|
4
|
+
// architecture-audit.md §8: "the deprecated UTM path is still the CLI's default" — ~2,460 lines of
|
|
5
|
+
// UTM-only code still ship, and until this file existed, using any of them said nothing about it. The
|
|
6
|
+
// repository owner stated plainly on 2026-09-05: "The UTM is deprecated, that was a testing thing." This
|
|
7
|
+
// is the loud half CLAUDE.md's own rule calls for: "a deprecated path that is still the first one
|
|
8
|
+
// documented is not deprecated" — printing nothing at the point of use is the runtime version of that
|
|
9
|
+
// same mistake.
|
|
10
|
+
//
|
|
11
|
+
// A notice, not a refusal: some of these entry points are still how an existing UTM guest is reached
|
|
12
|
+
// (worker-ctl.sh status, for instance), and this file does not decide which are safe to keep working —
|
|
13
|
+
// that is a separate proposal. This only makes sure nobody reaches any of them without being told.
|
|
14
|
+
/**
|
|
15
|
+
* @param {string} what The command or module the caller is about to run — named so the message is
|
|
16
|
+
* specific to what actually fired, not a generic banner every UTM-adjacent file prints identically.
|
|
17
|
+
*/
|
|
18
|
+
export function warnUtmDeprecated(what) {
|
|
19
|
+
process.stderr.write(`DEPRECATED: ${what} manages a local UTM worker VM. UTM was a testing path and is not the fleet.\n` +
|
|
20
|
+
"Capture on the bare-metal fleet instead: npm run fleet:status, npm run fleet:deploy. See CLAUDE.md's\n" +
|
|
21
|
+
'"Working on a Mac" section.\n');
|
|
22
|
+
}
|
|
23
|
+
//# sourceMappingURL=utm-deprecated.mjs.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"utm-deprecated.mjs","sourceRoot":"","sources":["../src/utm-deprecated.mjs"],"names":[],"mappings":"AAAA,YAAY;AACZ,mEAAmE;AACnE,EAAE;AACF,mGAAmG;AACnG,sGAAsG;AACtG,yGAAyG;AACzG,kGAAkG;AAClG,sGAAsG;AACtG,gBAAgB;AAChB,EAAE;AACF,qGAAqG;AACrG,uGAAuG;AACvG,mGAAmG;AAEnG;;;GAGG;AACH,MAAM,UAAU,iBAAiB,CAAC,IAAI;IACpC,OAAO,CAAC,MAAM,CAAC,KAAK,CAClB,eAAe,IAAI,gFAAgF;QACnG,wGAAwG;QACxG,+BAA+B,CAChC,CAAC;AACJ,CAAC"}
|
|
@@ -0,0 +1,29 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Refuse to capture with a fleet that is not running this checkout.
|
|
3
|
+
*
|
|
4
|
+
* Called at the boundary of every capture entry point, for the reason `assertWorkerUrl` is: the
|
|
5
|
+
* alternative is discovering it in the evidence weeks later, where a stale worker looks like a page that
|
|
6
|
+
* changed. **Both entry points, not one** — a remedy that reaches one of several paths is the shape this
|
|
7
|
+
* repo has paid for three times over (`anchorToTop`, `ensureSpeechChannel`, `waitForAnnouncement`), and
|
|
8
|
+
* `capture-preflight.test.ts` pins that both call it.
|
|
9
|
+
*
|
|
10
|
+
* A thin wrapper over `assertWorkersServe`, supplying the one thing only this file can compute: the hash.
|
|
11
|
+
*
|
|
12
|
+
* @param {string[]} workers
|
|
13
|
+
* @param {{when?: string, allow?: boolean, read?: (url: string) => Promise<string|null>, bareMetalUrls?: string[]}} options
|
|
14
|
+
*/
|
|
15
|
+
export function assertFleetRunsThisCheckout(workers: string[], options?: {
|
|
16
|
+
when?: string;
|
|
17
|
+
allow?: boolean;
|
|
18
|
+
read?: (url: string) => Promise<string | null>;
|
|
19
|
+
bareMetalUrls?: string[];
|
|
20
|
+
}): Promise<void>;
|
|
21
|
+
export function expectedWorkerCode(): string;
|
|
22
|
+
import { codeDrift } from "./code-drift.mjs";
|
|
23
|
+
import { describeCodeDrift } from "./code-drift.mjs";
|
|
24
|
+
import { describeEmptyPool } from "./code-drift.mjs";
|
|
25
|
+
import { readWorkerCode } from "./code-drift.mjs";
|
|
26
|
+
import { remedyLines } from "./code-drift.mjs";
|
|
27
|
+
import { workerSourceDirty } from "./code-drift.mjs";
|
|
28
|
+
export { codeDrift, describeCodeDrift, describeEmptyPool, readWorkerCode, remedyLines, workerSourceDirty };
|
|
29
|
+
//# sourceMappingURL=worker-code-check.d.mts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"worker-code-check.d.mts","sourceRoot":"","sources":["../src/worker-code-check.mjs"],"names":[],"mappings":"AAwEA;;;;;;;;;;;;;GAaG;AACH,qDAHW,MAAM,EAAE,YACR;IAAC,IAAI,CAAC,EAAE,MAAM,CAAC;IAAC,KAAK,CAAC,EAAE,OAAO,CAAC;IAAC,IAAI,CAAC,EAAE,CAAC,GAAG,EAAE,MAAM,KAAK,OAAO,CAAC,MAAM,GAAC,IAAI,CAAC,CAAC;IAAC,aAAa,CAAC,EAAE,MAAM,EAAE,CAAA;CAAC,iBAIlH;AAvBM,6CAA8C;0BAjBN,kBAAkB;kCAAlB,kBAAkB;kCAAlB,kBAAkB;+BAAlB,kBAAkB;4BAAlB,kBAAkB;kCAAlB,kBAAkB"}
|
|
@@ -0,0 +1,85 @@
|
|
|
1
|
+
// @ts-check
|
|
2
|
+
/**
|
|
3
|
+
* Is the fleet running the code this checkout expects — asked BEFORE a capture run, not after it.
|
|
4
|
+
*
|
|
5
|
+
* ## The hole this closes
|
|
6
|
+
*
|
|
7
|
+
* `run-job.yml` refuses to run at a commit other than the one asked for, and the comment above that
|
|
8
|
+
* refusal says why: *"a job that quietly runs four commits behind reports success for code you did not
|
|
9
|
+
* ask for."* That guard covers the LAB. It says nothing about the twelve machines that actually take the
|
|
10
|
+
* captures, and those are a second checkout, deployed by a separate command nobody is forced to run.
|
|
11
|
+
*
|
|
12
|
+
* So a capture run could be dispatched at the right commit, on a lab that proved it was at the right
|
|
13
|
+
* commit, and still capture with the PREVIOUS release of `capture-core.mjs`. Measured on 2026-08-25: after
|
|
14
|
+
* `MAX_TAB_STOPS` went 12 -> 150 and `collectByType` started recording `prevCount`, the real-page corpus
|
|
15
|
+
* held both populations at once, and the only way to read it was to bucket captures by whether they
|
|
16
|
+
* carried the new diagnostic mark at all. The evidence was mixed, the run reported success, and the
|
|
17
|
+
* separation had to be done by hand afterwards.
|
|
18
|
+
*
|
|
19
|
+
* `npm run worker:code` has answered this question correctly the whole time. It is a separate command a
|
|
20
|
+
* human must remember, which is this repo's own definition of a check that does not happen — and it was
|
|
21
|
+
* remembered by hand four times in one day before this existed.
|
|
22
|
+
*
|
|
23
|
+
* ## Why a REFUSAL, and why on any difference at all
|
|
24
|
+
*
|
|
25
|
+
* `workerCode` is deliberately outside the capture cache key ("it changes when a comment changes, and
|
|
26
|
+
* invalidating the WHOLE corpus over a reworded comment is how a cache becomes something people turn
|
|
27
|
+
* off") and deliberately outside
|
|
28
|
+
* `fleet-consistency.mjs`'s `MUST_MATCH` for the same reason. Both of those are the right call for
|
|
29
|
+
* the questions they answer — *is this evidence still valid* and *are these guests interchangeable*.
|
|
30
|
+
*
|
|
31
|
+
* This is a third question with a different answer: *am I about to capture with the code I asked for*. A
|
|
32
|
+
* comment-only drift is a false alarm here and it costs one `fleet:deploy`; a real drift costs a corpus and
|
|
33
|
+
* is invisible, because nothing downstream keys on `workerCode`. That asymmetry is the whole argument.
|
|
34
|
+
*
|
|
35
|
+
* It is a PRECONDITION and never a key: nothing here invalidates a cached capture.
|
|
36
|
+
*
|
|
37
|
+
* ## The comparison itself lives in `code-drift.mjs`, and this file is the reason for the split
|
|
38
|
+
*
|
|
39
|
+
* `expectedWorkerCode` below needs `codeVersion`/`workerSourceDir`, reached through a SUBPATH export
|
|
40
|
+
* (`@a11ign/screenreader-worker/code-version`) rather than a relative path — a relative one drags
|
|
41
|
+
* `nvda-worker`'s `.mjs` files into this package's own tsc project and the build dies with TS5055 ("would
|
|
42
|
+
* overwrite input file"). That subpath resolves through `node_modules`, which is exactly what
|
|
43
|
+
* `packages/control` does not have (ADR 0012) — so when `lab-job.mjs` needed this same comparison BEFORE
|
|
44
|
+
* dispatching to the lab, it could not import this file. `code-drift.mjs` is the part of this file with no
|
|
45
|
+
* opinion about what "expected" means: it takes the hash as a parameter, imports nothing but
|
|
46
|
+
* `node:child_process`, and is safe from both places. This file supplies the one thing only it can compute.
|
|
47
|
+
*/
|
|
48
|
+
import { codeDrift, describeCodeDrift, describeEmptyPool, readWorkerCode, remedyLines, workerSourceDirty, assertWorkersServe } from "./code-drift.mjs";
|
|
49
|
+
// A SUBPATH export, not a deep relative path: `../../nvda-worker/src/...` drags those .mjs files into
|
|
50
|
+
// worker-fleet's tsc project and the build dies with TS5055 "would overwrite input file". The subpath is
|
|
51
|
+
// also the shape already in use for the same reason -- `@a11ign/screenreader-fleet/worker-http`.
|
|
52
|
+
// `code-version.mjs` imports nothing but node stdlib and `worker-files.mjs`, which is why it is safe and
|
|
53
|
+
// why it is its own module. Still the ONE hasher: the subpath is the same function.
|
|
54
|
+
import { codeVersion, workerSourceDir } from "@a11ign/screenreader-worker/code-version";
|
|
55
|
+
/**
|
|
56
|
+
* The hash every worker is expected to be serving: the one the installed `@a11ign/screenreader-worker` was released with.
|
|
57
|
+
*
|
|
58
|
+
* NO ARGUMENT, and that is the point (a11ign/a11ign#3740). The package is BUILT, so `workerSourceDir()` is its `dist/`, which holds
|
|
59
|
+
* none of the files a guest runs (a guest runs the raw `src/*.mjs`, ADR 0031) and `codeVersion(workerSourceDir())` threw ENOENT.
|
|
60
|
+
* Built, `codeVersion()` returns the hash of the `src` the package was built from, which is what `/health.code` is over. Raw (0.1.0,
|
|
61
|
+
* which ships `src`), the same call hashes the module's own directory, which is the source: the one expression is right for both.
|
|
62
|
+
*/
|
|
63
|
+
export const expectedWorkerCode = () => codeVersion();
|
|
64
|
+
// Re-exported rather than duplicated: existing callers (`capture-real-pages.mjs`,
|
|
65
|
+
// `capture-screenreader-dataset.mjs`, and this module's own test) import these from here, and moving their
|
|
66
|
+
// implementation to `code-drift.mjs` must not become a second place either has to be found.
|
|
67
|
+
export { codeDrift, describeCodeDrift, describeEmptyPool, readWorkerCode, remedyLines, workerSourceDirty };
|
|
68
|
+
/**
|
|
69
|
+
* Refuse to capture with a fleet that is not running this checkout.
|
|
70
|
+
*
|
|
71
|
+
* Called at the boundary of every capture entry point, for the reason `assertWorkerUrl` is: the
|
|
72
|
+
* alternative is discovering it in the evidence weeks later, where a stale worker looks like a page that
|
|
73
|
+
* changed. **Both entry points, not one** — a remedy that reaches one of several paths is the shape this
|
|
74
|
+
* repo has paid for three times over (`anchorToTop`, `ensureSpeechChannel`, `waitForAnnouncement`), and
|
|
75
|
+
* `capture-preflight.test.ts` pins that both call it.
|
|
76
|
+
*
|
|
77
|
+
* A thin wrapper over `assertWorkersServe`, supplying the one thing only this file can compute: the hash.
|
|
78
|
+
*
|
|
79
|
+
* @param {string[]} workers
|
|
80
|
+
* @param {{when?: string, allow?: boolean, read?: (url: string) => Promise<string|null>, bareMetalUrls?: string[]}} options
|
|
81
|
+
*/
|
|
82
|
+
export async function assertFleetRunsThisCheckout(workers, options = {}) {
|
|
83
|
+
return assertWorkersServe(expectedWorkerCode(), workers, { ...options, sourceDir: workerSourceDir() });
|
|
84
|
+
}
|
|
85
|
+
//# sourceMappingURL=worker-code-check.mjs.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"worker-code-check.mjs","sourceRoot":"","sources":["../src/worker-code-check.mjs"],"names":[],"mappings":"AAAA,YAAY;AACZ;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;GA6CG;AACH,OAAO,EAAE,SAAS,EAAE,iBAAiB,EAAE,iBAAiB,EAAE,cAAc,EAAE,WAAW,EACnF,iBAAiB,EAAE,kBAAkB,EAAE,MAAM,kBAAkB,CAAC;AAElE,sGAAsG;AACtG,yGAAyG;AACzG,iGAAiG;AACjG,yGAAyG;AACzG,oFAAoF;AACpF,OAAO,EAAE,WAAW,EAAE,eAAe,EAAE,MAAM,0CAA0C,CAAC;AAExF;;;;;;;GAOG;AACH,MAAM,CAAC,MAAM,kBAAkB,GAAG,GAAG,EAAE,CAAC,WAAW,EAAE,CAAC;AAEtD,kFAAkF;AAClF,2GAA2G;AAC3G,4FAA4F;AAC5F,OAAO,EAAE,SAAS,EAAE,iBAAiB,EAAE,iBAAiB,EAAE,cAAc,EAAE,WAAW,EAAE,iBAAiB,EAAE,CAAC;AAE3G;;;;;;;;;;;;;GAaG;AACH,MAAM,CAAC,KAAK,UAAU,2BAA2B,CAAC,OAAO,EAAE,OAAO,GAAG,EAAE;IACrE,OAAO,kBAAkB,CAAC,kBAAkB,EAAE,EAAE,OAAO,EAAE,EAAE,GAAG,OAAO,EAAE,SAAS,EAAE,eAAe,EAAE,EAAE,CAAC,CAAC;AACzG,CAAC"}
|
|
@@ -0,0 +1,56 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Is a worker healthy, degraded, or unusable — from the vitals it reports.
|
|
3
|
+
*
|
|
4
|
+
* This exists because a guest whose NVDA was broken on **every single capture** sat in the pool at four
|
|
5
|
+
* times the cost of its neighbours and nothing noticed. Measured, same page, same code, same moment:
|
|
6
|
+
*
|
|
7
|
+
* worker 1 4 captures, 4 recoveries, 0 failures nvdaStart 19.1s/capture WALL 122.9s
|
|
8
|
+
* worker 2 9 captures, 0 recoveries, 0 failures nvdaStart 0.0s/capture WALL 40.6s
|
|
9
|
+
*
|
|
10
|
+
* Two things conspired to hide it. The worker's own retry absorbed every fault, so `failures` stayed 0
|
|
11
|
+
* and the run's eviction rule — three consecutive FAILURES — could never fire. And wall-clock time only
|
|
12
|
+
* said "slower", which I twice misattributed to Edge.
|
|
13
|
+
*
|
|
14
|
+
* So degradation is defined on the recovery RATE, not on failures. `recoveries` counts faults the worker
|
|
15
|
+
* papered over for the caller, which makes it the one number that rises while everything still appears
|
|
16
|
+
* to work.
|
|
17
|
+
*
|
|
18
|
+
* **Degraded workers keep taking work.** They are slow, not broken, and pulling one from a three-VM pool
|
|
19
|
+
* costs more throughput than it saves. This mirrors the standard health-check split — a degraded service
|
|
20
|
+
* returns 200 and is *surfaced* rather than restarted, because declaring degraded things unhealthy is how
|
|
21
|
+
* you end up with nothing left to serve (Distributed Systems with Node.js, ch. 4).
|
|
22
|
+
*/
|
|
23
|
+
/**
|
|
24
|
+
* Can this worker take a capture right now? ONE definition, because there were four that disagreed.
|
|
25
|
+
*
|
|
26
|
+
* `ready` is about the ENVIRONMENT — Edge resolvable, ForegroundLockTimeout 0, the worker free — and a
|
|
27
|
+
* worker reports `ready: false` while NVDA warms up after a boot, which is normal and self-correcting.
|
|
28
|
+
*
|
|
29
|
+
* The subtlety, and the reason this is `!== false` rather than `=== true`: **a worker predating the
|
|
30
|
+
* field reports neither.** Treating absent as ready keeps an un-redeployed guest working instead of
|
|
31
|
+
* stalling a run against it forever; staleness has its own detector in `npm run worker:code`. The
|
|
32
|
+
* dataset runner got this right and said so. `repeat-capture.mjs` tested `health.ready` for truthiness
|
|
33
|
+
* and `capture-real-pages.mjs` tested `=== true`, so both would have waited out their whole readiness
|
|
34
|
+
* budget against a perfectly good older worker and then blamed the page.
|
|
35
|
+
*
|
|
36
|
+
* @param {{ busy?: boolean, ready?: boolean } | null | undefined} health
|
|
37
|
+
* @returns {boolean}
|
|
38
|
+
*/
|
|
39
|
+
export function workerIsUsable(health: {
|
|
40
|
+
busy?: boolean;
|
|
41
|
+
ready?: boolean;
|
|
42
|
+
} | null | undefined): boolean;
|
|
43
|
+
/**
|
|
44
|
+
* @param {{ captures?: number, recoveries?: number, failures?: number } | null | undefined} vitals
|
|
45
|
+
* @returns {{ degraded: boolean, reason: string | null, recoveryShare: number | null }}
|
|
46
|
+
*/
|
|
47
|
+
export function assessWorker(vitals: {
|
|
48
|
+
captures?: number;
|
|
49
|
+
recoveries?: number;
|
|
50
|
+
failures?: number;
|
|
51
|
+
} | null | undefined): {
|
|
52
|
+
degraded: boolean;
|
|
53
|
+
reason: string | null;
|
|
54
|
+
recoveryShare: number | null;
|
|
55
|
+
};
|
|
56
|
+
//# sourceMappingURL=worker-health.d.mts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"worker-health.d.mts","sourceRoot":"","sources":["../src/worker-health.mjs"],"names":[],"mappings":"AACA;;;;;;;;;;;;;;;;;;;;;GAqBG;AAEH;;;;;;;;;;;;;;;GAeG;AACH,uCAHW;IAAE,IAAI,CAAC,EAAE,OAAO,CAAC;IAAC,KAAK,CAAC,EAAE,OAAO,CAAA;CAAE,GAAG,IAAI,GAAG,SAAS,GACpD,OAAO,CAKnB;AAQD;;;GAGG;AACH,qCAHW;IAAE,QAAQ,CAAC,EAAE,MAAM,CAAC;IAAC,UAAU,CAAC,EAAE,MAAM,CAAC;IAAC,QAAQ,CAAC,EAAE,MAAM,CAAA;CAAE,GAAG,IAAI,GAAG,SAAS,GAC9E;IAAE,QAAQ,EAAE,OAAO,CAAC;IAAC,MAAM,EAAE,MAAM,GAAG,IAAI,CAAC;IAAC,aAAa,EAAE,MAAM,GAAG,IAAI,CAAA;CAAE,CAqBtF"}
|
|
@@ -0,0 +1,73 @@
|
|
|
1
|
+
// @ts-check
|
|
2
|
+
/**
|
|
3
|
+
* Is a worker healthy, degraded, or unusable — from the vitals it reports.
|
|
4
|
+
*
|
|
5
|
+
* This exists because a guest whose NVDA was broken on **every single capture** sat in the pool at four
|
|
6
|
+
* times the cost of its neighbours and nothing noticed. Measured, same page, same code, same moment:
|
|
7
|
+
*
|
|
8
|
+
* worker 1 4 captures, 4 recoveries, 0 failures nvdaStart 19.1s/capture WALL 122.9s
|
|
9
|
+
* worker 2 9 captures, 0 recoveries, 0 failures nvdaStart 0.0s/capture WALL 40.6s
|
|
10
|
+
*
|
|
11
|
+
* Two things conspired to hide it. The worker's own retry absorbed every fault, so `failures` stayed 0
|
|
12
|
+
* and the run's eviction rule — three consecutive FAILURES — could never fire. And wall-clock time only
|
|
13
|
+
* said "slower", which I twice misattributed to Edge.
|
|
14
|
+
*
|
|
15
|
+
* So degradation is defined on the recovery RATE, not on failures. `recoveries` counts faults the worker
|
|
16
|
+
* papered over for the caller, which makes it the one number that rises while everything still appears
|
|
17
|
+
* to work.
|
|
18
|
+
*
|
|
19
|
+
* **Degraded workers keep taking work.** They are slow, not broken, and pulling one from a three-VM pool
|
|
20
|
+
* costs more throughput than it saves. This mirrors the standard health-check split — a degraded service
|
|
21
|
+
* returns 200 and is *surfaced* rather than restarted, because declaring degraded things unhealthy is how
|
|
22
|
+
* you end up with nothing left to serve (Distributed Systems with Node.js, ch. 4).
|
|
23
|
+
*/
|
|
24
|
+
/**
|
|
25
|
+
* Can this worker take a capture right now? ONE definition, because there were four that disagreed.
|
|
26
|
+
*
|
|
27
|
+
* `ready` is about the ENVIRONMENT — Edge resolvable, ForegroundLockTimeout 0, the worker free — and a
|
|
28
|
+
* worker reports `ready: false` while NVDA warms up after a boot, which is normal and self-correcting.
|
|
29
|
+
*
|
|
30
|
+
* The subtlety, and the reason this is `!== false` rather than `=== true`: **a worker predating the
|
|
31
|
+
* field reports neither.** Treating absent as ready keeps an un-redeployed guest working instead of
|
|
32
|
+
* stalling a run against it forever; staleness has its own detector in `npm run worker:code`. The
|
|
33
|
+
* dataset runner got this right and said so. `repeat-capture.mjs` tested `health.ready` for truthiness
|
|
34
|
+
* and `capture-real-pages.mjs` tested `=== true`, so both would have waited out their whole readiness
|
|
35
|
+
* budget against a perfectly good older worker and then blamed the page.
|
|
36
|
+
*
|
|
37
|
+
* @param {{ busy?: boolean, ready?: boolean } | null | undefined} health
|
|
38
|
+
* @returns {boolean}
|
|
39
|
+
*/
|
|
40
|
+
export function workerIsUsable(health) {
|
|
41
|
+
if (!health)
|
|
42
|
+
return false;
|
|
43
|
+
return !health.busy && health.ready !== false;
|
|
44
|
+
}
|
|
45
|
+
/** Below this many captures the rate is noise: one recovery out of one capture is not a pattern. */
|
|
46
|
+
const MIN_CAPTURES_TO_JUDGE = 4;
|
|
47
|
+
/** Above this share of captures needing a recovery, the guest is not merely unlucky. */
|
|
48
|
+
const DEGRADED_RECOVERY_SHARE = 0.5;
|
|
49
|
+
/**
|
|
50
|
+
* @param {{ captures?: number, recoveries?: number, failures?: number } | null | undefined} vitals
|
|
51
|
+
* @returns {{ degraded: boolean, reason: string | null, recoveryShare: number | null }}
|
|
52
|
+
*/
|
|
53
|
+
export function assessWorker(vitals) {
|
|
54
|
+
const captures = vitals?.captures ?? 0;
|
|
55
|
+
const recoveries = vitals?.recoveries ?? 0;
|
|
56
|
+
// Recoveries are counted per capture served, so the share can exceed nothing sensible above 1.
|
|
57
|
+
const attempted = captures + (vitals?.failures ?? 0);
|
|
58
|
+
if (attempted < MIN_CAPTURES_TO_JUDGE) {
|
|
59
|
+
return { degraded: false, reason: null, recoveryShare: null };
|
|
60
|
+
}
|
|
61
|
+
const recoveryShare = recoveries / attempted;
|
|
62
|
+
if (recoveryShare <= DEGRADED_RECOVERY_SHARE) {
|
|
63
|
+
return { degraded: false, reason: null, recoveryShare };
|
|
64
|
+
}
|
|
65
|
+
return {
|
|
66
|
+
degraded: true,
|
|
67
|
+
recoveryShare,
|
|
68
|
+
reason: `${recoveries} of ${attempted} captures needed a screen-reader recovery ` +
|
|
69
|
+
`(${Math.round(recoveryShare * 100)}%) — this guest's NVDA is failing and every capture pays for it. ` +
|
|
70
|
+
"Reinstall NVDA or re-provision it (docs/nvda-worker-runbook.md); it is still serving, just slowly.",
|
|
71
|
+
};
|
|
72
|
+
}
|
|
73
|
+
//# sourceMappingURL=worker-health.mjs.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"worker-health.mjs","sourceRoot":"","sources":["../src/worker-health.mjs"],"names":[],"mappings":"AAAA,YAAY;AACZ;;;;;;;;;;;;;;;;;;;;;GAqBG;AAEH;;;;;;;;;;;;;;;GAeG;AACH,MAAM,UAAU,cAAc,CAAC,MAAM;IACnC,IAAI,CAAC,MAAM;QAAE,OAAO,KAAK,CAAC;IAC1B,OAAO,CAAC,MAAM,CAAC,IAAI,IAAI,MAAM,CAAC,KAAK,KAAK,KAAK,CAAC;AAChD,CAAC;AAED,oGAAoG;AACpG,MAAM,qBAAqB,GAAG,CAAC,CAAC;AAEhC,wFAAwF;AACxF,MAAM,uBAAuB,GAAG,GAAG,CAAC;AAEpC;;;GAGG;AACH,MAAM,UAAU,YAAY,CAAC,MAAM;IACjC,MAAM,QAAQ,GAAG,MAAM,EAAE,QAAQ,IAAI,CAAC,CAAC;IACvC,MAAM,UAAU,GAAG,MAAM,EAAE,UAAU,IAAI,CAAC,CAAC;IAC3C,+FAA+F;IAC/F,MAAM,SAAS,GAAG,QAAQ,GAAG,CAAC,MAAM,EAAE,QAAQ,IAAI,CAAC,CAAC,CAAC;IACrD,IAAI,SAAS,GAAG,qBAAqB,EAAE,CAAC;QACtC,OAAO,EAAE,QAAQ,EAAE,KAAK,EAAE,MAAM,EAAE,IAAI,EAAE,aAAa,EAAE,IAAI,EAAE,CAAC;IAChE,CAAC;IACD,MAAM,aAAa,GAAG,UAAU,GAAG,SAAS,CAAC;IAC7C,IAAI,aAAa,IAAI,uBAAuB,EAAE,CAAC;QAC7C,OAAO,EAAE,QAAQ,EAAE,KAAK,EAAE,MAAM,EAAE,IAAI,EAAE,aAAa,EAAE,CAAC;IAC1D,CAAC;IACD,OAAO;QACL,QAAQ,EAAE,IAAI;QACd,aAAa;QACb,MAAM,EAAE,GAAG,UAAU,OAAO,SAAS,4CAA4C;YAC/E,IAAI,IAAI,CAAC,KAAK,CAAC,aAAa,GAAG,GAAG,CAAC,mEAAmE;YACtG,oGAAoG;KACvG,CAAC;AACJ,CAAC"}
|
|
@@ -0,0 +1,103 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* A worker address, validated at the BOUNDARY where it enters the program.
|
|
3
|
+
*
|
|
4
|
+
* `requestJson` already calls `new URL(url)`, which throws `ERR_INVALID_URL` on a malformed address — so an
|
|
5
|
+
* empty host dies in under a second, in principle. In practice it did not, and the way it did not is the
|
|
6
|
+
* reason this function exists.
|
|
7
|
+
*
|
|
8
|
+
* `--worker=http://:8765` reached `capture-real-pages.mjs` because nothing there did more than check the
|
|
9
|
+
* value was truthy, and `http://:8765` is truthy. The readiness loop then caught the resulting
|
|
10
|
+
* `ERR_INVALID_URL` in a bare `catch` whose only content was the comment "mid-boot or mid-restart; keep
|
|
11
|
+
* waiting", and so classified a permanent
|
|
12
|
+
* programmer error as a transient network condition: 60 attempts, 5 s apart, per page — then recorded
|
|
13
|
+
* "worker never became ready" as a failure of the PAGE. Four shards spent 29 minutes that way while every
|
|
14
|
+
* worker sat idle, and the run blamed the corpus.
|
|
15
|
+
*
|
|
16
|
+
* So the fix is two-part and both halves are needed: refuse the value here, and stop the readiness loop
|
|
17
|
+
* swallowing what it cannot recover from. Validating without fixing the catch leaves the next unrecoverable
|
|
18
|
+
* error to be absorbed the same way.
|
|
19
|
+
*
|
|
20
|
+
* Node's URL parser does the work. There is no regex here on purpose — a hand-rolled one would accept
|
|
21
|
+
* `http://:8765` again, since the only thing wrong with it is an empty host. Note `http:/x` and
|
|
22
|
+
* `http:///path` DO parse, to host `x` and host `path`; that is the parser's business and not something to
|
|
23
|
+
* second-guess here.
|
|
24
|
+
*
|
|
25
|
+
* @param {string | null | undefined} value the raw `--worker=` or `A11Y_WORKER` value
|
|
26
|
+
* @param {{ source?: string }} [options] what to name in the error, e.g. "--worker"
|
|
27
|
+
* @returns {string} the address, trailing slash removed
|
|
28
|
+
*/
|
|
29
|
+
export function assertWorkerUrl(value: string | null | undefined, { source }?: {
|
|
30
|
+
source?: string;
|
|
31
|
+
}): string;
|
|
32
|
+
/**
|
|
33
|
+
* One request, with a single deadline covering connect, headers and body.
|
|
34
|
+
*
|
|
35
|
+
* @param {string} url
|
|
36
|
+
* @param {{ method?: string, body?: unknown, timeoutMs?: number }} [options]
|
|
37
|
+
* @returns {Promise<{ status: number, ok: boolean, text: string, json: any }>}
|
|
38
|
+
* `json: any`, not `unknown`. This is a JSON body off the wire, and every caller reads named fields
|
|
39
|
+
* from it -- `body.error`, `body.fault`, `health.busy`, `body.transcript`. `unknown` makes each of
|
|
40
|
+
* those a cast, and a cast written to satisfy a checker asserts a shape nobody verified, which is
|
|
41
|
+
* strictly worse than saying the value is untyped. The SHAPE that matters is checked where it is
|
|
42
|
+
* defined: `capture-core`'s `Capture` typedef, and the worker's own `/health` contract.
|
|
43
|
+
*/
|
|
44
|
+
export function requestJson(url: string, { method, body, timeoutMs }?: {
|
|
45
|
+
method?: string;
|
|
46
|
+
body?: unknown;
|
|
47
|
+
timeoutMs?: number;
|
|
48
|
+
}): Promise<{
|
|
49
|
+
status: number;
|
|
50
|
+
ok: boolean;
|
|
51
|
+
text: string;
|
|
52
|
+
json: any;
|
|
53
|
+
}>;
|
|
54
|
+
/**
|
|
55
|
+
* How long a CLIENT should wait for a capture. One definition, because five had drifted.
|
|
56
|
+
*
|
|
57
|
+
* It must exceed the worker's own hard timeout (`CAPTURE_HARD_TIMEOUT_DEFAULT_MS`, 520 s) or the client
|
|
58
|
+
* gives up first, and a capture the worker would have completed is reported as a client failure. Five
|
|
59
|
+
* clients sat at 300 s -- `compare-workers`, `bench-capture`, `evidence-check`, `repeat-capture` and
|
|
60
|
+
* `capture-real-pages` -- against that 520 s. On the generated corpus nothing noticed, because a 1,338-byte
|
|
61
|
+
* page finishes in seconds. On REAL pages it silently dropped whatever used its budget, biasing the
|
|
62
|
+
* real-page corpus toward small simple pages: precisely the axis that corpus exists to add.
|
|
63
|
+
*
|
|
64
|
+
* 560,000 -> 620,000 on architecture-audit.md §14.5: `runCapture` (server.mjs) spends up to
|
|
65
|
+
* `DESKTOP_PREPARE_TIMEOUT_MS` (60 s) clearing the desktop BEFORE the hard-timeout-wrapped capture attempt
|
|
66
|
+
* even starts, sequentially rather than overlapping it -- so the true worst case a worker can legitimately
|
|
67
|
+
* take is 60 s + 520 s = 580 s, not 520 s alone. The old 560 s ceiling sat BELOW that, so a real page that
|
|
68
|
+
* used the full prepare budget and the full capture budget was killed by the CLIENT first and reported as
|
|
69
|
+
* a client failure for work the worker would have finished -- the exact shape this constant already exists
|
|
70
|
+
* to prevent, one rung further out. 620,000 keeps the same 40 s margin above the new true worst case that
|
|
71
|
+
* the original 560,000 kept above 520,000. `budget-ladder.test.ts` asserts the full sequence, not only the
|
|
72
|
+
* capture attempt inside it.
|
|
73
|
+
*
|
|
74
|
+
* Deliberately NOT imported from `@a11ign/screenreader-worker`: this package runs on macOS and Linux and must
|
|
75
|
+
* not depend on a win32-only one. `budget-ladder.test.ts` enforces the relationship instead, over every
|
|
76
|
+
* client it DISCOVERS rather than a list -- which is how the 300 s clients stayed invisible while a guard
|
|
77
|
+
* for exactly this existed and read one hardcoded path.
|
|
78
|
+
*
|
|
79
|
+
* `DATASET_CAPTURE_TIMEOUT_MS` still overrides it in the dataset runner, which is the only client that
|
|
80
|
+
* wants a per-run ceiling.
|
|
81
|
+
*/
|
|
82
|
+
export const CAPTURE_CLIENT_TIMEOUT_MS: 620000;
|
|
83
|
+
/**
|
|
84
|
+
* How long a capture's silent connection may idle before the OS proves it is still there.
|
|
85
|
+
*
|
|
86
|
+
* 15 s, and RAISING IT TO 60 s WAS TRIED AND REVERTED — by this file's own test, which refuses a delay
|
|
87
|
+
* above ~30 s because common NAT idle timeouts start there.
|
|
88
|
+
*
|
|
89
|
+
* The reasoning for raising it was that the async path removed the long-lived connection, so nothing needs
|
|
90
|
+
* an aggressive value. That is true of the async path and FALSE of the `A11Y_SYNC_CAPTURE` escape hatch,
|
|
91
|
+
* which still holds one socket silent for a whole capture. At 60 s the first probe would fire after the
|
|
92
|
+
* reap, so the hatch would be unprotected while the constant looked deliberate — a guard weakened for a
|
|
93
|
+
* reason that did not cover the case it exists for.
|
|
94
|
+
*
|
|
95
|
+
* It stays aggressive until item D explains why THIS path reaps in seconds when the literature says
|
|
96
|
+
* minutes. An unexplained number is not a solved problem.
|
|
97
|
+
*
|
|
98
|
+
* EXPORTED so its test can key on the exact value. Node's own HTTP SERVER calls `setKeepAlive(true, 5000)`
|
|
99
|
+
* on every socket it accepts, so a test that merely looked for "keepalive with a plausible delay" matched
|
|
100
|
+
* the server's call and passed with this hook DELETED — found by mutation, not by reading.
|
|
101
|
+
*/
|
|
102
|
+
export const KEEPALIVE_DELAY_MS: 15000;
|
|
103
|
+
//# sourceMappingURL=worker-http.d.mts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"worker-http.d.mts","sourceRoot":"","sources":["../src/worker-http.mjs"],"names":[],"mappings":"AAoGA;;;;;;;;;;;;;;;;;;;;;;;;;;;GA2BG;AACH,uCAJW,MAAM,GAAG,IAAI,GAAG,SAAS,eACzB;IAAE,MAAM,CAAC,EAAE,MAAM,CAAA;CAAE,GACjB,MAAM,CA0BlB;AAED;;;;;;;;;;;GAWG;AACH,iCATW,MAAM,gCACN;IAAE,MAAM,CAAC,EAAE,MAAM,CAAC;IAAC,IAAI,CAAC,EAAE,OAAO,CAAC;IAAC,SAAS,CAAC,EAAE,MAAM,CAAA;CAAE,GACrD,OAAO,CAAC;IAAE,MAAM,EAAE,MAAM,CAAC;IAAC,EAAE,EAAE,OAAO,CAAC;IAAC,IAAI,EAAE,MAAM,CAAC;IAAC,IAAI,EAAE,GAAG,CAAA;CAAE,CAAC,CAwF7E;AAtMD;;;;;;;;;;;;;;;;;;;;;;;;;;;GA2BG;AACH,wCAAyC,MAAO,CAAC;AAEjD;;;;;;;;;;;;;;;;;;GAkBG;AACH,iCAAkC,KAAM,CAAC"}
|