@a11ign/screenreader-fleet 0.0.0-reserved.0 → 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +661 -0
- package/README.md +94 -2
- package/dist/capture-client.d.mts +49 -0
- package/dist/capture-client.d.mts.map +1 -0
- package/dist/capture-client.mjs +352 -0
- package/dist/capture-client.mjs.map +1 -0
- package/dist/check-worker-code.d.mts +34 -0
- package/dist/check-worker-code.d.mts.map +1 -0
- package/dist/check-worker-code.mjs +173 -0
- package/dist/check-worker-code.mjs.map +1 -0
- package/dist/cli-flags.d.mts +71 -0
- package/dist/cli-flags.d.mts.map +1 -0
- package/dist/cli-flags.mjs +207 -0
- package/dist/cli-flags.mjs.map +1 -0
- package/dist/code-drift.d.mts +140 -0
- package/dist/code-drift.d.mts.map +1 -0
- package/dist/code-drift.mjs +284 -0
- package/dist/code-drift.mjs.map +1 -0
- package/dist/command-line-census.d.mts +33 -0
- package/dist/command-line-census.d.mts.map +1 -0
- package/dist/command-line-census.mjs +96 -0
- package/dist/command-line-census.mjs.map +1 -0
- package/dist/compare-workers.d.mts +3 -0
- package/dist/compare-workers.d.mts.map +1 -0
- package/dist/compare-workers.mjs +332 -0
- package/dist/compare-workers.mjs.map +1 -0
- package/dist/control-plane-isolation.d.mts +45 -0
- package/dist/control-plane-isolation.d.mts.map +1 -0
- package/dist/control-plane-isolation.mjs +67 -0
- package/dist/control-plane-isolation.mjs.map +1 -0
- package/dist/deploy-worker.d.mts +3 -0
- package/dist/deploy-worker.d.mts.map +1 -0
- package/dist/deploy-worker.mjs +333 -0
- package/dist/deploy-worker.mjs.map +1 -0
- package/dist/doctor.d.mts +216 -0
- package/dist/doctor.d.mts.map +1 -0
- package/dist/doctor.mjs +962 -0
- package/dist/doctor.mjs.map +1 -0
- package/dist/fleet-consistency.d.mts +235 -0
- package/dist/fleet-consistency.d.mts.map +1 -0
- package/dist/fleet-consistency.mjs +436 -0
- package/dist/fleet-consistency.mjs.map +1 -0
- package/dist/fleet-env.d.mts +228 -0
- package/dist/fleet-env.d.mts.map +1 -0
- package/dist/fleet-env.mjs +509 -0
- package/dist/fleet-env.mjs.map +1 -0
- package/dist/fleet-scripts.d.mts +11 -0
- package/dist/fleet-scripts.d.mts.map +1 -0
- package/dist/fleet-scripts.mjs +41 -0
- package/dist/fleet-scripts.mjs.map +1 -0
- package/dist/git-safe-env.d.mts +10 -0
- package/dist/git-safe-env.d.mts.map +1 -0
- package/dist/git-safe-env.mjs +44 -0
- package/dist/git-safe-env.mjs.map +1 -0
- package/dist/guest-run.d.mts +26 -0
- package/dist/guest-run.d.mts.map +1 -0
- package/dist/guest-run.mjs +164 -0
- package/dist/guest-run.mjs.map +1 -0
- package/dist/host-address.d.mts +33 -0
- package/dist/host-address.d.mts.map +1 -0
- package/dist/host-address.mjs +105 -0
- package/dist/host-address.mjs.map +1 -0
- package/dist/host-capacity.d.mts +64 -0
- package/dist/host-capacity.d.mts.map +1 -0
- package/dist/host-capacity.mjs +152 -0
- package/dist/host-capacity.mjs.map +1 -0
- package/dist/host-metrics.d.mts +116 -0
- package/dist/host-metrics.d.mts.map +1 -0
- package/dist/host-metrics.mjs +201 -0
- package/dist/host-metrics.mjs.map +1 -0
- package/dist/index.d.ts +23 -0
- package/dist/index.d.ts.map +1 -0
- package/dist/index.js +25 -0
- package/dist/index.js.map +1 -0
- package/dist/local-vm.d.ts +125 -0
- package/dist/local-vm.d.ts.map +1 -0
- package/dist/local-vm.js +360 -0
- package/dist/local-vm.js.map +1 -0
- package/dist/measure-guard.d.mts +34 -0
- package/dist/measure-guard.d.mts.map +1 -0
- package/dist/measure-guard.mjs +73 -0
- package/dist/measure-guard.mjs.map +1 -0
- package/dist/normalise-fleet.d.mts +2 -0
- package/dist/normalise-fleet.d.mts.map +1 -0
- package/dist/normalise-fleet.mjs +76 -0
- package/dist/normalise-fleet.mjs.map +1 -0
- package/dist/npm-cli-executable.d.mts +42 -0
- package/dist/npm-cli-executable.d.mts.map +1 -0
- package/dist/npm-cli-executable.mjs +159 -0
- package/dist/npm-cli-executable.mjs.map +1 -0
- package/dist/probe-outcome.d.mts +89 -0
- package/dist/probe-outcome.d.mts.map +1 -0
- package/dist/probe-outcome.mjs +104 -0
- package/dist/probe-outcome.mjs.map +1 -0
- package/dist/protocol-guard.d.mts +34 -0
- package/dist/protocol-guard.d.mts.map +1 -0
- package/dist/protocol-guard.mjs +121 -0
- package/dist/protocol-guard.mjs.map +1 -0
- package/dist/source-walk.d.mts +12 -0
- package/dist/source-walk.d.mts.map +1 -0
- package/dist/source-walk.mjs +56 -0
- package/dist/source-walk.mjs.map +1 -0
- package/dist/transient-fault.d.mts +6 -0
- package/dist/transient-fault.d.mts.map +1 -0
- package/dist/transient-fault.mjs +86 -0
- package/dist/transient-fault.mjs.map +1 -0
- package/dist/utm-deprecated.d.mts +6 -0
- package/dist/utm-deprecated.d.mts.map +1 -0
- package/dist/utm-deprecated.mjs +23 -0
- package/dist/utm-deprecated.mjs.map +1 -0
- package/dist/worker-code-check.d.mts +29 -0
- package/dist/worker-code-check.d.mts.map +1 -0
- package/dist/worker-code-check.mjs +78 -0
- package/dist/worker-code-check.mjs.map +1 -0
- package/dist/worker-health.d.mts +56 -0
- package/dist/worker-health.d.mts.map +1 -0
- package/dist/worker-health.mjs +73 -0
- package/dist/worker-health.mjs.map +1 -0
- package/dist/worker-http.d.mts +103 -0
- package/dist/worker-http.d.mts.map +1 -0
- package/dist/worker-http.mjs +277 -0
- package/dist/worker-http.mjs.map +1 -0
- package/dist/worker-stats.d.mts +66 -0
- package/dist/worker-stats.d.mts.map +1 -0
- package/dist/worker-stats.mjs +143 -0
- package/dist/worker-stats.mjs.map +1 -0
- package/package.json +96 -4
- package/src/local-worker/autounattend.xml +280 -0
- package/src/local-worker/build-vm.sh +218 -0
- package/src/local-worker/clone-worker.sh +141 -0
- package/src/local-worker/create-utm-vm.sh +202 -0
- package/src/local-worker/fetch-windows-iso.sh +238 -0
- package/src/local-worker/first-boot.cmd +58 -0
- package/src/local-worker/worker-ctl.sh +442 -0
- package/src/provisioning/README.md +28 -0
- package/src/provisioning/apply-foreground-lock-timeout.ps1 +71 -0
- package/src/provisioning/bare-metal/README.md +213 -0
- package/src/provisioning/bare-metal/a11y-bootstrap.service +58 -0
- package/src/provisioning/bare-metal/autounattend.xml +428 -0
- package/src/provisioning/bare-metal/serve-bootstrap.sh +86 -0
- package/src/provisioning/bootstrap-control-plane.sh +463 -0
- package/src/provisioning/bootstrap-windows-worker.ps1 +649 -0
- package/src/provisioning/build-lean-worker-image.ps1 +275 -0
- package/src/provisioning/diagnose-nvda-worker.ps1 +174 -0
- package/src/provisioning/provision-nvda-worker.ps1 +827 -0
- package/src/provisioning/set-display-mode.ps1 +411 -0
- package/src/provisioning/stamp-provision-revision.ps1 +184 -0
package/README.md
CHANGED
|
@@ -1,3 +1,95 @@
|
|
|
1
|
-
#
|
|
1
|
+
# `@a11ign/screenreader-fleet`
|
|
2
2
|
|
|
3
|
-
|
|
3
|
+
Host-side lifecycle, health and capacity for a fleet of Windows NVDA capture workers. Runs on the machine that
|
|
4
|
+
*drives* the workers — no NVDA, no guidepup, no Windows.
|
|
5
|
+
|
|
6
|
+
```bash
|
|
7
|
+
npm install @a11ign/screenreader-fleet
|
|
8
|
+
npx a11ign-doctor # can I run right now? every check names its own fix
|
|
9
|
+
```
|
|
10
|
+
|
|
11
|
+
```js
|
|
12
|
+
import { leaseWorker } from "@a11ign/screenreader-fleet";
|
|
13
|
+
|
|
14
|
+
const lease = await leaseWorker({ after: "stop" });
|
|
15
|
+
try {
|
|
16
|
+
// ... capture against lease.url
|
|
17
|
+
} finally {
|
|
18
|
+
await lease.release(); // puts the VM back the way it was found
|
|
19
|
+
}
|
|
20
|
+
```
|
|
21
|
+
|
|
22
|
+
## A lease, not a lifecycle you own
|
|
23
|
+
|
|
24
|
+
`leaseWorker` decides what to capture against in priority order: an explicit worker you named, then
|
|
25
|
+
`inventory.yml`'s bare-metal fleet if one is declared, then a **local UTM VM (deprecated — "The UTM is
|
|
26
|
+
deprecated, that was a testing thing," repository owner, 2026-09-05)**, then the historical
|
|
27
|
+
`http://localhost:8765` default. Reaching the UTM branch prints a warning naming `fleet:*` as the
|
|
28
|
+
replacement; it is not refused outright because some machines still only have a UTM guest to reach, and
|
|
29
|
+
deleting the ~2,460 lines that manage it is a separate decision from warning about it. Only the VM case
|
|
30
|
+
has a lifecycle to manage — a bare-metal box is always on — so `release()` is a no-op for every other
|
|
31
|
+
source. It starts what is missing and **puts a VM back as it found it**: one that was already running is
|
|
32
|
+
left running, a stopped one is stopped again. A long run must not shut down something another run is
|
|
33
|
+
using.
|
|
34
|
+
|
|
35
|
+
**Stopped worker VMs are the correct resting state.** `all stopped` is a READY state, not a fault — `a11ign-doctor`
|
|
36
|
+
says so explicitly, because "go and start a worker" is the wrong instinct and costs time.
|
|
37
|
+
|
|
38
|
+
**The local UTM VM is not how this project's own corpus is captured.** It is the DEPRECATED default for a
|
|
39
|
+
solo contributor with no bare-metal fleet, kept working precisely because it is still the right path for
|
|
40
|
+
that case — this repo's own corpus is captured on ten bare-metal workers driven by
|
|
41
|
+
`inventory.yml`, which `leaseWorker` checks first. If you are consuming this package standalone with your
|
|
42
|
+
own Windows box, the VM lifecycle below still applies to you exactly as written.
|
|
43
|
+
|
|
44
|
+
## Capacity is measured, never assumed
|
|
45
|
+
|
|
46
|
+
```js
|
|
47
|
+
import { availableHostMemoryMb, workersHostCanRun } from "@a11ign/screenreader-fleet/capacity";
|
|
48
|
+
```
|
|
49
|
+
|
|
50
|
+
A worker VM costs the host **~8 GB**, not the 4 GB it is configured with — QEMU's overhead on top of guest RAM
|
|
51
|
+
that Windows dirties and never returns. So three do not fit on a 36 GB Mac, and over-committing does not merely
|
|
52
|
+
slow a run: the same page took 44.5 s with three guests up and 27.4 s with one, and the starved guests produced
|
|
53
|
+
"NVDA is running but not speaking" failures and `/health` blackouts. From outside that reads as *the workers are
|
|
54
|
+
degrading*, which is how it was misdiagnosed for a day.
|
|
55
|
+
|
|
56
|
+
Two rules follow, both learned the hard way:
|
|
57
|
+
|
|
58
|
+
- **Never `os.freemem()`.** It reported 402 MB on a host with ~12 GB to give, because macOS counts compressed
|
|
59
|
+
and inactive pages as used. The reading comes from `vm_stat`.
|
|
60
|
+
- **`vm_stat` is distorted by the very condition it must detect** — a swapped-out guest's pages count as
|
|
61
|
+
available, so the estimate *rises* as the host gets sicker; it advertised 13.7 GB free while two guests were
|
|
62
|
+
starving. The cap is therefore the lower of that estimate and a ceiling derived from physical RAM, which no
|
|
63
|
+
feedback loop can move.
|
|
64
|
+
|
|
65
|
+
## Health: watch `recoveries`, not failures
|
|
66
|
+
|
|
67
|
+
```js
|
|
68
|
+
import { assessWorker } from "@a11ign/screenreader-fleet/health";
|
|
69
|
+
```
|
|
70
|
+
|
|
71
|
+
The worst worker fault this fleet has had produced **zero failures**. One guest's NVDA went mute on 4 of 4
|
|
72
|
+
captures, the worker's own retry absorbed every one, so every capture succeeded and the eviction rule — three
|
|
73
|
+
consecutive *failures* — could never fire. The only symptom was 122.9 s per capture against a healthy peer's
|
|
74
|
+
40.6 s.
|
|
75
|
+
|
|
76
|
+
`vitals.recoveries` counts the faults a worker papered over. It is the number that rises while everything still
|
|
77
|
+
appears to work.
|
|
78
|
+
|
|
79
|
+
## The provisioning scripts ship with the package
|
|
80
|
+
|
|
81
|
+
```js
|
|
82
|
+
import { fleetScriptPaths } from "@a11ign/screenreader-fleet";
|
|
83
|
+
fleetScriptPaths().workerCtl; // absolute path to worker-ctl.sh
|
|
84
|
+
```
|
|
85
|
+
|
|
86
|
+
They are shell, so a consumer spawns them, and they are resolved from the module rather than the cwd — a
|
|
87
|
+
cwd-relative path is right exactly when the cwd is the repo root. Getting that wrong is not subtle: during this
|
|
88
|
+
extraction `doctor` reported "no local VM tooling here" on a host with three registered VMs.
|
|
89
|
+
|
|
90
|
+
**macOS + UTM** is what the VM lifecycle assumes. `utmctl` needs the UTM app running, or a perfectly healthy VM
|
|
91
|
+
reports its state as `unknown`; and UTM cannot suspend a guest with an emulated NVMe device, so stop/start is
|
|
92
|
+
the only real lifecycle — a cold boot to ready is 15–45 s, which is fine.
|
|
93
|
+
|
|
94
|
+
Not exported: `host-metrics`, `worker-stats`, `fleet-consistency`. They are measurement internals whose shapes
|
|
95
|
+
change every time something new gets measured.
|
|
@@ -0,0 +1,49 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Ask the worker for a capture we already paid for but may not have received.
|
|
3
|
+
*
|
|
4
|
+
* Returns null when there is nothing to recover — a worker predating the endpoint (404 from the router's
|
|
5
|
+
* fallback), one that restarted and lost its memory, or a capture still running. Null means "capture
|
|
6
|
+
* again", which is what every client did before the endpoint existed.
|
|
7
|
+
*
|
|
8
|
+
* A recovered FAILURE is RETURNED rather than swallowed, so a replay is indistinguishable from the original
|
|
9
|
+
* response. That keeps the worker's `fault` code — the thing it worked out, and which we would otherwise
|
|
10
|
+
* replace with "no answer" — and lets the caller's own classification decide, as it would have all along.
|
|
11
|
+
* (The dataset runner's version THREW here, because it wrapped a `fetchJson` that rejects on non-2xx;
|
|
12
|
+
* `requestJson` resolves instead, so the same intent is expressed by returning the response.)
|
|
13
|
+
*
|
|
14
|
+
* @param {string} worker @param {string} captureId
|
|
15
|
+
*/
|
|
16
|
+
export function recoverCapture(worker: string, captureId: string): Promise<{
|
|
17
|
+
status: number;
|
|
18
|
+
ok: boolean;
|
|
19
|
+
text: string;
|
|
20
|
+
json: any;
|
|
21
|
+
} | null>;
|
|
22
|
+
/**
|
|
23
|
+
* POST a capture, and on a TRANSIENT failure ask whether it actually finished before paying again.
|
|
24
|
+
*
|
|
25
|
+
* Returns what `requestJson` returns — `{status, ok, text, json}` — plus `recovered`, so this is a
|
|
26
|
+
* drop-in at the call sites that already POST `/capture` and read `response.ok` / `body.error`. Keeping
|
|
27
|
+
* their own classification is deliberate: a shared client that also decided what counts as a failure would
|
|
28
|
+
* be changing ten behaviours at once while claiming to change one.
|
|
29
|
+
*
|
|
30
|
+
* `requestJson` RESOLVES on an HTTP error and REJECTS only on a transport one, so the `catch` below is
|
|
31
|
+
* reached exactly by the case this exists for — `ETIMEDOUT`, `ECONNRESET`, a socket dying mid-answer.
|
|
32
|
+
*
|
|
33
|
+
* `waitForWorker` is deliberately NOT called here. The dataset runner has one, tuned to a corpus run's
|
|
34
|
+
* tolerance for a box gone for minutes; a gate wants an answer quickly. So this asks once, immediately,
|
|
35
|
+
* and a caller that wants to wait first passes `beforeRecovery`.
|
|
36
|
+
*
|
|
37
|
+
* @param {{ worker: string, body: object, timeoutMs?: number,
|
|
38
|
+
* beforeRecovery?: (error: unknown) => Promise<void>,
|
|
39
|
+
* onProgress?: (progress: object) => void, sync?: boolean }} request
|
|
40
|
+
*/
|
|
41
|
+
export function captureTolerantly({ worker, body, timeoutMs, beforeRecovery, onProgress, sync }: {
|
|
42
|
+
worker: string;
|
|
43
|
+
body: object;
|
|
44
|
+
timeoutMs?: number;
|
|
45
|
+
beforeRecovery?: (error: unknown) => Promise<void>;
|
|
46
|
+
onProgress?: (progress: object) => void;
|
|
47
|
+
sync?: boolean;
|
|
48
|
+
}): Promise<any>;
|
|
49
|
+
//# sourceMappingURL=capture-client.d.mts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"capture-client.d.mts","sourceRoot":"","sources":["../src/capture-client.mjs"],"names":[],"mappings":"AAkFA;;;;;;;;;;;;;;GAcG;AACH,uCAFW,MAAM,aAAiB,MAAM;;;;;UAwBvC;AAED;;;;;;;;;;;;;;;;;;GAkBG;AACH,iGAJW;IAAE,MAAM,EAAE,MAAM,CAAC;IAAC,IAAI,EAAE,MAAM,CAAC;IAAC,SAAS,CAAC,EAAE,MAAM,CAAC;IACjD,cAAc,CAAC,EAAE,CAAC,KAAK,EAAE,OAAO,KAAK,OAAO,CAAC,IAAI,CAAC,CAAC;IACnD,UAAU,CAAC,EAAE,CAAC,QAAQ,EAAE,MAAM,KAAK,IAAI,CAAC;IAAC,IAAI,CAAC,EAAE,OAAO,CAAA;CAAE,gBAwBrE"}
|
|
@@ -0,0 +1,352 @@
|
|
|
1
|
+
// @ts-check
|
|
2
|
+
/**
|
|
3
|
+
* ONE CAPTURE, TOLERANT OF THE SOCKET DYING UNDER IT — and the only place that knows how.
|
|
4
|
+
*
|
|
5
|
+
* The worker has stored completed captures under a caller-chosen id since the `captureId` work landed:
|
|
6
|
+
* `POST /capture {captureId}` then `GET /capture/<id>` returns the original response verbatim, so a lost
|
|
7
|
+
* socket costs a round trip instead of 12-520 s of real screen-reader work. That is the idempotency-key
|
|
8
|
+
* shape, for the reason payment APIs use it.
|
|
9
|
+
*
|
|
10
|
+
* TEN CLIENTS POST TO `/capture`. ONE USED THE RECOVERY. This repo's most expensive recurring shape is a
|
|
11
|
+
* remedy applied at one call site when the behaviour reaches several — `anchorToTop`, `ensureSpeechChannel`,
|
|
12
|
+
* `waitForAnnouncement`, `refreshBrowseBuffer` — and this is that shape at its largest here.
|
|
13
|
+
*
|
|
14
|
+
* MEASURED COST, 2026-08-28: `gate:stability` lost THREE canaries to `FAILED read ETIMEDOUT` across two
|
|
15
|
+
* runs — `filter-status-silent-solar/bad` on one box, then `form-error-silent/bad` and
|
|
16
|
+
* `disclosure-state-silent/good` on two others. Different pages, different machines, so it is the transport
|
|
17
|
+
* and not either. Each one turned the determinism gate INCONCLUSIVE while the capture it lost had already
|
|
18
|
+
* COMPLETED and was sitting in the worker's store, which is precisely what the store exists to serve back.
|
|
19
|
+
*
|
|
20
|
+
* The worker is bare metal on real Ethernet with real power management, which is why this arrived now: on
|
|
21
|
+
* three VMs sharing one Mac the socket was a virtual bridge and effectively lossless.
|
|
22
|
+
*
|
|
23
|
+
* MOVED HERE from `packages/lab/src/capture/capture-client.mjs` — architecture-audit.md §5, item 6: the
|
|
24
|
+
* product CLI sent no `captureId` at all, so this recovery path was unavailable to the one caller that is
|
|
25
|
+
* a user. `packages/cli` can depend on `@a11ign/screenreader-fleet` (it already does, for `requestJson`)
|
|
26
|
+
* but must never depend on `@a11ign/lab`, which is private and never published — so this is the
|
|
27
|
+
* seventh capture client and the last one, not an eighth copy of the recovery logic beside it.
|
|
28
|
+
*/
|
|
29
|
+
import { randomUUID } from "node:crypto";
|
|
30
|
+
import { requestJson, CAPTURE_CLIENT_TIMEOUT_MS } from "./worker-http.mjs";
|
|
31
|
+
import { isTransient } from "./transient-fault.mjs";
|
|
32
|
+
/** Long enough to survive a worker that is briefly busy, short enough not to double a capture's cost. */
|
|
33
|
+
const RECOVERY_TIMEOUT_MS = 30_000;
|
|
34
|
+
/**
|
|
35
|
+
* THE ESCAPE HATCH BACK TO THE SYNCHRONOUS PATH, and it says so when used.
|
|
36
|
+
*
|
|
37
|
+
* Kept because a protocol change wants a way back that does not need a deploy, and removed only once a
|
|
38
|
+
* corpus run has gone through the async path end to end.
|
|
39
|
+
*/
|
|
40
|
+
/**
|
|
41
|
+
* READ PER CALL, NOT AT MODULE LOAD, and the difference is not stylistic. A value fixed at import is one
|
|
42
|
+
* no test can vary and no process can change -- `fileProductVersion` memoised Edge's version that way and
|
|
43
|
+
* stamped five days of captures with a build they were not taken under. So it is a PARAMETER with an env
|
|
44
|
+
* default, which also lets both paths be driven from one test file.
|
|
45
|
+
*/
|
|
46
|
+
const syncByEnv = () => process.env.A11Y_SYNC_CAPTURE === "1";
|
|
47
|
+
/** Accepting a capture is a handshake, not the work: it either answers in seconds or the box is unwell. */
|
|
48
|
+
const ACCEPT_TIMEOUT_MS = 30_000;
|
|
49
|
+
/** Between polls. Short enough that a 12 s capture is not padded, long enough not to hammer a busy box. */
|
|
50
|
+
const POLL_MS = 2_000;
|
|
51
|
+
/** A read of an array already in memory; if this cannot answer, the guest's event loop is blocked. */
|
|
52
|
+
const PROGRESS_TIMEOUT_MS = 10_000;
|
|
53
|
+
/**
|
|
54
|
+
* Consecutive failed POLLS before giving up.
|
|
55
|
+
*
|
|
56
|
+
* Not one: a single dropped poll is precisely the transport fault this design exists to survive, and
|
|
57
|
+
* treating it as a failed capture would reintroduce the defect through the new door. Not unbounded either,
|
|
58
|
+
* or a box that has genuinely gone would be waited on for the whole capture budget.
|
|
59
|
+
*/
|
|
60
|
+
const MAX_POLL_FAILURES = 5;
|
|
61
|
+
const sleep = (/** @type {number} */ ms) => new Promise((r) => setTimeout(r, ms));
|
|
62
|
+
const base = (/** @type {string} */ worker) => String(worker).replace(/\/$/, "");
|
|
63
|
+
/**
|
|
64
|
+
* What is left of the OPERATION'S budget, never negative — architecture-audit.md §14.5.
|
|
65
|
+
*
|
|
66
|
+
* Every wait inside the poll loop used its own fixed constant regardless of how little of `timeoutMs`
|
|
67
|
+
* remained: a two-second sleep, a ten-second progress read and a thirty-second result read, none clipped
|
|
68
|
+
* to what was actually left. A caller asking for `timeoutMs: 20` measured 2,012 ms before an answer,
|
|
69
|
+
* because the unconditional sleep ran to completion first regardless of the deadline it was about to blow
|
|
70
|
+
* past. This is the one place that number is computed, so every wait below shares one clock.
|
|
71
|
+
*
|
|
72
|
+
* @param {number} deadline
|
|
73
|
+
*/
|
|
74
|
+
const remaining = (deadline) => Math.max(0, deadline - Date.now());
|
|
75
|
+
/**
|
|
76
|
+
* Ask the worker for a capture we already paid for but may not have received.
|
|
77
|
+
*
|
|
78
|
+
* Returns null when there is nothing to recover — a worker predating the endpoint (404 from the router's
|
|
79
|
+
* fallback), one that restarted and lost its memory, or a capture still running. Null means "capture
|
|
80
|
+
* again", which is what every client did before the endpoint existed.
|
|
81
|
+
*
|
|
82
|
+
* A recovered FAILURE is RETURNED rather than swallowed, so a replay is indistinguishable from the original
|
|
83
|
+
* response. That keeps the worker's `fault` code — the thing it worked out, and which we would otherwise
|
|
84
|
+
* replace with "no answer" — and lets the caller's own classification decide, as it would have all along.
|
|
85
|
+
* (The dataset runner's version THREW here, because it wrapped a `fetchJson` that rejects on non-2xx;
|
|
86
|
+
* `requestJson` resolves instead, so the same intent is expressed by returning the response.)
|
|
87
|
+
*
|
|
88
|
+
* @param {string} worker @param {string} captureId
|
|
89
|
+
*/
|
|
90
|
+
export async function recoverCapture(worker, captureId) {
|
|
91
|
+
let response;
|
|
92
|
+
try {
|
|
93
|
+
response = await requestJson(`${base(worker)}/capture/${captureId}`, { timeoutMs: RECOVERY_TIMEOUT_MS });
|
|
94
|
+
}
|
|
95
|
+
catch (error) {
|
|
96
|
+
// The worker went away again mid-question. Not worth a second round trip; the caller falls back to
|
|
97
|
+
// capturing, which is what it would have done anyway.
|
|
98
|
+
void error;
|
|
99
|
+
return null;
|
|
100
|
+
}
|
|
101
|
+
// 500 IS AN ANSWER, NOT AN OBSTACLE: it is the worker's own account of a failed capture, carrying the
|
|
102
|
+
// `fault` code it worked out. Returning it lets the caller's existing classification decide, exactly as
|
|
103
|
+
// it would have on the original response -- and losing it replaces a diagnosis with "no answer", which
|
|
104
|
+
// this project has repeatedly misread as a dead machine.
|
|
105
|
+
if (response.status === 500)
|
|
106
|
+
return response;
|
|
107
|
+
// 404 from the endpoint, or from an older worker's router fallback: nothing kept, so capture again.
|
|
108
|
+
if (response.status === 404)
|
|
109
|
+
return null;
|
|
110
|
+
if (!response.ok)
|
|
111
|
+
return null;
|
|
112
|
+
// "Still running" and "never heard of it" are DIFFERENT ANSWERS and must stay that way. Neither is
|
|
113
|
+
// recoverable here, but only one means the work is still being done.
|
|
114
|
+
if ( /** @type {any} */(response.json)?.state === "running")
|
|
115
|
+
return null;
|
|
116
|
+
return response;
|
|
117
|
+
}
|
|
118
|
+
/**
|
|
119
|
+
* POST a capture, and on a TRANSIENT failure ask whether it actually finished before paying again.
|
|
120
|
+
*
|
|
121
|
+
* Returns what `requestJson` returns — `{status, ok, text, json}` — plus `recovered`, so this is a
|
|
122
|
+
* drop-in at the call sites that already POST `/capture` and read `response.ok` / `body.error`. Keeping
|
|
123
|
+
* their own classification is deliberate: a shared client that also decided what counts as a failure would
|
|
124
|
+
* be changing ten behaviours at once while claiming to change one.
|
|
125
|
+
*
|
|
126
|
+
* `requestJson` RESOLVES on an HTTP error and REJECTS only on a transport one, so the `catch` below is
|
|
127
|
+
* reached exactly by the case this exists for — `ETIMEDOUT`, `ECONNRESET`, a socket dying mid-answer.
|
|
128
|
+
*
|
|
129
|
+
* `waitForWorker` is deliberately NOT called here. The dataset runner has one, tuned to a corpus run's
|
|
130
|
+
* tolerance for a box gone for minutes; a gate wants an answer quickly. So this asks once, immediately,
|
|
131
|
+
* and a caller that wants to wait first passes `beforeRecovery`.
|
|
132
|
+
*
|
|
133
|
+
* @param {{ worker: string, body: object, timeoutMs?: number,
|
|
134
|
+
* beforeRecovery?: (error: unknown) => Promise<void>,
|
|
135
|
+
* onProgress?: (progress: object) => void, sync?: boolean }} request
|
|
136
|
+
*/
|
|
137
|
+
export async function captureTolerantly({ worker, body, timeoutMs = CAPTURE_CLIENT_TIMEOUT_MS, beforeRecovery, onProgress, sync = syncByEnv() }) {
|
|
138
|
+
const captureId = randomUUID();
|
|
139
|
+
if (!sync)
|
|
140
|
+
return pollForResult({ worker, body, captureId, timeoutMs, onProgress });
|
|
141
|
+
// Said once, at the moment it applies, rather than at import: the synchronous form holds a connection
|
|
142
|
+
// open and silent for the whole capture, which is the shape that lost 9 of 40 responses on this fleet.
|
|
143
|
+
process.stderr.write("A11Y_SYNC_CAPTURE — holding one connection open for this capture\n");
|
|
144
|
+
try {
|
|
145
|
+
return { ...await post(worker, { ...body, captureId }, timeoutMs), recovered: false, pollsSurvived: 0 };
|
|
146
|
+
}
|
|
147
|
+
catch (error) {
|
|
148
|
+
if (!isTransient(error))
|
|
149
|
+
throw error;
|
|
150
|
+
// The error is handed over so a caller can SAY why it is waiting. The dataset runner prints
|
|
151
|
+
// "worker unreachable (<message>)" before a multi-minute wait, and a wait with no stated cause is
|
|
152
|
+
// one an operator kills.
|
|
153
|
+
if (beforeRecovery)
|
|
154
|
+
await beforeRecovery(error);
|
|
155
|
+
const recovered = await recoverCapture(worker, captureId);
|
|
156
|
+
// A recovered capture is the ORIGINAL response, returned rather than re-requested. `recovered` travels
|
|
157
|
+
// with it because a caller measuring the transport needs to know this one cost a round trip and not a
|
|
158
|
+
// capture -- reporting it as a clean first attempt would hide the very fault this exists for.
|
|
159
|
+
if (recovered)
|
|
160
|
+
return { ...recovered, recovered: true, pollsSurvived: 0 };
|
|
161
|
+
return { ...await post(worker, { ...body, captureId: randomUUID() }, timeoutMs), recovered: false, pollsSurvived: 0 };
|
|
162
|
+
}
|
|
163
|
+
}
|
|
164
|
+
/**
|
|
165
|
+
* DISPATCH, THEN POLL — the async path, and the reason this module exists.
|
|
166
|
+
*
|
|
167
|
+
* `POST {async:true}` returns 202 in milliseconds, so no connection is held while NVDA reads a page. The
|
|
168
|
+
* result is collected from the store with `GET /capture/<id>`, which is the endpoint that has existed for
|
|
169
|
+
* this shape all along and was only ever reached after a failure. **The recovery path is now the normal
|
|
170
|
+
* path**, which is what stops it rotting: a route that runs only when something breaks is one nobody
|
|
171
|
+
* notices has broken.
|
|
172
|
+
*
|
|
173
|
+
* A dropped poll costs one round trip and is simply retried; the capture is unaffected because nothing is
|
|
174
|
+
* riding on that socket. That is the whole difference from the synchronous form, where the answer existed
|
|
175
|
+
* only in the connection that was carrying it.
|
|
176
|
+
*/
|
|
177
|
+
/**
|
|
178
|
+
* @param {{ worker: string, body: object, captureId: string, timeoutMs: number,
|
|
179
|
+
* onProgress?: (progress: object) => void }} request
|
|
180
|
+
*/
|
|
181
|
+
async function pollForResult({ worker, body, captureId, timeoutMs, onProgress }) {
|
|
182
|
+
const deadline = Date.now() + timeoutMs;
|
|
183
|
+
let accepted;
|
|
184
|
+
try {
|
|
185
|
+
accepted = await post(worker, { ...body, captureId, async: true }, ACCEPT_TIMEOUT_MS);
|
|
186
|
+
}
|
|
187
|
+
catch (error) {
|
|
188
|
+
if (!isTransient(error))
|
|
189
|
+
throw error;
|
|
190
|
+
const reconciled = await reconcileLostAcceptance(worker, captureId, deadline);
|
|
191
|
+
if (reconciled.state === "done")
|
|
192
|
+
return { ...reconciled.response, recovered: true, pollsSurvived: 0 };
|
|
193
|
+
if (reconciled.state === "unknown") {
|
|
194
|
+
// CONFIRMED nothing is running under this id -- only now is a fresh one safe, exactly as the
|
|
195
|
+
// synchronous escape hatch already does for the identical failure. Minting one on the first sign
|
|
196
|
+
// of trouble, before asking, is what the audit's remedy forbids: it would risk a second real
|
|
197
|
+
// capture running under a worker that already accepted the first.
|
|
198
|
+
return pollForResult({
|
|
199
|
+
worker, body, captureId: randomUUID(), timeoutMs: Math.max(0, deadline - Date.now()), onProgress,
|
|
200
|
+
});
|
|
201
|
+
}
|
|
202
|
+
// reconciled.state === "running": the worker DID accept it under the id we already hold -- the 202
|
|
203
|
+
// was lost, not the acceptance. Fall through to the ordinary poll loop below exactly as a received
|
|
204
|
+
// 202 would have.
|
|
205
|
+
}
|
|
206
|
+
// A worker too old to know `async` runs the capture SYNCHRONOUSLY and answers 200 with the result. That
|
|
207
|
+
// is not an error and must not be retried -- it is the additive-field contract this project uses for
|
|
208
|
+
// every wire change, and it means a host can be deployed before the fleet.
|
|
209
|
+
if (accepted && accepted.status !== 202)
|
|
210
|
+
return { ...accepted, recovered: false, pollsSurvived: 0 };
|
|
211
|
+
return awaitCompletion({ worker, captureId, deadline, timeoutMs, onProgress });
|
|
212
|
+
}
|
|
213
|
+
/**
|
|
214
|
+
* ASKING ABOUT A CAPTURE WE MAY OR MAY NOT HAVE STARTED, using the SAME three-way discrimination a
|
|
215
|
+
* dropped poll already relies on (`pollOnce`) rather than `recoverCapture`, which collapses "running"
|
|
216
|
+
* and "unknown" into one null and cannot tell them apart -- the exact distinction this exists to make.
|
|
217
|
+
*
|
|
218
|
+
* A 404 here is not quite `pollOnce`'s documented "worker restarted mid-capture": no 202 was ever
|
|
219
|
+
* received, so there is nothing to have restarted AWAY FROM. It means the POST itself never reached the
|
|
220
|
+
* worker, which is what makes minting a fresh id safe once this returns "unknown" and not before.
|
|
221
|
+
*
|
|
222
|
+
* The audit's own caveat applies in principle: "unknown does not prove execution never occurred" if the
|
|
223
|
+
* result were evicted (§14.4) before this ever asks — but that needs seven OTHER captures to finish on
|
|
224
|
+
* this worker in the seconds between the lost 202 and this reconciliation, which the worker's own `busy`
|
|
225
|
+
* gate (one capture at a time) makes impossible while nothing else is running under this id.
|
|
226
|
+
*
|
|
227
|
+
* @param {string} worker @param {string} captureId @param {number} deadline
|
|
228
|
+
* @returns {Promise<{ state: "running" } | { state: "unknown" } | { state: "done", response: any }>}
|
|
229
|
+
*/
|
|
230
|
+
async function reconcileLostAcceptance(worker, captureId, deadline) {
|
|
231
|
+
let lastError;
|
|
232
|
+
while (Date.now() < deadline) {
|
|
233
|
+
const poll = await pollOnce(worker, captureId, Math.min(RECOVERY_TIMEOUT_MS, remaining(deadline)));
|
|
234
|
+
if (poll.running)
|
|
235
|
+
return { state: "running" };
|
|
236
|
+
if (poll.done)
|
|
237
|
+
return { state: "done", response: poll.response };
|
|
238
|
+
if (poll.lost)
|
|
239
|
+
return { state: "unknown" };
|
|
240
|
+
lastError = poll.error;
|
|
241
|
+
await sleep(Math.min(POLL_MS, remaining(deadline)));
|
|
242
|
+
}
|
|
243
|
+
// Never resolved within the budget: neither confirmed running nor confirmed absent. The ORIGINAL
|
|
244
|
+
// acceptance failure is the fault that actually occurred, so it is what surfaces -- not a generic
|
|
245
|
+
// timeout that would hide which of the two things went wrong.
|
|
246
|
+
throw lastError ?? Object.assign(new Error(`could not confirm capture ${captureId} was accepted, within its remaining budget`), { code: "ETIMEDOUT" });
|
|
247
|
+
}
|
|
248
|
+
/**
|
|
249
|
+
* DISPATCH, THEN POLL — the async path, and the reason this module exists.
|
|
250
|
+
*
|
|
251
|
+
* `POST {async:true}` returns 202 in milliseconds, so no connection is held while NVDA reads a page. The
|
|
252
|
+
* result is collected from the store with `GET /capture/<id>`, which is the endpoint that has existed for
|
|
253
|
+
* this shape all along and was only ever reached after a failure. **The recovery path is now the normal
|
|
254
|
+
* path**, which is what stops it rotting: a route that runs only when something breaks is one nobody
|
|
255
|
+
* notices has broken.
|
|
256
|
+
*
|
|
257
|
+
* A dropped poll costs one round trip and is simply retried; the capture is unaffected because nothing is
|
|
258
|
+
* riding on that socket. That is the whole difference from the synchronous form, where the answer existed
|
|
259
|
+
* only in the connection that was carrying it.
|
|
260
|
+
*
|
|
261
|
+
* @param {{ worker: string, captureId: string, deadline: number, timeoutMs: number,
|
|
262
|
+
* onProgress?: (progress: object) => void }} request
|
|
263
|
+
*/
|
|
264
|
+
async function awaitCompletion({ worker, captureId, deadline, timeoutMs, onProgress }) {
|
|
265
|
+
let transportFailures = 0;
|
|
266
|
+
let survived = 0;
|
|
267
|
+
while (Date.now() < deadline) {
|
|
268
|
+
// CLIPPED, EACH TIME, TO WHAT IS LEFT — architecture-audit.md §14.5. A budget of 20 ms must not spend
|
|
269
|
+
// a full 2 s sleeping before it is even allowed to check the clock again.
|
|
270
|
+
await sleep(Math.min(POLL_MS, remaining(deadline)));
|
|
271
|
+
if (onProgress)
|
|
272
|
+
await readProgress(worker, onProgress, Math.min(PROGRESS_TIMEOUT_MS, remaining(deadline)));
|
|
273
|
+
const poll = await pollOnce(worker, captureId, Math.min(RECOVERY_TIMEOUT_MS, remaining(deadline)));
|
|
274
|
+
if (poll.done) {
|
|
275
|
+
// WHAT THE TRANSPORT DID, carried out with the result. Under the synchronous protocol a dropped
|
|
276
|
+
// response destroyed the capture; here it costs one poll -- but "harmless" and "not happening" are
|
|
277
|
+
// different facts, and only one of them means the network is healthy. Reporting it keeps the
|
|
278
|
+
// question answerable after the fix that stopped it mattering.
|
|
279
|
+
return { ...poll.response, recovered: false, pollsSurvived: survived };
|
|
280
|
+
}
|
|
281
|
+
if (poll.lost) {
|
|
282
|
+
throw Object.assign(new Error(`worker forgot capture ${captureId} after accepting it — it restarted `
|
|
283
|
+
+ "mid-capture, so the work is gone and the case must be re-issued"), { code: "CAPTURE_LOST" });
|
|
284
|
+
}
|
|
285
|
+
// A FAILED POLL IS NOT A FAILED CAPTURE, and conflating them would give back the defect this design
|
|
286
|
+
// removes: the worker is still working, we merely could not ask. Retried until a RUN of them says the
|
|
287
|
+
// box has gone, rather than on the first one -- a single dropped request is the exact fault this exists
|
|
288
|
+
// to survive.
|
|
289
|
+
if (poll.unreachable)
|
|
290
|
+
survived += 1;
|
|
291
|
+
transportFailures = poll.unreachable ? transportFailures + 1 : 0;
|
|
292
|
+
if (transportFailures >= MAX_POLL_FAILURES)
|
|
293
|
+
throw poll.error;
|
|
294
|
+
}
|
|
295
|
+
throw Object.assign(new Error(`capture ${captureId} did not finish within ${timeoutMs} ms`), { code: "ETIMEDOUT" });
|
|
296
|
+
}
|
|
297
|
+
/**
|
|
298
|
+
* ONE POLL, WITH FOUR DISTINCT ANSWERS — and keeping them apart is the whole job.
|
|
299
|
+
*
|
|
300
|
+
* done the capture finished (200 or 500); the 500 carries the worker's own fault code
|
|
301
|
+
* lost 404 AFTER a 202 acceptance: the worker restarted, the work is gone, re-issue
|
|
302
|
+
* running 202; keep waiting
|
|
303
|
+
* unreachable we could not ask; the capture is unaffected
|
|
304
|
+
*
|
|
305
|
+
* NOT `recoverCapture`, and that is the correction. It SWALLOWS a transport error and returns null, which
|
|
306
|
+
* is right for its own job — "we already failed, is the result there?" — and wrong here, because it makes
|
|
307
|
+
* "could not ask" indistinguishable from "still running". Written that way first, and the retry counter
|
|
308
|
+
* built on it was unreachable: found by mutation, since deleting the retry changed nothing.
|
|
309
|
+
*
|
|
310
|
+
* @param {string} worker @param {string} captureId @param {number} [timeoutMs] clipped to the operation's
|
|
311
|
+
* remaining budget by the caller — see `remaining()` — so this read cannot outlive the deadline it is
|
|
312
|
+
* answering to on its own.
|
|
313
|
+
*/
|
|
314
|
+
async function pollOnce(worker, captureId, timeoutMs = RECOVERY_TIMEOUT_MS) {
|
|
315
|
+
let response;
|
|
316
|
+
try {
|
|
317
|
+
response = await requestJson(`${base(worker)}/capture/${captureId}`, { timeoutMs });
|
|
318
|
+
}
|
|
319
|
+
catch (error) {
|
|
320
|
+
return { unreachable: true, error };
|
|
321
|
+
}
|
|
322
|
+
if (response.status === 404)
|
|
323
|
+
return { lost: true };
|
|
324
|
+
if (response.status === 202)
|
|
325
|
+
return { running: true };
|
|
326
|
+
// 200 and 500 are both ANSWERS: the second is the worker's diagnosis, and losing it would replace a
|
|
327
|
+
// fault code with silence -- which this project has repeatedly misread as a dead machine.
|
|
328
|
+
if (response.ok || response.status === 500)
|
|
329
|
+
return { done: true, response };
|
|
330
|
+
// Anything else is a worker speaking a protocol we do not know; treat it as unreachable rather than as
|
|
331
|
+
// an answer, so it is retried and then surfaces with its own status rather than being read as a capture.
|
|
332
|
+
return { unreachable: true, error: new Error(`unexpected ${response.status} polling ${captureId}`) };
|
|
333
|
+
}
|
|
334
|
+
/** The phase the worker is IN, so a caller can tell a slow capture from a wedged one. */
|
|
335
|
+
async function readProgress(/** @type {string} */ worker, /** @type {(p: object) => void} */ onProgress,
|
|
336
|
+
/** @type {number} */ timeoutMs = PROGRESS_TIMEOUT_MS) {
|
|
337
|
+
try {
|
|
338
|
+
const { json } = await requestJson(`${base(worker)}/progress`, { timeoutMs });
|
|
339
|
+
if (json && typeof json === "object")
|
|
340
|
+
onProgress(json);
|
|
341
|
+
}
|
|
342
|
+
catch (error) {
|
|
343
|
+
// Progress is a convenience; failing to read it must never fail a capture that is going fine.
|
|
344
|
+
void error;
|
|
345
|
+
}
|
|
346
|
+
}
|
|
347
|
+
/** @param {string} worker @param {object} body @param {number} timeoutMs */
|
|
348
|
+
function post(worker, body, timeoutMs) {
|
|
349
|
+
// `requestJson` serialises the body and sets the headers; passing a string here would double-encode it.
|
|
350
|
+
return requestJson(`${base(worker)}/capture`, { method: "POST", body, timeoutMs });
|
|
351
|
+
}
|
|
352
|
+
//# sourceMappingURL=capture-client.mjs.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"capture-client.mjs","sourceRoot":"","sources":["../src/capture-client.mjs"],"names":[],"mappings":"AAAA,YAAY;AACZ;;;;;;;;;;;;;;;;;;;;;;;;;;GA0BG;AACH,OAAO,EAAE,UAAU,EAAE,MAAM,aAAa,CAAC;AAEzC,OAAO,EAAE,WAAW,EAAE,yBAAyB,EAAE,MAAM,mBAAmB,CAAC;AAC3E,OAAO,EAAE,WAAW,EAAE,MAAM,uBAAuB,CAAC;AAEpD,yGAAyG;AACzG,MAAM,mBAAmB,GAAG,MAAM,CAAC;AAEnC;;;;;GAKG;AACH;;;;;GAKG;AACH,MAAM,SAAS,GAAG,GAAG,EAAE,CAAC,OAAO,CAAC,GAAG,CAAC,iBAAiB,KAAK,GAAG,CAAC;AAE9D,2GAA2G;AAC3G,MAAM,iBAAiB,GAAG,MAAM,CAAC;AACjC,2GAA2G;AAC3G,MAAM,OAAO,GAAG,KAAK,CAAC;AACtB,sGAAsG;AACtG,MAAM,mBAAmB,GAAG,MAAM,CAAC;AACnC;;;;;;GAMG;AACH,MAAM,iBAAiB,GAAG,CAAC,CAAC;AAE5B,MAAM,KAAK,GAAG,CAAC,qBAAqB,CAAC,EAAE,EAAE,EAAE,CAAC,IAAI,OAAO,CAAC,CAAC,CAAC,EAAE,EAAE,CAAC,UAAU,CAAC,CAAC,EAAE,EAAE,CAAC,CAAC,CAAC;AAElF,MAAM,IAAI,GAAG,CAAC,qBAAqB,CAAC,MAAM,EAAE,EAAE,CAAC,MAAM,CAAC,MAAM,CAAC,CAAC,OAAO,CAAC,KAAK,EAAE,EAAE,CAAC,CAAC;AAEjF;;;;;;;;;;GAUG;AACH,MAAM,SAAS,GAAG,CAAC,QAAQ,EAAE,EAAE,CAAC,IAAI,CAAC,GAAG,CAAC,CAAC,EAAE,QAAQ,GAAG,IAAI,CAAC,GAAG,EAAE,CAAC,CAAC;AAEnE;;;;;;;;;;;;;;GAcG;AACH,MAAM,CAAC,KAAK,UAAU,cAAc,CAAC,MAAM,EAAE,SAAS;IACpD,IAAI,QAAQ,CAAC;IACb,IAAI,CAAC;QACH,QAAQ,GAAG,MAAM,WAAW,CAAC,GAAG,IAAI,CAAC,MAAM,CAAC,YAAY,SAAS,EAAE,EAAE,EAAE,SAAS,EAAE,mBAAmB,EAAE,CAAC,CAAC;IAC3G,CAAC;IAAC,OAAO,KAAK,EAAE,CAAC;QACf,mGAAmG;QACnG,sDAAsD;QACtD,KAAK,KAAK,CAAC;QACX,OAAO,IAAI,CAAC;IACd,CAAC;IACD,sGAAsG;IACtG,wGAAwG;IACxG,uGAAuG;IACvG,yDAAyD;IACzD,IAAI,QAAQ,CAAC,MAAM,KAAK,GAAG;QAAE,OAAO,QAAQ,CAAC;IAC7C,oGAAoG;IACpG,IAAI,QAAQ,CAAC,MAAM,KAAK,GAAG;QAAE,OAAO,IAAI,CAAC;IACzC,IAAI,CAAC,QAAQ,CAAC,EAAE;QAAE,OAAO,IAAI,CAAC;IAC9B,mGAAmG;IACnG,qEAAqE;IACrE,KAAI,kBAAmB,CAAC,QAAQ,CAAC,IAAI,CAAC,EAAE,KAAK,KAAK,SAAS;QAAE,OAAO,IAAI,CAAC;IACzE,OAAO,QAAQ,CAAC;AAClB,CAAC;AAED;;;;;;;;;;;;;;;;;;GAkBG;AACH,MAAM,CAAC,KAAK,UAAU,iBAAiB,CAAC,EAAE,MAAM,EAAE,IAAI,EAAE,SAAS,GAAG,yBAAyB,EAAE,cAAc,EAC3G,UAAU,EAAE,IAAI,GAAG,SAAS,EAAE,EAAE;IAChC,MAAM,SAAS,GAAG,UAAU,EAAE,CAAC;IAC/B,IAAI,CAAC,IAAI;QAAE,OAAO,aAAa,CAAC,EAAE,MAAM,EAAE,IAAI,EAAE,SAAS,EAAE,SAAS,EAAE,UAAU,EAAE,CAAC,CAAC;IACpF,sGAAsG;IACtG,uGAAuG;IACvG,OAAO,CAAC,MAAM,CAAC,KAAK,CAAC,oEAAoE,CAAC,CAAC;IAC3F,IAAI,CAAC;QACH,OAAO,EAAE,GAAG,MAAM,IAAI,CAAC,MAAM,EAAE,EAAE,GAAG,IAAI,EAAE,SAAS,EAAE,EAAE,SAAS,CAAC,EAAE,SAAS,EAAE,KAAK,EAAE,aAAa,EAAE,CAAC,EAAE,CAAC;IAC1G,CAAC;IAAC,OAAO,KAAK,EAAE,CAAC;QACf,IAAI,CAAC,WAAW,CAAC,KAAK,CAAC;YAAE,MAAM,KAAK,CAAC;QACrC,4FAA4F;QAC5F,kGAAkG;QAClG,yBAAyB;QACzB,IAAI,cAAc;YAAE,MAAM,cAAc,CAAC,KAAK,CAAC,CAAC;QAChD,MAAM,SAAS,GAAG,MAAM,cAAc,CAAC,MAAM,EAAE,SAAS,CAAC,CAAC;QAC1D,uGAAuG;QACvG,sGAAsG;QACtG,8FAA8F;QAC9F,IAAI,SAAS;YAAE,OAAO,EAAE,GAAG,SAAS,EAAE,SAAS,EAAE,IAAI,EAAE,aAAa,EAAE,CAAC,EAAE,CAAC;QAC1E,OAAO,EAAE,GAAG,MAAM,IAAI,CAAC,MAAM,EAAE,EAAE,GAAG,IAAI,EAAE,SAAS,EAAE,UAAU,EAAE,EAAE,EAAE,SAAS,CAAC,EAAE,SAAS,EAAE,KAAK,EAAE,aAAa,EAAE,CAAC,EAAE,CAAC;IACxH,CAAC;AACH,CAAC;AAED;;;;;;;;;;;;GAYG;AACH;;;GAGG;AACH,KAAK,UAAU,aAAa,CAAC,EAAE,MAAM,EAAE,IAAI,EAAE,SAAS,EAAE,SAAS,EAAE,UAAU,EAAE;IAC7E,MAAM,QAAQ,GAAG,IAAI,CAAC,GAAG,EAAE,GAAG,SAAS,CAAC;IACxC,IAAI,QAAQ,CAAC;IACb,IAAI,CAAC;QACH,QAAQ,GAAG,MAAM,IAAI,CAAC,MAAM,EAAE,EAAE,GAAG,IAAI,EAAE,SAAS,EAAE,KAAK,EAAE,IAAI,EAAE,EAAE,iBAAiB,CAAC,CAAC;IACxF,CAAC;IAAC,OAAO,KAAK,EAAE,CAAC;QACf,IAAI,CAAC,WAAW,CAAC,KAAK,CAAC;YAAE,MAAM,KAAK,CAAC;QACrC,MAAM,UAAU,GAAG,MAAM,uBAAuB,CAAC,MAAM,EAAE,SAAS,EAAE,QAAQ,CAAC,CAAC;QAC9E,IAAI,UAAU,CAAC,KAAK,KAAK,MAAM;YAAE,OAAO,EAAE,GAAG,UAAU,CAAC,QAAQ,EAAE,SAAS,EAAE,IAAI,EAAE,aAAa,EAAE,CAAC,EAAE,CAAC;QACtG,IAAI,UAAU,CAAC,KAAK,KAAK,SAAS,EAAE,CAAC;YACnC,6FAA6F;YAC7F,iGAAiG;YACjG,6FAA6F;YAC7F,kEAAkE;YAClE,OAAO,aAAa,CAAC;gBACnB,MAAM,EAAE,IAAI,EAAE,SAAS,EAAE,UAAU,EAAE,EAAE,SAAS,EAAE,IAAI,CAAC,GAAG,CAAC,CAAC,EAAE,QAAQ,GAAG,IAAI,CAAC,GAAG,EAAE,CAAC,EAAE,UAAU;aACjG,CAAC,CAAC;QACL,CAAC;QACD,mGAAmG;QACnG,mGAAmG;QACnG,kBAAkB;IACpB,CAAC;IACD,wGAAwG;IACxG,qGAAqG;IACrG,2EAA2E;IAC3E,IAAI,QAAQ,IAAI,QAAQ,CAAC,MAAM,KAAK,GAAG;QAAE,OAAO,EAAE,GAAG,QAAQ,EAAE,SAAS,EAAE,KAAK,EAAE,aAAa,EAAE,CAAC,EAAE,CAAC;IACpG,OAAO,eAAe,CAAC,EAAE,MAAM,EAAE,SAAS,EAAE,QAAQ,EAAE,SAAS,EAAE,UAAU,EAAE,CAAC,CAAC;AACjF,CAAC;AAED;;;;;;;;;;;;;;;;GAgBG;AACH,KAAK,UAAU,uBAAuB,CAAC,MAAM,EAAE,SAAS,EAAE,QAAQ;IAChE,IAAI,SAAS,CAAC;IACd,OAAO,IAAI,CAAC,GAAG,EAAE,GAAG,QAAQ,EAAE,CAAC;QAC7B,MAAM,IAAI,GAAG,MAAM,QAAQ,CAAC,MAAM,EAAE,SAAS,EAAE,IAAI,CAAC,GAAG,CAAC,mBAAmB,EAAE,SAAS,CAAC,QAAQ,CAAC,CAAC,CAAC,CAAC;QACnG,IAAI,IAAI,CAAC,OAAO;YAAE,OAAO,EAAE,KAAK,EAAE,SAAS,EAAE,CAAC;QAC9C,IAAI,IAAI,CAAC,IAAI;YAAE,OAAO,EAAE,KAAK,EAAE,MAAM,EAAE,QAAQ,EAAE,IAAI,CAAC,QAAQ,EAAE,CAAC;QACjE,IAAI,IAAI,CAAC,IAAI;YAAE,OAAO,EAAE,KAAK,EAAE,SAAS,EAAE,CAAC;QAC3C,SAAS,GAAG,IAAI,CAAC,KAAK,CAAC;QACvB,MAAM,KAAK,CAAC,IAAI,CAAC,GAAG,CAAC,OAAO,EAAE,SAAS,CAAC,QAAQ,CAAC,CAAC,CAAC,CAAC;IACtD,CAAC;IACD,iGAAiG;IACjG,kGAAkG;IAClG,8DAA8D;IAC9D,MAAM,SAAS,IAAI,MAAM,CAAC,MAAM,CAC9B,IAAI,KAAK,CAAC,6BAA6B,SAAS,4CAA4C,CAAC,EAC7F,EAAE,IAAI,EAAE,WAAW,EAAE,CAAC,CAAC;AAC3B,CAAC;AAED;;;;;;;;;;;;;;;GAeG;AACH,KAAK,UAAU,eAAe,CAAC,EAAE,MAAM,EAAE,SAAS,EAAE,QAAQ,EAAE,SAAS,EAAE,UAAU,EAAE;IACnF,IAAI,iBAAiB,GAAG,CAAC,CAAC;IAC1B,IAAI,QAAQ,GAAG,CAAC,CAAC;IACjB,OAAO,IAAI,CAAC,GAAG,EAAE,GAAG,QAAQ,EAAE,CAAC;QAC7B,sGAAsG;QACtG,0EAA0E;QAC1E,MAAM,KAAK,CAAC,IAAI,CAAC,GAAG,CAAC,OAAO,EAAE,SAAS,CAAC,QAAQ,CAAC,CAAC,CAAC,CAAC;QACpD,IAAI,UAAU;YAAE,MAAM,YAAY,CAAC,MAAM,EAAE,UAAU,EAAE,IAAI,CAAC,GAAG,CAAC,mBAAmB,EAAE,SAAS,CAAC,QAAQ,CAAC,CAAC,CAAC,CAAC;QAC3G,MAAM,IAAI,GAAG,MAAM,QAAQ,CAAC,MAAM,EAAE,SAAS,EAAE,IAAI,CAAC,GAAG,CAAC,mBAAmB,EAAE,SAAS,CAAC,QAAQ,CAAC,CAAC,CAAC,CAAC;QACnG,IAAI,IAAI,CAAC,IAAI,EAAE,CAAC;YACd,gGAAgG;YAChG,mGAAmG;YACnG,6FAA6F;YAC7F,+DAA+D;YAC/D,OAAO,EAAE,GAAG,IAAI,CAAC,QAAQ,EAAE,SAAS,EAAE,KAAK,EAAE,aAAa,EAAE,QAAQ,EAAE,CAAC;QACzE,CAAC;QACD,IAAI,IAAI,CAAC,IAAI,EAAE,CAAC;YACd,MAAM,MAAM,CAAC,MAAM,CAAC,IAAI,KAAK,CAAC,yBAAyB,SAAS,qCAAqC;kBACjG,iEAAiE,CAAC,EAAE,EAAE,IAAI,EAAE,cAAc,EAAE,CAAC,CAAC;QACpG,CAAC;QACD,oGAAoG;QACpG,sGAAsG;QACtG,wGAAwG;QACxG,cAAc;QACd,IAAI,IAAI,CAAC,WAAW;YAAE,QAAQ,IAAI,CAAC,CAAC;QACpC,iBAAiB,GAAG,IAAI,CAAC,WAAW,CAAC,CAAC,CAAC,iBAAiB,GAAG,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC;QACjE,IAAI,iBAAiB,IAAI,iBAAiB;YAAE,MAAM,IAAI,CAAC,KAAK,CAAC;IAC/D,CAAC;IACD,MAAM,MAAM,CAAC,MAAM,CAAC,IAAI,KAAK,CAAC,WAAW,SAAS,0BAA0B,SAAS,KAAK,CAAC,EACzF,EAAE,IAAI,EAAE,WAAW,EAAE,CAAC,CAAC;AAC3B,CAAC;AAED;;;;;;;;;;;;;;;;GAgBG;AACH,KAAK,UAAU,QAAQ,CAAC,MAAM,EAAE,SAAS,EAAE,SAAS,GAAG,mBAAmB;IACxE,IAAI,QAAQ,CAAC;IACb,IAAI,CAAC;QACH,QAAQ,GAAG,MAAM,WAAW,CAAC,GAAG,IAAI,CAAC,MAAM,CAAC,YAAY,SAAS,EAAE,EAAE,EAAE,SAAS,EAAE,CAAC,CAAC;IACtF,CAAC;IAAC,OAAO,KAAK,EAAE,CAAC;QACf,OAAO,EAAE,WAAW,EAAE,IAAI,EAAE,KAAK,EAAE,CAAC;IACtC,CAAC;IACD,IAAI,QAAQ,CAAC,MAAM,KAAK,GAAG;QAAE,OAAO,EAAE,IAAI,EAAE,IAAI,EAAE,CAAC;IACnD,IAAI,QAAQ,CAAC,MAAM,KAAK,GAAG;QAAE,OAAO,EAAE,OAAO,EAAE,IAAI,EAAE,CAAC;IACtD,oGAAoG;IACpG,0FAA0F;IAC1F,IAAI,QAAQ,CAAC,EAAE,IAAI,QAAQ,CAAC,MAAM,KAAK,GAAG;QAAE,OAAO,EAAE,IAAI,EAAE,IAAI,EAAE,QAAQ,EAAE,CAAC;IAC5E,uGAAuG;IACvG,yGAAyG;IACzG,OAAO,EAAE,WAAW,EAAE,IAAI,EAAE,KAAK,EAAE,IAAI,KAAK,CAAC,cAAc,QAAQ,CAAC,MAAM,YAAY,SAAS,EAAE,CAAC,EAAE,CAAC;AACvG,CAAC;AAED,yFAAyF;AACzF,KAAK,UAAU,YAAY,CAAC,qBAAqB,CAAC,MAAM,EAAE,kCAAkC,CAAC,UAAU;AACrG,qBAAqB,CAAC,SAAS,GAAG,mBAAmB;IACrD,IAAI,CAAC;QACH,MAAM,EAAE,IAAI,EAAE,GAAG,MAAM,WAAW,CAAC,GAAG,IAAI,CAAC,MAAM,CAAC,WAAW,EAAE,EAAE,SAAS,EAAE,CAAC,CAAC;QAC9E,IAAI,IAAI,IAAI,OAAO,IAAI,KAAK,QAAQ;YAAE,UAAU,CAAC,IAAI,CAAC,CAAC;IACzD,CAAC;IAAC,OAAO,KAAK,EAAE,CAAC;QACf,8FAA8F;QAC9F,KAAK,KAAK,CAAC;IACb,CAAC;AACH,CAAC;AAED,4EAA4E;AAC5E,SAAS,IAAI,CAAC,MAAM,EAAE,IAAI,EAAE,SAAS;IACnC,wGAAwG;IACxG,OAAO,WAAW,CAAC,GAAG,IAAI,CAAC,MAAM,CAAC,UAAU,EAAE,EAAE,MAAM,EAAE,MAAM,EAAE,IAAI,EAAE,SAAS,EAAE,CAAC,CAAC;AACrF,CAAC"}
|
|
@@ -0,0 +1,34 @@
|
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
/**
|
|
3
|
+
* Which workers to ask, AND WHERE THAT LIST CAME FROM — the second half is not decoration.
|
|
4
|
+
*
|
|
5
|
+
* This used to answer the local UTM pool or nothing, and print "no worker is running — nothing to compare"
|
|
6
|
+
* when it found neither. Measured 2026-08-28 with five bare-metal workers serving `/health` and all five
|
|
7
|
+
* STALE against this checkout: the bare command reported nothing to compare, and the same command with
|
|
8
|
+
* `A11Y_WORKERS` set reported `5 stale worker(s)`. One env var apart, and the quiet answer was the wrong one.
|
|
9
|
+
*
|
|
10
|
+
* That is `lab:inventory`'s lesson at a different layer — *"'none here' and 'none anywhere' are different
|
|
11
|
+
* answers, and it now refuses to turn the first into the second"* — and it lands harder here, because this
|
|
12
|
+
* command exists to stop a corpus being captured on the wrong code. A false clean from it is the failure it
|
|
13
|
+
* was written to prevent, delivered by the tool itself.
|
|
14
|
+
*
|
|
15
|
+
* The inventory was already imported and already read, twelve lines below, to print the REMEDY. So the
|
|
16
|
+
* command could name the five workers it should have checked while insisting it had none to check.
|
|
17
|
+
*
|
|
18
|
+
* Reading it here is safe in a way it would not be for a capture run: this probes `/health` and starts
|
|
19
|
+
* nothing, so the rule that naming workers means you are managing them does not apply.
|
|
20
|
+
*/
|
|
21
|
+
export function workerUrls({ named, local, inventory, }?: {
|
|
22
|
+
named?: typeof configuredWorkers | undefined;
|
|
23
|
+
local?: typeof localPoolUrls | undefined;
|
|
24
|
+
inventory?: typeof inventoryWorkerUrls | undefined;
|
|
25
|
+
}): {
|
|
26
|
+
urls: string[];
|
|
27
|
+
source: string;
|
|
28
|
+
};
|
|
29
|
+
import { configuredWorkers } from "./fleet-env.mjs";
|
|
30
|
+
/** The local UTM pool, or none — `utmctl` is absent on a machine that never had one. */
|
|
31
|
+
declare function localPoolUrls(): any;
|
|
32
|
+
import { inventoryWorkerUrls } from "./fleet-env.mjs";
|
|
33
|
+
export {};
|
|
34
|
+
//# sourceMappingURL=check-worker-code.d.mts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"check-worker-code.d.mts","sourceRoot":"","sources":["../src/check-worker-code.mjs"],"names":[],"mappings":";AAwEA;;;;;;;;;;;;;;;;;;GAkBG;AACH;;;;;;;EAQC;kCA5EyE,iBAAiB;AA8E3F,wFAAwF;AACxF,sCAWC;oCA1FyE,iBAAiB"}
|