@a11ign/screenreader-fleet 0.2.0 → 0.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/capture-client.d.mts +0 -1
- package/dist/capture-client.mjs +144 -306
- package/dist/check-worker-code.d.mts +0 -1
- package/dist/check-worker-code.mjs +42 -110
- package/dist/cli-flags.d.mts +0 -1
- package/dist/cli-flags.mjs +33 -179
- package/dist/code-drift.d.mts +0 -1
- package/dist/command-line-census.d.mts +0 -1
- package/dist/compare-workers.d.mts +0 -1
- package/dist/compare-workers.mjs +383 -255
- package/dist/control-plane-isolation.d.mts +0 -1
- package/dist/deploy-worker.d.mts +0 -1
- package/dist/deploy-worker.mjs +101 -242
- package/dist/doctor.d.mts +0 -1
- package/dist/doctor.mjs +361 -809
- package/dist/fleet-consistency.d.mts +0 -1
- package/dist/fleet-consistency.mjs +155 -379
- package/dist/fleet-env.d.mts +0 -1
- package/dist/fleet-env.mjs +148 -438
- package/dist/fleet-scripts.d.mts +0 -1
- package/dist/git-safe-env.d.mts +0 -1
- package/dist/guest-run.d.mts +0 -1
- package/dist/host-address.d.mts +0 -1
- package/dist/host-address.mjs +19 -90
- package/dist/host-capacity.d.mts +0 -1
- package/dist/host-capacity.mjs +22 -136
- package/dist/host-metrics.d.mts +0 -1
- package/dist/index.d.ts +0 -1
- package/dist/index.mjs +231 -0
- package/dist/local-vm.d.ts +0 -1
- package/dist/measure-guard.d.mts +0 -1
- package/dist/normalise-fleet.d.mts +0 -1
- package/dist/npm-cli-executable.d.mts +0 -1
- package/dist/probe-outcome.d.mts +0 -1
- package/dist/probe-outcome.mjs +50 -96
- package/dist/protocol-guard.d.mts +0 -1
- package/dist/source-walk.d.mts +0 -1
- package/dist/src_fleet-scripts_mjs.mjs +16 -0
- package/dist/src_git-safe-env_mjs.mjs +9 -0
- package/dist/src_utm-deprecated_mjs.mjs +4 -0
- package/dist/transient-fault.d.mts +0 -1
- package/dist/transient-fault.mjs +21 -81
- package/dist/utm-deprecated.d.mts +0 -1
- package/dist/worker-code-check.d.mts +0 -1
- package/dist/worker-code-check.mjs +127 -83
- package/dist/worker-health.d.mts +0 -1
- package/dist/worker-health.mjs +16 -61
- package/dist/worker-http.d.mts +0 -1
- package/dist/worker-http.mjs +42 -234
- package/dist/worker-stats.d.mts +0 -1
- package/package.json +12 -5
- package/dist/capture-client.d.mts.map +0 -1
- package/dist/capture-client.mjs.map +0 -1
- package/dist/check-worker-code.d.mts.map +0 -1
- package/dist/check-worker-code.mjs.map +0 -1
- package/dist/cli-flags.d.mts.map +0 -1
- package/dist/cli-flags.mjs.map +0 -1
- package/dist/code-drift.d.mts.map +0 -1
- package/dist/code-drift.mjs +0 -284
- package/dist/code-drift.mjs.map +0 -1
- package/dist/command-line-census.d.mts.map +0 -1
- package/dist/command-line-census.mjs +0 -96
- package/dist/command-line-census.mjs.map +0 -1
- package/dist/compare-workers.d.mts.map +0 -1
- package/dist/compare-workers.mjs.map +0 -1
- package/dist/control-plane-isolation.d.mts.map +0 -1
- package/dist/control-plane-isolation.mjs +0 -67
- package/dist/control-plane-isolation.mjs.map +0 -1
- package/dist/deploy-worker.d.mts.map +0 -1
- package/dist/deploy-worker.mjs.map +0 -1
- package/dist/doctor.d.mts.map +0 -1
- package/dist/doctor.mjs.map +0 -1
- package/dist/fleet-consistency.d.mts.map +0 -1
- package/dist/fleet-consistency.mjs.map +0 -1
- package/dist/fleet-env.d.mts.map +0 -1
- package/dist/fleet-env.mjs.map +0 -1
- package/dist/fleet-scripts.d.mts.map +0 -1
- package/dist/fleet-scripts.mjs +0 -41
- package/dist/fleet-scripts.mjs.map +0 -1
- package/dist/git-safe-env.d.mts.map +0 -1
- package/dist/git-safe-env.mjs +0 -44
- package/dist/git-safe-env.mjs.map +0 -1
- package/dist/guest-run.d.mts.map +0 -1
- package/dist/guest-run.mjs +0 -164
- package/dist/guest-run.mjs.map +0 -1
- package/dist/host-address.d.mts.map +0 -1
- package/dist/host-address.mjs.map +0 -1
- package/dist/host-capacity.d.mts.map +0 -1
- package/dist/host-capacity.mjs.map +0 -1
- package/dist/host-metrics.d.mts.map +0 -1
- package/dist/host-metrics.mjs +0 -201
- package/dist/host-metrics.mjs.map +0 -1
- package/dist/index.d.ts.map +0 -1
- package/dist/index.js +0 -25
- package/dist/index.js.map +0 -1
- package/dist/local-vm.d.ts.map +0 -1
- package/dist/local-vm.js +0 -360
- package/dist/local-vm.js.map +0 -1
- package/dist/measure-guard.d.mts.map +0 -1
- package/dist/measure-guard.mjs +0 -73
- package/dist/measure-guard.mjs.map +0 -1
- package/dist/normalise-fleet.d.mts.map +0 -1
- package/dist/normalise-fleet.mjs +0 -76
- package/dist/normalise-fleet.mjs.map +0 -1
- package/dist/npm-cli-executable.d.mts.map +0 -1
- package/dist/npm-cli-executable.mjs +0 -159
- package/dist/npm-cli-executable.mjs.map +0 -1
- package/dist/probe-outcome.d.mts.map +0 -1
- package/dist/probe-outcome.mjs.map +0 -1
- package/dist/protocol-guard.d.mts.map +0 -1
- package/dist/protocol-guard.mjs +0 -121
- package/dist/protocol-guard.mjs.map +0 -1
- package/dist/source-walk.d.mts.map +0 -1
- package/dist/source-walk.mjs +0 -56
- package/dist/source-walk.mjs.map +0 -1
- package/dist/transient-fault.d.mts.map +0 -1
- package/dist/transient-fault.mjs.map +0 -1
- package/dist/utm-deprecated.d.mts.map +0 -1
- package/dist/utm-deprecated.mjs +0 -23
- package/dist/utm-deprecated.mjs.map +0 -1
- package/dist/worker-code-check.d.mts.map +0 -1
- package/dist/worker-code-check.mjs.map +0 -1
- package/dist/worker-health.d.mts.map +0 -1
- package/dist/worker-health.mjs.map +0 -1
- package/dist/worker-http.d.mts.map +0 -1
- package/dist/worker-http.mjs.map +0 -1
- package/dist/worker-stats.d.mts.map +0 -1
- package/dist/worker-stats.mjs +0 -143
- package/dist/worker-stats.mjs.map +0 -1
package/dist/capture-client.mjs
CHANGED
|
@@ -1,352 +1,190 @@
|
|
|
1
|
-
// @ts-check
|
|
2
|
-
/**
|
|
3
|
-
* ONE CAPTURE, TOLERANT OF THE SOCKET DYING UNDER IT — and the only place that knows how.
|
|
4
|
-
*
|
|
5
|
-
* The worker has stored completed captures under a caller-chosen id since the `captureId` work landed:
|
|
6
|
-
* `POST /capture {captureId}` then `GET /capture/<id>` returns the original response verbatim, so a lost
|
|
7
|
-
* socket costs a round trip instead of 12-520 s of real screen-reader work. That is the idempotency-key
|
|
8
|
-
* shape, for the reason payment APIs use it.
|
|
9
|
-
*
|
|
10
|
-
* TEN CLIENTS POST TO `/capture`. ONE USED THE RECOVERY. This repo's most expensive recurring shape is a
|
|
11
|
-
* remedy applied at one call site when the behaviour reaches several — `anchorToTop`, `ensureSpeechChannel`,
|
|
12
|
-
* `waitForAnnouncement`, `refreshBrowseBuffer` — and this is that shape at its largest here.
|
|
13
|
-
*
|
|
14
|
-
* MEASURED COST, 2026-08-28: `gate:stability` lost THREE canaries to `FAILED read ETIMEDOUT` across two
|
|
15
|
-
* runs — `filter-status-silent-solar/bad` on one box, then `form-error-silent/bad` and
|
|
16
|
-
* `disclosure-state-silent/good` on two others. Different pages, different machines, so it is the transport
|
|
17
|
-
* and not either. Each one turned the determinism gate INCONCLUSIVE while the capture it lost had already
|
|
18
|
-
* COMPLETED and was sitting in the worker's store, which is precisely what the store exists to serve back.
|
|
19
|
-
*
|
|
20
|
-
* The worker is bare metal on real Ethernet with real power management, which is why this arrived now: on
|
|
21
|
-
* three VMs sharing one Mac the socket was a virtual bridge and effectively lossless.
|
|
22
|
-
*
|
|
23
|
-
* MOVED HERE from `packages/lab/src/capture/capture-client.mjs` — architecture-audit.md §5, item 6: the
|
|
24
|
-
* product CLI sent no `captureId` at all, so this recovery path was unavailable to the one caller that is
|
|
25
|
-
* a user. `packages/cli` can depend on `@a11ign/screenreader-fleet` (it already does, for `requestJson`)
|
|
26
|
-
* but must never depend on `@a11ign/lab`, which is private and never published — so this is the
|
|
27
|
-
* seventh capture client and the last one, not an eighth copy of the recovery logic beside it.
|
|
28
|
-
*/
|
|
29
1
|
import { randomUUID } from "node:crypto";
|
|
30
|
-
import {
|
|
2
|
+
import { CAPTURE_CLIENT_TIMEOUT_MS, requestJson } from "./worker-http.mjs";
|
|
31
3
|
import { isTransient } from "./transient-fault.mjs";
|
|
32
|
-
|
|
33
|
-
const
|
|
34
|
-
|
|
35
|
-
|
|
36
|
-
|
|
37
|
-
* Kept because a protocol change wants a way back that does not need a deploy, and removed only once a
|
|
38
|
-
* corpus run has gone through the async path end to end.
|
|
39
|
-
*/
|
|
40
|
-
/**
|
|
41
|
-
* READ PER CALL, NOT AT MODULE LOAD, and the difference is not stylistic. A value fixed at import is one
|
|
42
|
-
* no test can vary and no process can change -- `fileProductVersion` memoised Edge's version that way and
|
|
43
|
-
* stamped five days of captures with a build they were not taken under. So it is a PARAMETER with an env
|
|
44
|
-
* default, which also lets both paths be driven from one test file.
|
|
45
|
-
*/
|
|
46
|
-
const syncByEnv = () => process.env.A11Y_SYNC_CAPTURE === "1";
|
|
47
|
-
/** Accepting a capture is a handshake, not the work: it either answers in seconds or the box is unwell. */
|
|
48
|
-
const ACCEPT_TIMEOUT_MS = 30_000;
|
|
49
|
-
/** Between polls. Short enough that a 12 s capture is not padded, long enough not to hammer a busy box. */
|
|
50
|
-
const POLL_MS = 2_000;
|
|
51
|
-
/** A read of an array already in memory; if this cannot answer, the guest's event loop is blocked. */
|
|
52
|
-
const PROGRESS_TIMEOUT_MS = 10_000;
|
|
53
|
-
/**
|
|
54
|
-
* Consecutive failed POLLS before giving up.
|
|
55
|
-
*
|
|
56
|
-
* Not one: a single dropped poll is precisely the transport fault this design exists to survive, and
|
|
57
|
-
* treating it as a failed capture would reintroduce the defect through the new door. Not unbounded either,
|
|
58
|
-
* or a box that has genuinely gone would be waited on for the whole capture budget.
|
|
59
|
-
*/
|
|
4
|
+
const RECOVERY_TIMEOUT_MS = 30000;
|
|
5
|
+
const syncByEnv = ()=>"1" === process.env.A11Y_SYNC_CAPTURE;
|
|
6
|
+
const ACCEPT_TIMEOUT_MS = 30000;
|
|
7
|
+
const POLL_MS = 2000;
|
|
8
|
+
const PROGRESS_TIMEOUT_MS = 10000;
|
|
60
9
|
const MAX_POLL_FAILURES = 5;
|
|
61
|
-
const sleep = (
|
|
62
|
-
const base = (
|
|
63
|
-
|
|
64
|
-
|
|
65
|
-
*
|
|
66
|
-
* Every wait inside the poll loop used its own fixed constant regardless of how little of `timeoutMs`
|
|
67
|
-
* remained: a two-second sleep, a ten-second progress read and a thirty-second result read, none clipped
|
|
68
|
-
* to what was actually left. A caller asking for `timeoutMs: 20` measured 2,012 ms before an answer,
|
|
69
|
-
* because the unconditional sleep ran to completion first regardless of the deadline it was about to blow
|
|
70
|
-
* past. This is the one place that number is computed, so every wait below shares one clock.
|
|
71
|
-
*
|
|
72
|
-
* @param {number} deadline
|
|
73
|
-
*/
|
|
74
|
-
const remaining = (deadline) => Math.max(0, deadline - Date.now());
|
|
75
|
-
/**
|
|
76
|
-
* Ask the worker for a capture we already paid for but may not have received.
|
|
77
|
-
*
|
|
78
|
-
* Returns null when there is nothing to recover — a worker predating the endpoint (404 from the router's
|
|
79
|
-
* fallback), one that restarted and lost its memory, or a capture still running. Null means "capture
|
|
80
|
-
* again", which is what every client did before the endpoint existed.
|
|
81
|
-
*
|
|
82
|
-
* A recovered FAILURE is RETURNED rather than swallowed, so a replay is indistinguishable from the original
|
|
83
|
-
* response. That keeps the worker's `fault` code — the thing it worked out, and which we would otherwise
|
|
84
|
-
* replace with "no answer" — and lets the caller's own classification decide, as it would have all along.
|
|
85
|
-
* (The dataset runner's version THREW here, because it wrapped a `fetchJson` that rejects on non-2xx;
|
|
86
|
-
* `requestJson` resolves instead, so the same intent is expressed by returning the response.)
|
|
87
|
-
*
|
|
88
|
-
* @param {string} worker @param {string} captureId
|
|
89
|
-
*/
|
|
90
|
-
export async function recoverCapture(worker, captureId) {
|
|
10
|
+
const sleep = (ms)=>new Promise((r)=>setTimeout(r, ms));
|
|
11
|
+
const base = (worker)=>String(worker).replace(/\/$/, "");
|
|
12
|
+
const remaining = (deadline)=>Math.max(0, deadline - Date.now());
|
|
13
|
+
async function recoverCapture(worker, captureId) {
|
|
91
14
|
let response;
|
|
92
15
|
try {
|
|
93
|
-
response = await requestJson(`${base(worker)}/capture/${captureId}`, {
|
|
94
|
-
|
|
95
|
-
|
|
96
|
-
|
|
97
|
-
// capturing, which is what it would have done anyway.
|
|
98
|
-
void error;
|
|
16
|
+
response = await requestJson(`${base(worker)}/capture/${captureId}`, {
|
|
17
|
+
timeoutMs: RECOVERY_TIMEOUT_MS
|
|
18
|
+
});
|
|
19
|
+
} catch (error) {
|
|
99
20
|
return null;
|
|
100
21
|
}
|
|
101
|
-
|
|
102
|
-
|
|
103
|
-
|
|
104
|
-
|
|
105
|
-
if (response.status === 500)
|
|
106
|
-
return response;
|
|
107
|
-
// 404 from the endpoint, or from an older worker's router fallback: nothing kept, so capture again.
|
|
108
|
-
if (response.status === 404)
|
|
109
|
-
return null;
|
|
110
|
-
if (!response.ok)
|
|
111
|
-
return null;
|
|
112
|
-
// "Still running" and "never heard of it" are DIFFERENT ANSWERS and must stay that way. Neither is
|
|
113
|
-
// recoverable here, but only one means the work is still being done.
|
|
114
|
-
if ( /** @type {any} */(response.json)?.state === "running")
|
|
115
|
-
return null;
|
|
22
|
+
if (500 === response.status) return response;
|
|
23
|
+
if (404 === response.status) return null;
|
|
24
|
+
if (!response.ok) return null;
|
|
25
|
+
if (response.json?.state === "running") return null;
|
|
116
26
|
return response;
|
|
117
27
|
}
|
|
118
|
-
|
|
119
|
-
* POST a capture, and on a TRANSIENT failure ask whether it actually finished before paying again.
|
|
120
|
-
*
|
|
121
|
-
* Returns what `requestJson` returns — `{status, ok, text, json}` — plus `recovered`, so this is a
|
|
122
|
-
* drop-in at the call sites that already POST `/capture` and read `response.ok` / `body.error`. Keeping
|
|
123
|
-
* their own classification is deliberate: a shared client that also decided what counts as a failure would
|
|
124
|
-
* be changing ten behaviours at once while claiming to change one.
|
|
125
|
-
*
|
|
126
|
-
* `requestJson` RESOLVES on an HTTP error and REJECTS only on a transport one, so the `catch` below is
|
|
127
|
-
* reached exactly by the case this exists for — `ETIMEDOUT`, `ECONNRESET`, a socket dying mid-answer.
|
|
128
|
-
*
|
|
129
|
-
* `waitForWorker` is deliberately NOT called here. The dataset runner has one, tuned to a corpus run's
|
|
130
|
-
* tolerance for a box gone for minutes; a gate wants an answer quickly. So this asks once, immediately,
|
|
131
|
-
* and a caller that wants to wait first passes `beforeRecovery`.
|
|
132
|
-
*
|
|
133
|
-
* @param {{ worker: string, body: object, timeoutMs?: number,
|
|
134
|
-
* beforeRecovery?: (error: unknown) => Promise<void>,
|
|
135
|
-
* onProgress?: (progress: object) => void, sync?: boolean }} request
|
|
136
|
-
*/
|
|
137
|
-
export async function captureTolerantly({ worker, body, timeoutMs = CAPTURE_CLIENT_TIMEOUT_MS, beforeRecovery, onProgress, sync = syncByEnv() }) {
|
|
28
|
+
async function captureTolerantly({ worker, body, timeoutMs = CAPTURE_CLIENT_TIMEOUT_MS, beforeRecovery, onProgress, sync = syncByEnv() }) {
|
|
138
29
|
const captureId = randomUUID();
|
|
139
|
-
if (!sync)
|
|
140
|
-
|
|
141
|
-
|
|
142
|
-
|
|
30
|
+
if (!sync) return pollForResult({
|
|
31
|
+
worker,
|
|
32
|
+
body,
|
|
33
|
+
captureId,
|
|
34
|
+
timeoutMs,
|
|
35
|
+
onProgress
|
|
36
|
+
});
|
|
143
37
|
process.stderr.write("A11Y_SYNC_CAPTURE — holding one connection open for this capture\n");
|
|
144
38
|
try {
|
|
145
|
-
return {
|
|
146
|
-
|
|
147
|
-
|
|
148
|
-
|
|
149
|
-
|
|
150
|
-
|
|
151
|
-
|
|
152
|
-
|
|
153
|
-
|
|
154
|
-
|
|
39
|
+
return {
|
|
40
|
+
...await post(worker, {
|
|
41
|
+
...body,
|
|
42
|
+
captureId
|
|
43
|
+
}, timeoutMs),
|
|
44
|
+
recovered: false,
|
|
45
|
+
pollsSurvived: 0
|
|
46
|
+
};
|
|
47
|
+
} catch (error) {
|
|
48
|
+
if (!isTransient(error)) throw error;
|
|
49
|
+
if (beforeRecovery) await beforeRecovery(error);
|
|
155
50
|
const recovered = await recoverCapture(worker, captureId);
|
|
156
|
-
|
|
157
|
-
|
|
158
|
-
|
|
159
|
-
|
|
160
|
-
|
|
161
|
-
return {
|
|
51
|
+
if (recovered) return {
|
|
52
|
+
...recovered,
|
|
53
|
+
recovered: true,
|
|
54
|
+
pollsSurvived: 0
|
|
55
|
+
};
|
|
56
|
+
return {
|
|
57
|
+
...await post(worker, {
|
|
58
|
+
...body,
|
|
59
|
+
captureId: randomUUID()
|
|
60
|
+
}, timeoutMs),
|
|
61
|
+
recovered: false,
|
|
62
|
+
pollsSurvived: 0
|
|
63
|
+
};
|
|
162
64
|
}
|
|
163
65
|
}
|
|
164
|
-
/**
|
|
165
|
-
* DISPATCH, THEN POLL — the async path, and the reason this module exists.
|
|
166
|
-
*
|
|
167
|
-
* `POST {async:true}` returns 202 in milliseconds, so no connection is held while NVDA reads a page. The
|
|
168
|
-
* result is collected from the store with `GET /capture/<id>`, which is the endpoint that has existed for
|
|
169
|
-
* this shape all along and was only ever reached after a failure. **The recovery path is now the normal
|
|
170
|
-
* path**, which is what stops it rotting: a route that runs only when something breaks is one nobody
|
|
171
|
-
* notices has broken.
|
|
172
|
-
*
|
|
173
|
-
* A dropped poll costs one round trip and is simply retried; the capture is unaffected because nothing is
|
|
174
|
-
* riding on that socket. That is the whole difference from the synchronous form, where the answer existed
|
|
175
|
-
* only in the connection that was carrying it.
|
|
176
|
-
*/
|
|
177
|
-
/**
|
|
178
|
-
* @param {{ worker: string, body: object, captureId: string, timeoutMs: number,
|
|
179
|
-
* onProgress?: (progress: object) => void }} request
|
|
180
|
-
*/
|
|
181
66
|
async function pollForResult({ worker, body, captureId, timeoutMs, onProgress }) {
|
|
182
67
|
const deadline = Date.now() + timeoutMs;
|
|
183
68
|
let accepted;
|
|
184
69
|
try {
|
|
185
|
-
accepted = await post(worker, {
|
|
186
|
-
|
|
187
|
-
|
|
188
|
-
|
|
189
|
-
|
|
70
|
+
accepted = await post(worker, {
|
|
71
|
+
...body,
|
|
72
|
+
captureId,
|
|
73
|
+
async: true
|
|
74
|
+
}, ACCEPT_TIMEOUT_MS);
|
|
75
|
+
} catch (error) {
|
|
76
|
+
if (!isTransient(error)) throw error;
|
|
190
77
|
const reconciled = await reconcileLostAcceptance(worker, captureId, deadline);
|
|
191
|
-
if (reconciled.state
|
|
192
|
-
|
|
193
|
-
|
|
194
|
-
|
|
195
|
-
|
|
196
|
-
|
|
197
|
-
|
|
198
|
-
|
|
199
|
-
|
|
200
|
-
|
|
201
|
-
|
|
202
|
-
|
|
203
|
-
// was lost, not the acceptance. Fall through to the ordinary poll loop below exactly as a received
|
|
204
|
-
// 202 would have.
|
|
78
|
+
if ("done" === reconciled.state) return {
|
|
79
|
+
...reconciled.response,
|
|
80
|
+
recovered: true,
|
|
81
|
+
pollsSurvived: 0
|
|
82
|
+
};
|
|
83
|
+
if ("unknown" === reconciled.state) return pollForResult({
|
|
84
|
+
worker,
|
|
85
|
+
body,
|
|
86
|
+
captureId: randomUUID(),
|
|
87
|
+
timeoutMs: Math.max(0, deadline - Date.now()),
|
|
88
|
+
onProgress
|
|
89
|
+
});
|
|
205
90
|
}
|
|
206
|
-
|
|
207
|
-
|
|
208
|
-
|
|
209
|
-
|
|
210
|
-
|
|
211
|
-
return awaitCompletion({
|
|
91
|
+
if (accepted && 202 !== accepted.status) return {
|
|
92
|
+
...accepted,
|
|
93
|
+
recovered: false,
|
|
94
|
+
pollsSurvived: 0
|
|
95
|
+
};
|
|
96
|
+
return awaitCompletion({
|
|
97
|
+
worker,
|
|
98
|
+
captureId,
|
|
99
|
+
deadline,
|
|
100
|
+
timeoutMs,
|
|
101
|
+
onProgress
|
|
102
|
+
});
|
|
212
103
|
}
|
|
213
|
-
/**
|
|
214
|
-
* ASKING ABOUT A CAPTURE WE MAY OR MAY NOT HAVE STARTED, using the SAME three-way discrimination a
|
|
215
|
-
* dropped poll already relies on (`pollOnce`) rather than `recoverCapture`, which collapses "running"
|
|
216
|
-
* and "unknown" into one null and cannot tell them apart -- the exact distinction this exists to make.
|
|
217
|
-
*
|
|
218
|
-
* A 404 here is not quite `pollOnce`'s documented "worker restarted mid-capture": no 202 was ever
|
|
219
|
-
* received, so there is nothing to have restarted AWAY FROM. It means the POST itself never reached the
|
|
220
|
-
* worker, which is what makes minting a fresh id safe once this returns "unknown" and not before.
|
|
221
|
-
*
|
|
222
|
-
* The audit's own caveat applies in principle: "unknown does not prove execution never occurred" if the
|
|
223
|
-
* result were evicted (§14.4) before this ever asks — but that needs seven OTHER captures to finish on
|
|
224
|
-
* this worker in the seconds between the lost 202 and this reconciliation, which the worker's own `busy`
|
|
225
|
-
* gate (one capture at a time) makes impossible while nothing else is running under this id.
|
|
226
|
-
*
|
|
227
|
-
* @param {string} worker @param {string} captureId @param {number} deadline
|
|
228
|
-
* @returns {Promise<{ state: "running" } | { state: "unknown" } | { state: "done", response: any }>}
|
|
229
|
-
*/
|
|
230
104
|
async function reconcileLostAcceptance(worker, captureId, deadline) {
|
|
231
105
|
let lastError;
|
|
232
|
-
while
|
|
106
|
+
while(Date.now() < deadline){
|
|
233
107
|
const poll = await pollOnce(worker, captureId, Math.min(RECOVERY_TIMEOUT_MS, remaining(deadline)));
|
|
234
|
-
if (poll.running)
|
|
235
|
-
|
|
236
|
-
|
|
237
|
-
|
|
238
|
-
|
|
239
|
-
|
|
108
|
+
if (poll.running) return {
|
|
109
|
+
state: "running"
|
|
110
|
+
};
|
|
111
|
+
if (poll.done) return {
|
|
112
|
+
state: "done",
|
|
113
|
+
response: poll.response
|
|
114
|
+
};
|
|
115
|
+
if (poll.lost) return {
|
|
116
|
+
state: "unknown"
|
|
117
|
+
};
|
|
240
118
|
lastError = poll.error;
|
|
241
119
|
await sleep(Math.min(POLL_MS, remaining(deadline)));
|
|
242
120
|
}
|
|
243
|
-
|
|
244
|
-
|
|
245
|
-
|
|
246
|
-
throw lastError ?? Object.assign(new Error(`could not confirm capture ${captureId} was accepted, within its remaining budget`), { code: "ETIMEDOUT" });
|
|
121
|
+
throw lastError ?? Object.assign(new Error(`could not confirm capture ${captureId} was accepted, within its remaining budget`), {
|
|
122
|
+
code: "ETIMEDOUT"
|
|
123
|
+
});
|
|
247
124
|
}
|
|
248
|
-
/**
|
|
249
|
-
* DISPATCH, THEN POLL — the async path, and the reason this module exists.
|
|
250
|
-
*
|
|
251
|
-
* `POST {async:true}` returns 202 in milliseconds, so no connection is held while NVDA reads a page. The
|
|
252
|
-
* result is collected from the store with `GET /capture/<id>`, which is the endpoint that has existed for
|
|
253
|
-
* this shape all along and was only ever reached after a failure. **The recovery path is now the normal
|
|
254
|
-
* path**, which is what stops it rotting: a route that runs only when something breaks is one nobody
|
|
255
|
-
* notices has broken.
|
|
256
|
-
*
|
|
257
|
-
* A dropped poll costs one round trip and is simply retried; the capture is unaffected because nothing is
|
|
258
|
-
* riding on that socket. That is the whole difference from the synchronous form, where the answer existed
|
|
259
|
-
* only in the connection that was carrying it.
|
|
260
|
-
*
|
|
261
|
-
* @param {{ worker: string, captureId: string, deadline: number, timeoutMs: number,
|
|
262
|
-
* onProgress?: (progress: object) => void }} request
|
|
263
|
-
*/
|
|
264
125
|
async function awaitCompletion({ worker, captureId, deadline, timeoutMs, onProgress }) {
|
|
265
126
|
let transportFailures = 0;
|
|
266
127
|
let survived = 0;
|
|
267
|
-
while
|
|
268
|
-
// CLIPPED, EACH TIME, TO WHAT IS LEFT — architecture-audit.md §14.5. A budget of 20 ms must not spend
|
|
269
|
-
// a full 2 s sleeping before it is even allowed to check the clock again.
|
|
128
|
+
while(Date.now() < deadline){
|
|
270
129
|
await sleep(Math.min(POLL_MS, remaining(deadline)));
|
|
271
|
-
if (onProgress)
|
|
272
|
-
await readProgress(worker, onProgress, Math.min(PROGRESS_TIMEOUT_MS, remaining(deadline)));
|
|
130
|
+
if (onProgress) await readProgress(worker, onProgress, Math.min(PROGRESS_TIMEOUT_MS, remaining(deadline)));
|
|
273
131
|
const poll = await pollOnce(worker, captureId, Math.min(RECOVERY_TIMEOUT_MS, remaining(deadline)));
|
|
274
|
-
if (poll.done) {
|
|
275
|
-
|
|
276
|
-
|
|
277
|
-
|
|
278
|
-
|
|
279
|
-
|
|
280
|
-
|
|
281
|
-
|
|
282
|
-
|
|
283
|
-
+ "mid-capture, so the work is gone and the case must be re-issued"), { code: "CAPTURE_LOST" });
|
|
284
|
-
}
|
|
285
|
-
// A FAILED POLL IS NOT A FAILED CAPTURE, and conflating them would give back the defect this design
|
|
286
|
-
// removes: the worker is still working, we merely could not ask. Retried until a RUN of them says the
|
|
287
|
-
// box has gone, rather than on the first one -- a single dropped request is the exact fault this exists
|
|
288
|
-
// to survive.
|
|
289
|
-
if (poll.unreachable)
|
|
290
|
-
survived += 1;
|
|
132
|
+
if (poll.done) return {
|
|
133
|
+
...poll.response,
|
|
134
|
+
recovered: false,
|
|
135
|
+
pollsSurvived: survived
|
|
136
|
+
};
|
|
137
|
+
if (poll.lost) throw Object.assign(new Error(`worker forgot capture ${captureId} after accepting it — it restarted mid-capture, so the work is gone and the case must be re-issued`), {
|
|
138
|
+
code: "CAPTURE_LOST"
|
|
139
|
+
});
|
|
140
|
+
if (poll.unreachable) survived += 1;
|
|
291
141
|
transportFailures = poll.unreachable ? transportFailures + 1 : 0;
|
|
292
|
-
if (transportFailures >= MAX_POLL_FAILURES)
|
|
293
|
-
throw poll.error;
|
|
142
|
+
if (transportFailures >= MAX_POLL_FAILURES) throw poll.error;
|
|
294
143
|
}
|
|
295
|
-
throw Object.assign(new Error(`capture ${captureId} did not finish within ${timeoutMs} ms`), {
|
|
144
|
+
throw Object.assign(new Error(`capture ${captureId} did not finish within ${timeoutMs} ms`), {
|
|
145
|
+
code: "ETIMEDOUT"
|
|
146
|
+
});
|
|
296
147
|
}
|
|
297
|
-
/**
|
|
298
|
-
* ONE POLL, WITH FOUR DISTINCT ANSWERS — and keeping them apart is the whole job.
|
|
299
|
-
*
|
|
300
|
-
* done the capture finished (200 or 500); the 500 carries the worker's own fault code
|
|
301
|
-
* lost 404 AFTER a 202 acceptance: the worker restarted, the work is gone, re-issue
|
|
302
|
-
* running 202; keep waiting
|
|
303
|
-
* unreachable we could not ask; the capture is unaffected
|
|
304
|
-
*
|
|
305
|
-
* NOT `recoverCapture`, and that is the correction. It SWALLOWS a transport error and returns null, which
|
|
306
|
-
* is right for its own job — "we already failed, is the result there?" — and wrong here, because it makes
|
|
307
|
-
* "could not ask" indistinguishable from "still running". Written that way first, and the retry counter
|
|
308
|
-
* built on it was unreachable: found by mutation, since deleting the retry changed nothing.
|
|
309
|
-
*
|
|
310
|
-
* @param {string} worker @param {string} captureId @param {number} [timeoutMs] clipped to the operation's
|
|
311
|
-
* remaining budget by the caller — see `remaining()` — so this read cannot outlive the deadline it is
|
|
312
|
-
* answering to on its own.
|
|
313
|
-
*/
|
|
314
148
|
async function pollOnce(worker, captureId, timeoutMs = RECOVERY_TIMEOUT_MS) {
|
|
315
149
|
let response;
|
|
316
150
|
try {
|
|
317
|
-
response = await requestJson(`${base(worker)}/capture/${captureId}`, {
|
|
318
|
-
|
|
319
|
-
|
|
320
|
-
|
|
151
|
+
response = await requestJson(`${base(worker)}/capture/${captureId}`, {
|
|
152
|
+
timeoutMs
|
|
153
|
+
});
|
|
154
|
+
} catch (error) {
|
|
155
|
+
return {
|
|
156
|
+
unreachable: true,
|
|
157
|
+
error
|
|
158
|
+
};
|
|
321
159
|
}
|
|
322
|
-
if (response.status
|
|
323
|
-
|
|
324
|
-
|
|
325
|
-
|
|
326
|
-
|
|
327
|
-
|
|
328
|
-
if (response.ok || response.status
|
|
329
|
-
|
|
330
|
-
|
|
331
|
-
|
|
332
|
-
return {
|
|
160
|
+
if (404 === response.status) return {
|
|
161
|
+
lost: true
|
|
162
|
+
};
|
|
163
|
+
if (202 === response.status) return {
|
|
164
|
+
running: true
|
|
165
|
+
};
|
|
166
|
+
if (response.ok || 500 === response.status) return {
|
|
167
|
+
done: true,
|
|
168
|
+
response
|
|
169
|
+
};
|
|
170
|
+
return {
|
|
171
|
+
unreachable: true,
|
|
172
|
+
error: new Error(`unexpected ${response.status} polling ${captureId}`)
|
|
173
|
+
};
|
|
333
174
|
}
|
|
334
|
-
|
|
335
|
-
async function readProgress(/** @type {string} */ worker, /** @type {(p: object) => void} */ onProgress,
|
|
336
|
-
/** @type {number} */ timeoutMs = PROGRESS_TIMEOUT_MS) {
|
|
175
|
+
async function readProgress(worker, onProgress, timeoutMs = PROGRESS_TIMEOUT_MS) {
|
|
337
176
|
try {
|
|
338
|
-
const { json } = await requestJson(`${base(worker)}/progress`, {
|
|
339
|
-
|
|
340
|
-
|
|
341
|
-
|
|
342
|
-
catch (error) {
|
|
343
|
-
// Progress is a convenience; failing to read it must never fail a capture that is going fine.
|
|
344
|
-
void error;
|
|
345
|
-
}
|
|
177
|
+
const { json } = await requestJson(`${base(worker)}/progress`, {
|
|
178
|
+
timeoutMs
|
|
179
|
+
});
|
|
180
|
+
if (json && "object" == typeof json) onProgress(json);
|
|
181
|
+
} catch (error) {}
|
|
346
182
|
}
|
|
347
|
-
/** @param {string} worker @param {object} body @param {number} timeoutMs */
|
|
348
183
|
function post(worker, body, timeoutMs) {
|
|
349
|
-
|
|
350
|
-
|
|
184
|
+
return requestJson(`${base(worker)}/capture`, {
|
|
185
|
+
method: "POST",
|
|
186
|
+
body,
|
|
187
|
+
timeoutMs
|
|
188
|
+
});
|
|
351
189
|
}
|
|
352
|
-
|
|
190
|
+
export { captureTolerantly, recoverCapture };
|