@a11ign/screenreader-fleet 0.2.0 → 0.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (130) hide show
  1. package/dist/capture-client.d.mts +0 -1
  2. package/dist/capture-client.mjs +144 -306
  3. package/dist/check-worker-code.d.mts +0 -1
  4. package/dist/check-worker-code.mjs +42 -110
  5. package/dist/cli-flags.d.mts +0 -1
  6. package/dist/cli-flags.mjs +33 -179
  7. package/dist/code-drift.d.mts +4 -4
  8. package/dist/command-line-census.d.mts +0 -1
  9. package/dist/compare-workers.d.mts +0 -1
  10. package/dist/compare-workers.mjs +383 -255
  11. package/dist/control-plane-isolation.d.mts +0 -1
  12. package/dist/doctor.d.mts +0 -1
  13. package/dist/doctor.mjs +361 -809
  14. package/dist/fleet-consistency.d.mts +0 -1
  15. package/dist/fleet-consistency.mjs +155 -379
  16. package/dist/fleet-env.d.mts +1 -2
  17. package/dist/fleet-env.mjs +148 -438
  18. package/dist/fleet-scripts.d.mts +0 -1
  19. package/dist/git-safe-env.d.mts +0 -1
  20. package/dist/guest-run.d.mts +0 -1
  21. package/dist/host-address.d.mts +0 -1
  22. package/dist/host-address.mjs +19 -90
  23. package/dist/host-capacity.d.mts +0 -1
  24. package/dist/host-capacity.mjs +22 -136
  25. package/dist/host-metrics.d.mts +0 -1
  26. package/dist/index.d.ts +0 -1
  27. package/dist/index.mjs +233 -0
  28. package/dist/local-vm.d.ts +0 -1
  29. package/dist/measure-guard.d.mts +0 -1
  30. package/dist/normalise-fleet.d.mts +0 -1
  31. package/dist/npm-cli-executable.d.mts +0 -1
  32. package/dist/probe-outcome.d.mts +0 -1
  33. package/dist/probe-outcome.mjs +50 -96
  34. package/dist/protocol-guard.d.mts +0 -1
  35. package/dist/source-walk.d.mts +0 -1
  36. package/dist/src_fleet-scripts_mjs.mjs +16 -0
  37. package/dist/src_git-safe-env_mjs.mjs +9 -0
  38. package/dist/transient-fault.d.mts +0 -1
  39. package/dist/transient-fault.mjs +21 -81
  40. package/dist/utm-deprecated.d.mts +0 -1
  41. package/dist/worker-code-check.d.mts +0 -1
  42. package/dist/worker-code-check.mjs +127 -83
  43. package/dist/worker-health.d.mts +0 -1
  44. package/dist/worker-health.mjs +16 -61
  45. package/dist/worker-http.d.mts +0 -1
  46. package/dist/worker-http.mjs +42 -234
  47. package/dist/worker-stats.d.mts +0 -1
  48. package/package.json +13 -7
  49. package/src/local-worker/worker-ctl.sh +1 -1
  50. package/src/provisioning/bootstrap-windows-worker.ps1 +1 -1
  51. package/dist/capture-client.d.mts.map +0 -1
  52. package/dist/capture-client.mjs.map +0 -1
  53. package/dist/check-worker-code.d.mts.map +0 -1
  54. package/dist/check-worker-code.mjs.map +0 -1
  55. package/dist/cli-flags.d.mts.map +0 -1
  56. package/dist/cli-flags.mjs.map +0 -1
  57. package/dist/code-drift.d.mts.map +0 -1
  58. package/dist/code-drift.mjs +0 -284
  59. package/dist/code-drift.mjs.map +0 -1
  60. package/dist/command-line-census.d.mts.map +0 -1
  61. package/dist/command-line-census.mjs +0 -96
  62. package/dist/command-line-census.mjs.map +0 -1
  63. package/dist/compare-workers.d.mts.map +0 -1
  64. package/dist/compare-workers.mjs.map +0 -1
  65. package/dist/control-plane-isolation.d.mts.map +0 -1
  66. package/dist/control-plane-isolation.mjs +0 -67
  67. package/dist/control-plane-isolation.mjs.map +0 -1
  68. package/dist/deploy-worker.d.mts +0 -3
  69. package/dist/deploy-worker.d.mts.map +0 -1
  70. package/dist/deploy-worker.mjs +0 -333
  71. package/dist/deploy-worker.mjs.map +0 -1
  72. package/dist/doctor.d.mts.map +0 -1
  73. package/dist/doctor.mjs.map +0 -1
  74. package/dist/fleet-consistency.d.mts.map +0 -1
  75. package/dist/fleet-consistency.mjs.map +0 -1
  76. package/dist/fleet-env.d.mts.map +0 -1
  77. package/dist/fleet-env.mjs.map +0 -1
  78. package/dist/fleet-scripts.d.mts.map +0 -1
  79. package/dist/fleet-scripts.mjs +0 -41
  80. package/dist/fleet-scripts.mjs.map +0 -1
  81. package/dist/git-safe-env.d.mts.map +0 -1
  82. package/dist/git-safe-env.mjs +0 -44
  83. package/dist/git-safe-env.mjs.map +0 -1
  84. package/dist/guest-run.d.mts.map +0 -1
  85. package/dist/guest-run.mjs +0 -164
  86. package/dist/guest-run.mjs.map +0 -1
  87. package/dist/host-address.d.mts.map +0 -1
  88. package/dist/host-address.mjs.map +0 -1
  89. package/dist/host-capacity.d.mts.map +0 -1
  90. package/dist/host-capacity.mjs.map +0 -1
  91. package/dist/host-metrics.d.mts.map +0 -1
  92. package/dist/host-metrics.mjs +0 -201
  93. package/dist/host-metrics.mjs.map +0 -1
  94. package/dist/index.d.ts.map +0 -1
  95. package/dist/index.js +0 -25
  96. package/dist/index.js.map +0 -1
  97. package/dist/local-vm.d.ts.map +0 -1
  98. package/dist/local-vm.js +0 -360
  99. package/dist/local-vm.js.map +0 -1
  100. package/dist/measure-guard.d.mts.map +0 -1
  101. package/dist/measure-guard.mjs +0 -73
  102. package/dist/measure-guard.mjs.map +0 -1
  103. package/dist/normalise-fleet.d.mts.map +0 -1
  104. package/dist/normalise-fleet.mjs +0 -76
  105. package/dist/normalise-fleet.mjs.map +0 -1
  106. package/dist/npm-cli-executable.d.mts.map +0 -1
  107. package/dist/npm-cli-executable.mjs +0 -159
  108. package/dist/npm-cli-executable.mjs.map +0 -1
  109. package/dist/probe-outcome.d.mts.map +0 -1
  110. package/dist/probe-outcome.mjs.map +0 -1
  111. package/dist/protocol-guard.d.mts.map +0 -1
  112. package/dist/protocol-guard.mjs +0 -121
  113. package/dist/protocol-guard.mjs.map +0 -1
  114. package/dist/source-walk.d.mts.map +0 -1
  115. package/dist/source-walk.mjs +0 -56
  116. package/dist/source-walk.mjs.map +0 -1
  117. package/dist/transient-fault.d.mts.map +0 -1
  118. package/dist/transient-fault.mjs.map +0 -1
  119. package/dist/utm-deprecated.d.mts.map +0 -1
  120. package/dist/utm-deprecated.mjs +0 -23
  121. package/dist/utm-deprecated.mjs.map +0 -1
  122. package/dist/worker-code-check.d.mts.map +0 -1
  123. package/dist/worker-code-check.mjs.map +0 -1
  124. package/dist/worker-health.d.mts.map +0 -1
  125. package/dist/worker-health.mjs.map +0 -1
  126. package/dist/worker-http.d.mts.map +0 -1
  127. package/dist/worker-http.mjs.map +0 -1
  128. package/dist/worker-stats.d.mts.map +0 -1
  129. package/dist/worker-stats.mjs +0 -143
  130. package/dist/worker-stats.mjs.map +0 -1
@@ -46,4 +46,3 @@ export function captureTolerantly({ worker, body, timeoutMs, beforeRecovery, onP
46
46
  onProgress?: (progress: object) => void;
47
47
  sync?: boolean;
48
48
  }): Promise<any>;
49
- //# sourceMappingURL=capture-client.d.mts.map
@@ -1,352 +1,190 @@
1
- // @ts-check
2
- /**
3
- * ONE CAPTURE, TOLERANT OF THE SOCKET DYING UNDER IT — and the only place that knows how.
4
- *
5
- * The worker has stored completed captures under a caller-chosen id since the `captureId` work landed:
6
- * `POST /capture {captureId}` then `GET /capture/<id>` returns the original response verbatim, so a lost
7
- * socket costs a round trip instead of 12-520 s of real screen-reader work. That is the idempotency-key
8
- * shape, for the reason payment APIs use it.
9
- *
10
- * TEN CLIENTS POST TO `/capture`. ONE USED THE RECOVERY. This repo's most expensive recurring shape is a
11
- * remedy applied at one call site when the behaviour reaches several — `anchorToTop`, `ensureSpeechChannel`,
12
- * `waitForAnnouncement`, `refreshBrowseBuffer` — and this is that shape at its largest here.
13
- *
14
- * MEASURED COST, 2026-08-28: `gate:stability` lost THREE canaries to `FAILED read ETIMEDOUT` across two
15
- * runs — `filter-status-silent-solar/bad` on one box, then `form-error-silent/bad` and
16
- * `disclosure-state-silent/good` on two others. Different pages, different machines, so it is the transport
17
- * and not either. Each one turned the determinism gate INCONCLUSIVE while the capture it lost had already
18
- * COMPLETED and was sitting in the worker's store, which is precisely what the store exists to serve back.
19
- *
20
- * The worker is bare metal on real Ethernet with real power management, which is why this arrived now: on
21
- * three VMs sharing one Mac the socket was a virtual bridge and effectively lossless.
22
- *
23
- * MOVED HERE from `packages/lab/src/capture/capture-client.mjs` — architecture-audit.md §5, item 6: the
24
- * product CLI sent no `captureId` at all, so this recovery path was unavailable to the one caller that is
25
- * a user. `packages/cli` can depend on `@a11ign/screenreader-fleet` (it already does, for `requestJson`)
26
- * but must never depend on `@a11ign/lab`, which is private and never published — so this is the
27
- * seventh capture client and the last one, not an eighth copy of the recovery logic beside it.
28
- */
29
1
  import { randomUUID } from "node:crypto";
30
- import { requestJson, CAPTURE_CLIENT_TIMEOUT_MS } from "./worker-http.mjs";
2
+ import { CAPTURE_CLIENT_TIMEOUT_MS, requestJson } from "./worker-http.mjs";
31
3
  import { isTransient } from "./transient-fault.mjs";
32
- /** Long enough to survive a worker that is briefly busy, short enough not to double a capture's cost. */
33
- const RECOVERY_TIMEOUT_MS = 30_000;
34
- /**
35
- * THE ESCAPE HATCH BACK TO THE SYNCHRONOUS PATH, and it says so when used.
36
- *
37
- * Kept because a protocol change wants a way back that does not need a deploy, and removed only once a
38
- * corpus run has gone through the async path end to end.
39
- */
40
- /**
41
- * READ PER CALL, NOT AT MODULE LOAD, and the difference is not stylistic. A value fixed at import is one
42
- * no test can vary and no process can change -- `fileProductVersion` memoised Edge's version that way and
43
- * stamped five days of captures with a build they were not taken under. So it is a PARAMETER with an env
44
- * default, which also lets both paths be driven from one test file.
45
- */
46
- const syncByEnv = () => process.env.A11Y_SYNC_CAPTURE === "1";
47
- /** Accepting a capture is a handshake, not the work: it either answers in seconds or the box is unwell. */
48
- const ACCEPT_TIMEOUT_MS = 30_000;
49
- /** Between polls. Short enough that a 12 s capture is not padded, long enough not to hammer a busy box. */
50
- const POLL_MS = 2_000;
51
- /** A read of an array already in memory; if this cannot answer, the guest's event loop is blocked. */
52
- const PROGRESS_TIMEOUT_MS = 10_000;
53
- /**
54
- * Consecutive failed POLLS before giving up.
55
- *
56
- * Not one: a single dropped poll is precisely the transport fault this design exists to survive, and
57
- * treating it as a failed capture would reintroduce the defect through the new door. Not unbounded either,
58
- * or a box that has genuinely gone would be waited on for the whole capture budget.
59
- */
4
+ const RECOVERY_TIMEOUT_MS = 30000;
5
+ const syncByEnv = ()=>"1" === process.env.A11Y_SYNC_CAPTURE;
6
+ const ACCEPT_TIMEOUT_MS = 30000;
7
+ const POLL_MS = 2000;
8
+ const PROGRESS_TIMEOUT_MS = 10000;
60
9
  const MAX_POLL_FAILURES = 5;
61
- const sleep = (/** @type {number} */ ms) => new Promise((r) => setTimeout(r, ms));
62
- const base = (/** @type {string} */ worker) => String(worker).replace(/\/$/, "");
63
- /**
64
- * What is left of the OPERATION'S budget, never negative — architecture-audit.md §14.5.
65
- *
66
- * Every wait inside the poll loop used its own fixed constant regardless of how little of `timeoutMs`
67
- * remained: a two-second sleep, a ten-second progress read and a thirty-second result read, none clipped
68
- * to what was actually left. A caller asking for `timeoutMs: 20` measured 2,012 ms before an answer,
69
- * because the unconditional sleep ran to completion first regardless of the deadline it was about to blow
70
- * past. This is the one place that number is computed, so every wait below shares one clock.
71
- *
72
- * @param {number} deadline
73
- */
74
- const remaining = (deadline) => Math.max(0, deadline - Date.now());
75
- /**
76
- * Ask the worker for a capture we already paid for but may not have received.
77
- *
78
- * Returns null when there is nothing to recover — a worker predating the endpoint (404 from the router's
79
- * fallback), one that restarted and lost its memory, or a capture still running. Null means "capture
80
- * again", which is what every client did before the endpoint existed.
81
- *
82
- * A recovered FAILURE is RETURNED rather than swallowed, so a replay is indistinguishable from the original
83
- * response. That keeps the worker's `fault` code — the thing it worked out, and which we would otherwise
84
- * replace with "no answer" — and lets the caller's own classification decide, as it would have all along.
85
- * (The dataset runner's version THREW here, because it wrapped a `fetchJson` that rejects on non-2xx;
86
- * `requestJson` resolves instead, so the same intent is expressed by returning the response.)
87
- *
88
- * @param {string} worker @param {string} captureId
89
- */
90
- export async function recoverCapture(worker, captureId) {
10
+ const sleep = (ms)=>new Promise((r)=>setTimeout(r, ms));
11
+ const base = (worker)=>String(worker).replace(/\/$/, "");
12
+ const remaining = (deadline)=>Math.max(0, deadline - Date.now());
13
+ async function recoverCapture(worker, captureId) {
91
14
  let response;
92
15
  try {
93
- response = await requestJson(`${base(worker)}/capture/${captureId}`, { timeoutMs: RECOVERY_TIMEOUT_MS });
94
- }
95
- catch (error) {
96
- // The worker went away again mid-question. Not worth a second round trip; the caller falls back to
97
- // capturing, which is what it would have done anyway.
98
- void error;
16
+ response = await requestJson(`${base(worker)}/capture/${captureId}`, {
17
+ timeoutMs: RECOVERY_TIMEOUT_MS
18
+ });
19
+ } catch (error) {
99
20
  return null;
100
21
  }
101
- // 500 IS AN ANSWER, NOT AN OBSTACLE: it is the worker's own account of a failed capture, carrying the
102
- // `fault` code it worked out. Returning it lets the caller's existing classification decide, exactly as
103
- // it would have on the original response -- and losing it replaces a diagnosis with "no answer", which
104
- // this project has repeatedly misread as a dead machine.
105
- if (response.status === 500)
106
- return response;
107
- // 404 from the endpoint, or from an older worker's router fallback: nothing kept, so capture again.
108
- if (response.status === 404)
109
- return null;
110
- if (!response.ok)
111
- return null;
112
- // "Still running" and "never heard of it" are DIFFERENT ANSWERS and must stay that way. Neither is
113
- // recoverable here, but only one means the work is still being done.
114
- if ( /** @type {any} */(response.json)?.state === "running")
115
- return null;
22
+ if (500 === response.status) return response;
23
+ if (404 === response.status) return null;
24
+ if (!response.ok) return null;
25
+ if (response.json?.state === "running") return null;
116
26
  return response;
117
27
  }
118
- /**
119
- * POST a capture, and on a TRANSIENT failure ask whether it actually finished before paying again.
120
- *
121
- * Returns what `requestJson` returns — `{status, ok, text, json}` — plus `recovered`, so this is a
122
- * drop-in at the call sites that already POST `/capture` and read `response.ok` / `body.error`. Keeping
123
- * their own classification is deliberate: a shared client that also decided what counts as a failure would
124
- * be changing ten behaviours at once while claiming to change one.
125
- *
126
- * `requestJson` RESOLVES on an HTTP error and REJECTS only on a transport one, so the `catch` below is
127
- * reached exactly by the case this exists for — `ETIMEDOUT`, `ECONNRESET`, a socket dying mid-answer.
128
- *
129
- * `waitForWorker` is deliberately NOT called here. The dataset runner has one, tuned to a corpus run's
130
- * tolerance for a box gone for minutes; a gate wants an answer quickly. So this asks once, immediately,
131
- * and a caller that wants to wait first passes `beforeRecovery`.
132
- *
133
- * @param {{ worker: string, body: object, timeoutMs?: number,
134
- * beforeRecovery?: (error: unknown) => Promise<void>,
135
- * onProgress?: (progress: object) => void, sync?: boolean }} request
136
- */
137
- export async function captureTolerantly({ worker, body, timeoutMs = CAPTURE_CLIENT_TIMEOUT_MS, beforeRecovery, onProgress, sync = syncByEnv() }) {
28
+ async function captureTolerantly({ worker, body, timeoutMs = CAPTURE_CLIENT_TIMEOUT_MS, beforeRecovery, onProgress, sync = syncByEnv() }) {
138
29
  const captureId = randomUUID();
139
- if (!sync)
140
- return pollForResult({ worker, body, captureId, timeoutMs, onProgress });
141
- // Said once, at the moment it applies, rather than at import: the synchronous form holds a connection
142
- // open and silent for the whole capture, which is the shape that lost 9 of 40 responses on this fleet.
30
+ if (!sync) return pollForResult({
31
+ worker,
32
+ body,
33
+ captureId,
34
+ timeoutMs,
35
+ onProgress
36
+ });
143
37
  process.stderr.write("A11Y_SYNC_CAPTURE — holding one connection open for this capture\n");
144
38
  try {
145
- return { ...await post(worker, { ...body, captureId }, timeoutMs), recovered: false, pollsSurvived: 0 };
146
- }
147
- catch (error) {
148
- if (!isTransient(error))
149
- throw error;
150
- // The error is handed over so a caller can SAY why it is waiting. The dataset runner prints
151
- // "worker unreachable (<message>)" before a multi-minute wait, and a wait with no stated cause is
152
- // one an operator kills.
153
- if (beforeRecovery)
154
- await beforeRecovery(error);
39
+ return {
40
+ ...await post(worker, {
41
+ ...body,
42
+ captureId
43
+ }, timeoutMs),
44
+ recovered: false,
45
+ pollsSurvived: 0
46
+ };
47
+ } catch (error) {
48
+ if (!isTransient(error)) throw error;
49
+ if (beforeRecovery) await beforeRecovery(error);
155
50
  const recovered = await recoverCapture(worker, captureId);
156
- // A recovered capture is the ORIGINAL response, returned rather than re-requested. `recovered` travels
157
- // with it because a caller measuring the transport needs to know this one cost a round trip and not a
158
- // capture -- reporting it as a clean first attempt would hide the very fault this exists for.
159
- if (recovered)
160
- return { ...recovered, recovered: true, pollsSurvived: 0 };
161
- return { ...await post(worker, { ...body, captureId: randomUUID() }, timeoutMs), recovered: false, pollsSurvived: 0 };
51
+ if (recovered) return {
52
+ ...recovered,
53
+ recovered: true,
54
+ pollsSurvived: 0
55
+ };
56
+ return {
57
+ ...await post(worker, {
58
+ ...body,
59
+ captureId: randomUUID()
60
+ }, timeoutMs),
61
+ recovered: false,
62
+ pollsSurvived: 0
63
+ };
162
64
  }
163
65
  }
164
- /**
165
- * DISPATCH, THEN POLL — the async path, and the reason this module exists.
166
- *
167
- * `POST {async:true}` returns 202 in milliseconds, so no connection is held while NVDA reads a page. The
168
- * result is collected from the store with `GET /capture/<id>`, which is the endpoint that has existed for
169
- * this shape all along and was only ever reached after a failure. **The recovery path is now the normal
170
- * path**, which is what stops it rotting: a route that runs only when something breaks is one nobody
171
- * notices has broken.
172
- *
173
- * A dropped poll costs one round trip and is simply retried; the capture is unaffected because nothing is
174
- * riding on that socket. That is the whole difference from the synchronous form, where the answer existed
175
- * only in the connection that was carrying it.
176
- */
177
- /**
178
- * @param {{ worker: string, body: object, captureId: string, timeoutMs: number,
179
- * onProgress?: (progress: object) => void }} request
180
- */
181
66
  async function pollForResult({ worker, body, captureId, timeoutMs, onProgress }) {
182
67
  const deadline = Date.now() + timeoutMs;
183
68
  let accepted;
184
69
  try {
185
- accepted = await post(worker, { ...body, captureId, async: true }, ACCEPT_TIMEOUT_MS);
186
- }
187
- catch (error) {
188
- if (!isTransient(error))
189
- throw error;
70
+ accepted = await post(worker, {
71
+ ...body,
72
+ captureId,
73
+ async: true
74
+ }, ACCEPT_TIMEOUT_MS);
75
+ } catch (error) {
76
+ if (!isTransient(error)) throw error;
190
77
  const reconciled = await reconcileLostAcceptance(worker, captureId, deadline);
191
- if (reconciled.state === "done")
192
- return { ...reconciled.response, recovered: true, pollsSurvived: 0 };
193
- if (reconciled.state === "unknown") {
194
- // CONFIRMED nothing is running under this id -- only now is a fresh one safe, exactly as the
195
- // synchronous escape hatch already does for the identical failure. Minting one on the first sign
196
- // of trouble, before asking, is what the audit's remedy forbids: it would risk a second real
197
- // capture running under a worker that already accepted the first.
198
- return pollForResult({
199
- worker, body, captureId: randomUUID(), timeoutMs: Math.max(0, deadline - Date.now()), onProgress,
200
- });
201
- }
202
- // reconciled.state === "running": the worker DID accept it under the id we already hold -- the 202
203
- // was lost, not the acceptance. Fall through to the ordinary poll loop below exactly as a received
204
- // 202 would have.
78
+ if ("done" === reconciled.state) return {
79
+ ...reconciled.response,
80
+ recovered: true,
81
+ pollsSurvived: 0
82
+ };
83
+ if ("unknown" === reconciled.state) return pollForResult({
84
+ worker,
85
+ body,
86
+ captureId: randomUUID(),
87
+ timeoutMs: Math.max(0, deadline - Date.now()),
88
+ onProgress
89
+ });
205
90
  }
206
- // A worker too old to know `async` runs the capture SYNCHRONOUSLY and answers 200 with the result. That
207
- // is not an error and must not be retried -- it is the additive-field contract this project uses for
208
- // every wire change, and it means a host can be deployed before the fleet.
209
- if (accepted && accepted.status !== 202)
210
- return { ...accepted, recovered: false, pollsSurvived: 0 };
211
- return awaitCompletion({ worker, captureId, deadline, timeoutMs, onProgress });
91
+ if (accepted && 202 !== accepted.status) return {
92
+ ...accepted,
93
+ recovered: false,
94
+ pollsSurvived: 0
95
+ };
96
+ return awaitCompletion({
97
+ worker,
98
+ captureId,
99
+ deadline,
100
+ timeoutMs,
101
+ onProgress
102
+ });
212
103
  }
213
- /**
214
- * ASKING ABOUT A CAPTURE WE MAY OR MAY NOT HAVE STARTED, using the SAME three-way discrimination a
215
- * dropped poll already relies on (`pollOnce`) rather than `recoverCapture`, which collapses "running"
216
- * and "unknown" into one null and cannot tell them apart -- the exact distinction this exists to make.
217
- *
218
- * A 404 here is not quite `pollOnce`'s documented "worker restarted mid-capture": no 202 was ever
219
- * received, so there is nothing to have restarted AWAY FROM. It means the POST itself never reached the
220
- * worker, which is what makes minting a fresh id safe once this returns "unknown" and not before.
221
- *
222
- * The audit's own caveat applies in principle: "unknown does not prove execution never occurred" if the
223
- * result were evicted (§14.4) before this ever asks — but that needs seven OTHER captures to finish on
224
- * this worker in the seconds between the lost 202 and this reconciliation, which the worker's own `busy`
225
- * gate (one capture at a time) makes impossible while nothing else is running under this id.
226
- *
227
- * @param {string} worker @param {string} captureId @param {number} deadline
228
- * @returns {Promise<{ state: "running" } | { state: "unknown" } | { state: "done", response: any }>}
229
- */
230
104
  async function reconcileLostAcceptance(worker, captureId, deadline) {
231
105
  let lastError;
232
- while (Date.now() < deadline) {
106
+ while(Date.now() < deadline){
233
107
  const poll = await pollOnce(worker, captureId, Math.min(RECOVERY_TIMEOUT_MS, remaining(deadline)));
234
- if (poll.running)
235
- return { state: "running" };
236
- if (poll.done)
237
- return { state: "done", response: poll.response };
238
- if (poll.lost)
239
- return { state: "unknown" };
108
+ if (poll.running) return {
109
+ state: "running"
110
+ };
111
+ if (poll.done) return {
112
+ state: "done",
113
+ response: poll.response
114
+ };
115
+ if (poll.lost) return {
116
+ state: "unknown"
117
+ };
240
118
  lastError = poll.error;
241
119
  await sleep(Math.min(POLL_MS, remaining(deadline)));
242
120
  }
243
- // Never resolved within the budget: neither confirmed running nor confirmed absent. The ORIGINAL
244
- // acceptance failure is the fault that actually occurred, so it is what surfaces -- not a generic
245
- // timeout that would hide which of the two things went wrong.
246
- throw lastError ?? Object.assign(new Error(`could not confirm capture ${captureId} was accepted, within its remaining budget`), { code: "ETIMEDOUT" });
121
+ throw lastError ?? Object.assign(new Error(`could not confirm capture ${captureId} was accepted, within its remaining budget`), {
122
+ code: "ETIMEDOUT"
123
+ });
247
124
  }
248
- /**
249
- * DISPATCH, THEN POLL — the async path, and the reason this module exists.
250
- *
251
- * `POST {async:true}` returns 202 in milliseconds, so no connection is held while NVDA reads a page. The
252
- * result is collected from the store with `GET /capture/<id>`, which is the endpoint that has existed for
253
- * this shape all along and was only ever reached after a failure. **The recovery path is now the normal
254
- * path**, which is what stops it rotting: a route that runs only when something breaks is one nobody
255
- * notices has broken.
256
- *
257
- * A dropped poll costs one round trip and is simply retried; the capture is unaffected because nothing is
258
- * riding on that socket. That is the whole difference from the synchronous form, where the answer existed
259
- * only in the connection that was carrying it.
260
- *
261
- * @param {{ worker: string, captureId: string, deadline: number, timeoutMs: number,
262
- * onProgress?: (progress: object) => void }} request
263
- */
264
125
  async function awaitCompletion({ worker, captureId, deadline, timeoutMs, onProgress }) {
265
126
  let transportFailures = 0;
266
127
  let survived = 0;
267
- while (Date.now() < deadline) {
268
- // CLIPPED, EACH TIME, TO WHAT IS LEFT — architecture-audit.md §14.5. A budget of 20 ms must not spend
269
- // a full 2 s sleeping before it is even allowed to check the clock again.
128
+ while(Date.now() < deadline){
270
129
  await sleep(Math.min(POLL_MS, remaining(deadline)));
271
- if (onProgress)
272
- await readProgress(worker, onProgress, Math.min(PROGRESS_TIMEOUT_MS, remaining(deadline)));
130
+ if (onProgress) await readProgress(worker, onProgress, Math.min(PROGRESS_TIMEOUT_MS, remaining(deadline)));
273
131
  const poll = await pollOnce(worker, captureId, Math.min(RECOVERY_TIMEOUT_MS, remaining(deadline)));
274
- if (poll.done) {
275
- // WHAT THE TRANSPORT DID, carried out with the result. Under the synchronous protocol a dropped
276
- // response destroyed the capture; here it costs one poll -- but "harmless" and "not happening" are
277
- // different facts, and only one of them means the network is healthy. Reporting it keeps the
278
- // question answerable after the fix that stopped it mattering.
279
- return { ...poll.response, recovered: false, pollsSurvived: survived };
280
- }
281
- if (poll.lost) {
282
- throw Object.assign(new Error(`worker forgot capture ${captureId} after accepting it — it restarted `
283
- + "mid-capture, so the work is gone and the case must be re-issued"), { code: "CAPTURE_LOST" });
284
- }
285
- // A FAILED POLL IS NOT A FAILED CAPTURE, and conflating them would give back the defect this design
286
- // removes: the worker is still working, we merely could not ask. Retried until a RUN of them says the
287
- // box has gone, rather than on the first one -- a single dropped request is the exact fault this exists
288
- // to survive.
289
- if (poll.unreachable)
290
- survived += 1;
132
+ if (poll.done) return {
133
+ ...poll.response,
134
+ recovered: false,
135
+ pollsSurvived: survived
136
+ };
137
+ if (poll.lost) throw Object.assign(new Error(`worker forgot capture ${captureId} after accepting it — it restarted mid-capture, so the work is gone and the case must be re-issued`), {
138
+ code: "CAPTURE_LOST"
139
+ });
140
+ if (poll.unreachable) survived += 1;
291
141
  transportFailures = poll.unreachable ? transportFailures + 1 : 0;
292
- if (transportFailures >= MAX_POLL_FAILURES)
293
- throw poll.error;
142
+ if (transportFailures >= MAX_POLL_FAILURES) throw poll.error;
294
143
  }
295
- throw Object.assign(new Error(`capture ${captureId} did not finish within ${timeoutMs} ms`), { code: "ETIMEDOUT" });
144
+ throw Object.assign(new Error(`capture ${captureId} did not finish within ${timeoutMs} ms`), {
145
+ code: "ETIMEDOUT"
146
+ });
296
147
  }
297
- /**
298
- * ONE POLL, WITH FOUR DISTINCT ANSWERS — and keeping them apart is the whole job.
299
- *
300
- * done the capture finished (200 or 500); the 500 carries the worker's own fault code
301
- * lost 404 AFTER a 202 acceptance: the worker restarted, the work is gone, re-issue
302
- * running 202; keep waiting
303
- * unreachable we could not ask; the capture is unaffected
304
- *
305
- * NOT `recoverCapture`, and that is the correction. It SWALLOWS a transport error and returns null, which
306
- * is right for its own job — "we already failed, is the result there?" — and wrong here, because it makes
307
- * "could not ask" indistinguishable from "still running". Written that way first, and the retry counter
308
- * built on it was unreachable: found by mutation, since deleting the retry changed nothing.
309
- *
310
- * @param {string} worker @param {string} captureId @param {number} [timeoutMs] clipped to the operation's
311
- * remaining budget by the caller — see `remaining()` — so this read cannot outlive the deadline it is
312
- * answering to on its own.
313
- */
314
148
  async function pollOnce(worker, captureId, timeoutMs = RECOVERY_TIMEOUT_MS) {
315
149
  let response;
316
150
  try {
317
- response = await requestJson(`${base(worker)}/capture/${captureId}`, { timeoutMs });
318
- }
319
- catch (error) {
320
- return { unreachable: true, error };
151
+ response = await requestJson(`${base(worker)}/capture/${captureId}`, {
152
+ timeoutMs
153
+ });
154
+ } catch (error) {
155
+ return {
156
+ unreachable: true,
157
+ error
158
+ };
321
159
  }
322
- if (response.status === 404)
323
- return { lost: true };
324
- if (response.status === 202)
325
- return { running: true };
326
- // 200 and 500 are both ANSWERS: the second is the worker's diagnosis, and losing it would replace a
327
- // fault code with silence -- which this project has repeatedly misread as a dead machine.
328
- if (response.ok || response.status === 500)
329
- return { done: true, response };
330
- // Anything else is a worker speaking a protocol we do not know; treat it as unreachable rather than as
331
- // an answer, so it is retried and then surfaces with its own status rather than being read as a capture.
332
- return { unreachable: true, error: new Error(`unexpected ${response.status} polling ${captureId}`) };
160
+ if (404 === response.status) return {
161
+ lost: true
162
+ };
163
+ if (202 === response.status) return {
164
+ running: true
165
+ };
166
+ if (response.ok || 500 === response.status) return {
167
+ done: true,
168
+ response
169
+ };
170
+ return {
171
+ unreachable: true,
172
+ error: new Error(`unexpected ${response.status} polling ${captureId}`)
173
+ };
333
174
  }
334
- /** The phase the worker is IN, so a caller can tell a slow capture from a wedged one. */
335
- async function readProgress(/** @type {string} */ worker, /** @type {(p: object) => void} */ onProgress,
336
- /** @type {number} */ timeoutMs = PROGRESS_TIMEOUT_MS) {
175
+ async function readProgress(worker, onProgress, timeoutMs = PROGRESS_TIMEOUT_MS) {
337
176
  try {
338
- const { json } = await requestJson(`${base(worker)}/progress`, { timeoutMs });
339
- if (json && typeof json === "object")
340
- onProgress(json);
341
- }
342
- catch (error) {
343
- // Progress is a convenience; failing to read it must never fail a capture that is going fine.
344
- void error;
345
- }
177
+ const { json } = await requestJson(`${base(worker)}/progress`, {
178
+ timeoutMs
179
+ });
180
+ if (json && "object" == typeof json) onProgress(json);
181
+ } catch (error) {}
346
182
  }
347
- /** @param {string} worker @param {object} body @param {number} timeoutMs */
348
183
  function post(worker, body, timeoutMs) {
349
- // `requestJson` serialises the body and sets the headers; passing a string here would double-encode it.
350
- return requestJson(`${base(worker)}/capture`, { method: "POST", body, timeoutMs });
184
+ return requestJson(`${base(worker)}/capture`, {
185
+ method: "POST",
186
+ body,
187
+ timeoutMs
188
+ });
351
189
  }
352
- //# sourceMappingURL=capture-client.mjs.map
190
+ export { captureTolerantly, recoverCapture };
@@ -31,4 +31,3 @@ import { configuredWorkers } from "./fleet-env.mjs";
31
31
  declare function localPoolUrls(): any;
32
32
  import { inventoryWorkerUrls } from "./fleet-env.mjs";
33
33
  export {};
34
- //# sourceMappingURL=check-worker-code.d.mts.map