@a11ign/screenreader-fleet 0.1.0 → 0.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/capture-client.d.mts +0 -1
- package/dist/capture-client.mjs +144 -306
- package/dist/check-worker-code.d.mts +0 -1
- package/dist/check-worker-code.mjs +42 -141
- package/dist/cli-flags.d.mts +0 -1
- package/dist/cli-flags.mjs +33 -179
- package/dist/code-drift.d.mts +0 -1
- package/dist/command-line-census.d.mts +0 -1
- package/dist/compare-workers.d.mts +0 -1
- package/dist/compare-workers.mjs +383 -255
- package/dist/control-plane-isolation.d.mts +0 -1
- package/dist/deploy-worker.d.mts +0 -1
- package/dist/deploy-worker.mjs +101 -242
- package/dist/doctor.d.mts +0 -1
- package/dist/doctor.mjs +361 -809
- package/dist/fleet-consistency.d.mts +0 -1
- package/dist/fleet-consistency.mjs +155 -379
- package/dist/fleet-env.d.mts +0 -1
- package/dist/fleet-env.mjs +148 -438
- package/dist/fleet-scripts.d.mts +0 -1
- package/dist/git-safe-env.d.mts +0 -1
- package/dist/guest-run.d.mts +0 -1
- package/dist/host-address.d.mts +0 -1
- package/dist/host-address.mjs +19 -90
- package/dist/host-capacity.d.mts +0 -1
- package/dist/host-capacity.mjs +22 -136
- package/dist/host-metrics.d.mts +0 -1
- package/dist/index.d.ts +0 -1
- package/dist/index.mjs +231 -0
- package/dist/local-vm.d.ts +0 -1
- package/dist/measure-guard.d.mts +0 -1
- package/dist/normalise-fleet.d.mts +0 -1
- package/dist/npm-cli-executable.d.mts +0 -1
- package/dist/probe-outcome.d.mts +0 -1
- package/dist/probe-outcome.mjs +50 -96
- package/dist/protocol-guard.d.mts +0 -1
- package/dist/source-walk.d.mts +0 -1
- package/dist/src_fleet-scripts_mjs.mjs +16 -0
- package/dist/src_git-safe-env_mjs.mjs +9 -0
- package/dist/src_utm-deprecated_mjs.mjs +4 -0
- package/dist/transient-fault.d.mts +0 -1
- package/dist/transient-fault.mjs +21 -81
- package/dist/utm-deprecated.d.mts +0 -1
- package/dist/worker-code-check.d.mts +0 -1
- package/dist/worker-code-check.mjs +127 -76
- package/dist/worker-health.d.mts +0 -1
- package/dist/worker-health.mjs +16 -61
- package/dist/worker-http.d.mts +0 -1
- package/dist/worker-http.mjs +42 -234
- package/dist/worker-stats.d.mts +0 -1
- package/package.json +12 -5
- package/dist/capture-client.d.mts.map +0 -1
- package/dist/capture-client.mjs.map +0 -1
- package/dist/check-worker-code.d.mts.map +0 -1
- package/dist/check-worker-code.mjs.map +0 -1
- package/dist/cli-flags.d.mts.map +0 -1
- package/dist/cli-flags.mjs.map +0 -1
- package/dist/code-drift.d.mts.map +0 -1
- package/dist/code-drift.mjs +0 -284
- package/dist/code-drift.mjs.map +0 -1
- package/dist/command-line-census.d.mts.map +0 -1
- package/dist/command-line-census.mjs +0 -96
- package/dist/command-line-census.mjs.map +0 -1
- package/dist/compare-workers.d.mts.map +0 -1
- package/dist/compare-workers.mjs.map +0 -1
- package/dist/control-plane-isolation.d.mts.map +0 -1
- package/dist/control-plane-isolation.mjs +0 -67
- package/dist/control-plane-isolation.mjs.map +0 -1
- package/dist/deploy-worker.d.mts.map +0 -1
- package/dist/deploy-worker.mjs.map +0 -1
- package/dist/doctor.d.mts.map +0 -1
- package/dist/doctor.mjs.map +0 -1
- package/dist/fleet-consistency.d.mts.map +0 -1
- package/dist/fleet-consistency.mjs.map +0 -1
- package/dist/fleet-env.d.mts.map +0 -1
- package/dist/fleet-env.mjs.map +0 -1
- package/dist/fleet-scripts.d.mts.map +0 -1
- package/dist/fleet-scripts.mjs +0 -41
- package/dist/fleet-scripts.mjs.map +0 -1
- package/dist/git-safe-env.d.mts.map +0 -1
- package/dist/git-safe-env.mjs +0 -44
- package/dist/git-safe-env.mjs.map +0 -1
- package/dist/guest-run.d.mts.map +0 -1
- package/dist/guest-run.mjs +0 -164
- package/dist/guest-run.mjs.map +0 -1
- package/dist/host-address.d.mts.map +0 -1
- package/dist/host-address.mjs.map +0 -1
- package/dist/host-capacity.d.mts.map +0 -1
- package/dist/host-capacity.mjs.map +0 -1
- package/dist/host-metrics.d.mts.map +0 -1
- package/dist/host-metrics.mjs +0 -201
- package/dist/host-metrics.mjs.map +0 -1
- package/dist/index.d.ts.map +0 -1
- package/dist/index.js +0 -25
- package/dist/index.js.map +0 -1
- package/dist/local-vm.d.ts.map +0 -1
- package/dist/local-vm.js +0 -360
- package/dist/local-vm.js.map +0 -1
- package/dist/measure-guard.d.mts.map +0 -1
- package/dist/measure-guard.mjs +0 -73
- package/dist/measure-guard.mjs.map +0 -1
- package/dist/normalise-fleet.d.mts.map +0 -1
- package/dist/normalise-fleet.mjs +0 -76
- package/dist/normalise-fleet.mjs.map +0 -1
- package/dist/npm-cli-executable.d.mts.map +0 -1
- package/dist/npm-cli-executable.mjs +0 -159
- package/dist/npm-cli-executable.mjs.map +0 -1
- package/dist/probe-outcome.d.mts.map +0 -1
- package/dist/probe-outcome.mjs.map +0 -1
- package/dist/protocol-guard.d.mts.map +0 -1
- package/dist/protocol-guard.mjs +0 -121
- package/dist/protocol-guard.mjs.map +0 -1
- package/dist/source-walk.d.mts.map +0 -1
- package/dist/source-walk.mjs +0 -56
- package/dist/source-walk.mjs.map +0 -1
- package/dist/transient-fault.d.mts.map +0 -1
- package/dist/transient-fault.mjs.map +0 -1
- package/dist/utm-deprecated.d.mts.map +0 -1
- package/dist/utm-deprecated.mjs +0 -23
- package/dist/utm-deprecated.mjs.map +0 -1
- package/dist/worker-code-check.d.mts.map +0 -1
- package/dist/worker-code-check.mjs.map +0 -1
- package/dist/worker-health.d.mts.map +0 -1
- package/dist/worker-health.mjs.map +0 -1
- package/dist/worker-http.d.mts.map +0 -1
- package/dist/worker-http.mjs.map +0 -1
- package/dist/worker-stats.d.mts.map +0 -1
- package/dist/worker-stats.mjs +0 -143
- package/dist/worker-stats.mjs.map +0 -1
package/dist/deploy-worker.d.mts
CHANGED
package/dist/deploy-worker.mjs
CHANGED
|
@@ -1,295 +1,165 @@
|
|
|
1
1
|
#!/usr/bin/env node
|
|
2
|
-
// @ts-check
|
|
3
|
-
// Deploy the worker's code to the guests, in one command, and prove it landed.
|
|
4
|
-
//
|
|
5
|
-
// node scripts/deploy-worker.mjs # every local worker VM, one at a time
|
|
6
|
-
// node scripts/deploy-worker.mjs --vm=a11y-worker-2
|
|
7
|
-
//
|
|
8
|
-
// Why this exists: deploying was a documented twelve-step manual dance — push four files with
|
|
9
|
-
// `utmctl file push`, stop the VM, start it, then run `worker:code` and hope. Two things about that
|
|
10
|
-
// were unacceptable for something we rely on:
|
|
11
|
-
//
|
|
12
|
-
// - **It is easy to get wrong, and I got it wrong.** Pushing three of the four files leaves a guest
|
|
13
|
-
// running a mix; the symptom is a `worker:code` mismatch with no clue which file is stale.
|
|
14
|
-
// - **`utmctl exec` cannot be trusted to restart the worker**, so the reboot is mandatory and easy to
|
|
15
|
-
// skip. Skipping it makes the guest serve the previous code while reporting success — which cost two
|
|
16
|
-
// workers an hour of running stale code once already.
|
|
17
|
-
//
|
|
18
|
-
// So: one command, every hashed file, a real reboot, and a hash check over HTTP afterwards. The hash
|
|
19
|
-
// check is the whole point — it shares no failure mode with the push, which is why `/health.code`
|
|
20
|
-
// exists rather than reading the guest's files back through the same broken channel.
|
|
21
|
-
//
|
|
22
|
-
// Deploys the WORKING TREE, deliberately: that is what you are testing. Roll back by checking out the
|
|
23
|
-
// ref you want and running this again — git is the source of truth for "the previous version", so there
|
|
24
|
-
// is no bespoke backup to go stale.
|
|
25
2
|
import { pathToFileURL } from "node:url";
|
|
26
3
|
import { execFile, execFileSync } from "node:child_process";
|
|
27
|
-
import { sandboxGitEnv } from "./git-safe-env.mjs";
|
|
28
4
|
import { createReadStream, realpathSync } from "node:fs";
|
|
29
5
|
import { promisify } from "node:util";
|
|
30
|
-
import { resolve } from "node:path";
|
|
31
|
-
// By SUBPATH, never the package ROOT: the index re-exports `capture-core.mjs`, which imports guidepup and
|
|
32
|
-
// throws `No available supported screen readers` at import on any host without one. This file only
|
|
33
|
-
// runs on a Mac, where VoiceOver makes that throw invisible — which is exactly why it went unnoticed.
|
|
34
|
-
// `no-win32-imports.test.ts` found it.
|
|
6
|
+
import { resolve as external_node_path_resolve } from "node:path";
|
|
35
7
|
import { WORKER_FILES } from "@a11ign/screenreader-worker/worker-files";
|
|
36
|
-
import {
|
|
37
|
-
|
|
38
|
-
|
|
39
|
-
|
|
40
|
-
// has to scrape TEXT, because `git show` returns a historical file's bytes, not something importable.
|
|
41
|
-
import { CAPTURE_PROTOCOL_VERSION as PROTOCOL_IN_TREE } from "@a11ign/screenreader-worker/protocol-version";
|
|
42
|
-
import { fleetScriptPaths } from "./fleet-scripts.mjs";
|
|
8
|
+
import { codeVersion, workerSourceDir } from "@a11ign/screenreader-worker/code-version";
|
|
9
|
+
import { CAPTURE_PROTOCOL_VERSION } from "@a11ign/screenreader-worker/protocol-version";
|
|
10
|
+
import { requestJson as worker_http_requestJson } from "./worker-http.mjs";
|
|
11
|
+
import { fleetScriptPaths } from "./src_fleet-scripts_mjs.mjs";
|
|
43
12
|
import { refuseUnknownFlags, flagValue } from "./cli-flags.mjs";
|
|
44
|
-
import {
|
|
45
|
-
import {
|
|
46
|
-
|
|
47
|
-
|
|
48
|
-
*
|
|
49
|
-
|
|
50
|
-
|
|
51
|
-
|
|
52
|
-
|
|
53
|
-
|
|
54
|
-
|
|
13
|
+
import { sandboxGitEnv } from "./src_git-safe-env_mjs.mjs";
|
|
14
|
+
import { warnUtmDeprecated } from "./src_utm-deprecated_mjs.mjs";
|
|
15
|
+
const MANIFEST_CASES = 1715;
|
|
16
|
+
const CAPTURES_PER_CASE = 2;
|
|
17
|
+
const RECAPTURE_COST = `${(MANIFEST_CASES * CAPTURES_PER_CASE).toLocaleString("en-US")} captures (${MANIFEST_CASES.toLocaleString("en-US")} cases in manifest.json x ${CAPTURES_PER_CASE}, read 2026-09-23T14:26Z; how long that takes depends on the fleet, so time a run rather than trust a figure)`;
|
|
18
|
+
refuseUnknownFlags([
|
|
19
|
+
"--vm=",
|
|
20
|
+
"--allow-protocol-change"
|
|
21
|
+
], {
|
|
22
|
+
entry: import.meta.url,
|
|
23
|
+
command: "npm run worker:deploy"
|
|
24
|
+
});
|
|
55
25
|
const run = promisify(execFile);
|
|
56
|
-
// From the worker PACKAGE, not from the cwd. This was `resolve("src/capture/nvda")` and then
|
|
57
|
-
// `resolve("packages/nvda-worker/src")` — a repo-layout guess that had to be edited every time the worker
|
|
58
|
-
// moved, and that silently pointed at nothing whenever the cwd was not the repo root.
|
|
59
26
|
const NVDA_DIR = workerSourceDir();
|
|
60
|
-
// The guest's layout deliberately does NOT mirror the repo's. It is where provisioning put the files and
|
|
61
|
-
// where the scheduled task points, so renaming it means re-provisioning every guest — and M5 moving the host
|
|
62
|
-
// directory to `packages/nvda-worker/src` changed nothing here. All the worker needs is that its files land in
|
|
63
|
-
// one directory together.
|
|
64
27
|
const GUEST_DIR = "C:\\Users\\witness\\a11y-witness\\src\\capture\\nvda";
|
|
65
|
-
// Resolved from THIS module: the fleet scripts ship with this package, so a cwd-relative path was only ever
|
|
66
|
-
// right when run from the repo root.
|
|
67
28
|
const CTL = fleetScriptPaths().workerCtl;
|
|
68
|
-
const LIFECYCLE_TIMEOUT_MS =
|
|
69
|
-
|
|
70
|
-
|
|
71
|
-
const POOL_TIMEOUT_MS = 240_000;
|
|
72
|
-
const HEALTH_TIMEOUT_MS = 20_000;
|
|
29
|
+
const LIFECYCLE_TIMEOUT_MS = 420000;
|
|
30
|
+
const POOL_TIMEOUT_MS = 240000;
|
|
31
|
+
const deploy_worker_HEALTH_TIMEOUT_MS = 20000;
|
|
73
32
|
const only = flagValue(process.argv, "vm");
|
|
74
|
-
/**
|
|
75
|
-
* The files that make up the worker's code version.
|
|
76
|
-
*
|
|
77
|
-
* Imported from the one module that defines them. This used to parse `check-worker-code.mjs`'s SOURCE with a
|
|
78
|
-
* regex for the same list — better than a third copy, which is what the comment here used to argue, but it
|
|
79
|
-
* still broke silently if that loop were ever rewritten, and "a file missing from the list deploys invisibly"
|
|
80
|
-
* is the failure it was guarding against.
|
|
81
|
-
*/
|
|
82
33
|
function hashedFiles() {
|
|
83
34
|
return WORKER_FILES;
|
|
84
35
|
}
|
|
85
|
-
/**
|
|
86
|
-
* The SHARED hasher, not a local copy of it.
|
|
87
|
-
*
|
|
88
|
-
* This used to hash raw bytes while `codeVersion()` normalises CRLF to LF -- and that difference is not
|
|
89
|
-
* cosmetic: a worker whose repo was git-cloned on Windows checks out CRLF, so the two sides hashed
|
|
90
|
-
* different bytes for identical code and the deploy verification reported STALE for ever. Measured on the
|
|
91
|
-
* first bare-metal worker: 31979b551b7a2cfa against a checkout's 22822b7a3a08969c.
|
|
92
|
-
*
|
|
93
|
-
* `code-version.test.ts` claims to enforce "one hasher" but only greps for the file list, so this file
|
|
94
|
-
* satisfied it while keeping its own implementation. Two implementations of a comparison are two chances
|
|
95
|
-
* to disagree, and the whole point of this check is that both sides agree.
|
|
96
|
-
*/
|
|
97
36
|
function localVersion() {
|
|
98
37
|
return codeVersion(NVDA_DIR);
|
|
99
38
|
}
|
|
100
39
|
async function pool() {
|
|
101
|
-
const { stdout } = await run(CTL, [
|
|
40
|
+
const { stdout } = await run(CTL, [
|
|
41
|
+
"pool"
|
|
42
|
+
], {
|
|
43
|
+
timeout: POOL_TIMEOUT_MS,
|
|
44
|
+
encoding: "utf8"
|
|
45
|
+
});
|
|
102
46
|
const all = JSON.parse(stdout);
|
|
103
|
-
return only ? all.filter((
|
|
47
|
+
return only ? all.filter((vm)=>vm.name === only) : all;
|
|
104
48
|
}
|
|
105
|
-
/** @param {string} action @param {string} vmName */
|
|
106
49
|
function ctl(action, vmName) {
|
|
107
|
-
return run(CTL, [
|
|
108
|
-
|
|
109
|
-
|
|
50
|
+
return run(CTL, [
|
|
51
|
+
action
|
|
52
|
+
], {
|
|
53
|
+
timeout: LIFECYCLE_TIMEOUT_MS,
|
|
54
|
+
encoding: "utf8",
|
|
55
|
+
env: {
|
|
56
|
+
...process.env,
|
|
57
|
+
A11Y_VM_NAME: vmName
|
|
58
|
+
}
|
|
110
59
|
});
|
|
111
60
|
}
|
|
112
|
-
/**
|
|
113
|
-
* Push one file. `utmctl file push` reads the content from stdin, which execFile cannot stream, so the
|
|
114
|
-
* child is spawned and the file piped in.
|
|
115
|
-
*/
|
|
116
|
-
/** @param {string} uuid @param {string} file @returns {Promise<void>} */
|
|
117
61
|
function push(uuid, file) {
|
|
118
|
-
return new Promise((done, fail)
|
|
119
|
-
const child = execFile("utmctl", [
|
|
120
|
-
|
|
121
|
-
|
|
62
|
+
return new Promise((done, fail)=>{
|
|
63
|
+
const child = execFile("utmctl", [
|
|
64
|
+
"file",
|
|
65
|
+
"push",
|
|
66
|
+
uuid,
|
|
67
|
+
`${GUEST_DIR}\\${file}`
|
|
68
|
+
], (error)=>error ? fail(new Error(`push ${file}: ${error.message}`)) : done());
|
|
69
|
+
if (child.stdin) createReadStream(external_node_path_resolve(NVDA_DIR, file)).pipe(child.stdin);
|
|
122
70
|
});
|
|
123
71
|
}
|
|
124
|
-
/** @param {string} ip @param {number} port */
|
|
125
72
|
async function healthCode(ip, port) {
|
|
126
|
-
|
|
127
|
-
|
|
128
|
-
|
|
129
|
-
|
|
130
|
-
|
|
131
|
-
const response = await requestJson(`http://${ip}:${port}/health`, { timeoutMs: HEALTH_TIMEOUT_MS });
|
|
132
|
-
if (!response.ok)
|
|
133
|
-
throw new Error(`HTTP ${response.status} from /health`);
|
|
134
|
-
// `requestJson` returns `undefined` for unparseable JSON rather than throwing (its own docstring: a
|
|
135
|
-
// cache miss is a normal outcome for its usual callers) -- `fetch`'s `.json()` threw, and
|
|
136
|
-
// `healthCodeWhenAwake`'s retry loop relies on that to keep polling on garbage the same as on silence.
|
|
137
|
-
if (response.json === undefined)
|
|
138
|
-
throw new Error(`invalid JSON from http://${ip}:${port}/health`);
|
|
73
|
+
const response = await worker_http_requestJson(`http://${ip}:${port}/health`, {
|
|
74
|
+
timeoutMs: deploy_worker_HEALTH_TIMEOUT_MS
|
|
75
|
+
});
|
|
76
|
+
if (!response.ok) throw new Error(`HTTP ${response.status} from /health`);
|
|
77
|
+
if (void 0 === response.json) throw new Error(`invalid JSON from http://${ip}:${port}/health`);
|
|
139
78
|
return response.json.code;
|
|
140
79
|
}
|
|
141
|
-
|
|
142
|
-
const
|
|
143
|
-
const VERIFY_POLL_MS = 10_000;
|
|
144
|
-
/**
|
|
145
|
-
* Read the guest's code hash, waiting for it to finish booting first.
|
|
146
|
-
*
|
|
147
|
-
* A guest that is not answering YET is not a failed deploy, and reading `/health` once immediately after the
|
|
148
|
-
* reboot conflated the two: `worker:deploy` printed "stale or failed" while `npm run worker:code` — run a
|
|
149
|
-
* minute later against the same guest — reported `matches`. That false alarm sent me redeploying guests that
|
|
150
|
-
* had deployed correctly, repeatedly, and the deploy is the tool whose whole job is telling you whether the
|
|
151
|
-
* push landed.
|
|
152
|
-
*
|
|
153
|
-
* Only SILENCE is waited on. A hash that answers and differs is returned straight to the caller, which
|
|
154
|
-
* compares it — so a genuine stale deploy still fails immediately and only a booting guest costs time. Boot
|
|
155
|
-
* times measured on this fleet run from 30 s to 147 s depending on how much Edge-profile hygiene the guest has
|
|
156
|
-
* to do first, so the budget is well above the slowest honest answer.
|
|
157
|
-
*/
|
|
158
|
-
/** @param {string} ip @param {number} port */
|
|
80
|
+
const VERIFY_BUDGET_MS = 240000;
|
|
81
|
+
const VERIFY_POLL_MS = 10000;
|
|
159
82
|
async function healthCodeWhenAwake(ip, port) {
|
|
160
83
|
const deadline = Date.now() + VERIFY_BUDGET_MS;
|
|
161
84
|
let last = "no answer";
|
|
162
85
|
let waited = false;
|
|
163
|
-
while
|
|
164
|
-
|
|
165
|
-
|
|
166
|
-
|
|
167
|
-
|
|
168
|
-
|
|
169
|
-
|
|
170
|
-
|
|
171
|
-
|
|
172
|
-
|
|
173
|
-
process.stdout.write(" waiting for the guest to answer /health ");
|
|
174
|
-
waited = true;
|
|
175
|
-
process.stdout.write(".");
|
|
176
|
-
await new Promise((resolve) => setTimeout(resolve, VERIFY_POLL_MS));
|
|
177
|
-
}
|
|
86
|
+
while(Date.now() < deadline)try {
|
|
87
|
+
const actual = await healthCode(ip, port);
|
|
88
|
+
if (waited) process.stdout.write("\n");
|
|
89
|
+
return actual;
|
|
90
|
+
} catch (error) {
|
|
91
|
+
last = error instanceof Error ? error.message : String(error);
|
|
92
|
+
if (!waited) process.stdout.write(" waiting for the guest to answer /health ");
|
|
93
|
+
waited = true;
|
|
94
|
+
process.stdout.write(".");
|
|
95
|
+
await new Promise((resolve)=>setTimeout(resolve, VERIFY_POLL_MS));
|
|
178
96
|
}
|
|
179
|
-
if (waited)
|
|
180
|
-
process.stdout.write("\n");
|
|
97
|
+
if (waited) process.stdout.write("\n");
|
|
181
98
|
throw new Error(`${ip}:${port} never answered /health within ${VERIFY_BUDGET_MS / 1000}s (last: ${last})`);
|
|
182
99
|
}
|
|
183
|
-
|
|
184
|
-
* Wait until a VM is no longer `stopping`, so the next deploy does not start a guest on top of one still
|
|
185
|
-
* holding its memory.
|
|
186
|
-
*
|
|
187
|
-
* Bounded and non-fatal: if it never settles we continue and let the next deploy report its own failure,
|
|
188
|
-
* because a deploy that hangs forever is worse than one that reports a stale worker.
|
|
189
|
-
*/
|
|
190
|
-
/** @param {string} name @param {number} [limitMs] */
|
|
191
|
-
async function waitUntilSettled(name, limitMs = 120_000) {
|
|
100
|
+
async function waitUntilSettled(name, limitMs = 120000) {
|
|
192
101
|
const deadline = Date.now() + limitMs;
|
|
193
|
-
while
|
|
194
|
-
const vm = (await pool()).find((
|
|
195
|
-
if (!vm || vm.state
|
|
196
|
-
|
|
197
|
-
await new Promise((resolve) => setTimeout(resolve, 5_000));
|
|
102
|
+
while(Date.now() < deadline){
|
|
103
|
+
const vm = (await pool()).find((v)=>v.name === name);
|
|
104
|
+
if (!vm || "stopping" !== vm.state) return;
|
|
105
|
+
await new Promise((resolve)=>setTimeout(resolve, 5000));
|
|
198
106
|
}
|
|
199
107
|
process.stdout.write(` note: ${name} is still stopping; continuing anyway\n`);
|
|
200
108
|
}
|
|
201
|
-
/** @param {Record<string, any>} vm @param {string[]} files @param {string} expected */
|
|
202
109
|
async function deployTo(vm, files, expected) {
|
|
203
110
|
process.stdout.write(`\n=== ${vm.name} ===\n`);
|
|
204
|
-
// Push needs the guest running; the reboot afterwards is what actually loads the new code.
|
|
205
111
|
await ctl("up", vm.name);
|
|
206
|
-
for (const file of files)
|
|
112
|
+
for (const file of files){
|
|
207
113
|
await push(vm.uuid, file);
|
|
208
114
|
process.stdout.write(` pushed ${file}\n`);
|
|
209
115
|
}
|
|
210
116
|
process.stdout.write(" rebooting (utmctl exec cannot be trusted to restart the worker) ...\n");
|
|
211
117
|
await ctl("stop", vm.name);
|
|
212
118
|
await ctl("up", vm.name);
|
|
213
|
-
// The restore is in a `finally` because it used to be on the SUCCESS path only, and the failure path is
|
|
214
|
-
// exactly when it matters. A guest whose health check threw was left RUNNING, and the loop then started
|
|
215
|
-
// the next VM on top of it — on a host `doctor` reports as having room for one of two, that guarantees
|
|
216
|
-
// the second times out too. It is then printed as "stale or failed", which reads as a broken guest and
|
|
217
|
-
// sent me looking to rebuild one that boots to ready in 33 s.
|
|
218
119
|
try {
|
|
219
120
|
const fresh = await pool();
|
|
220
|
-
const back = fresh.find((
|
|
221
|
-
if (!back?.ip)
|
|
222
|
-
throw new Error(`${vm.name} did not come back with an address`);
|
|
121
|
+
const back = fresh.find((v)=>v.name === vm.name);
|
|
122
|
+
if (!back?.ip) throw new Error(`${vm.name} did not come back with an address`);
|
|
223
123
|
const actual = await healthCodeWhenAwake(back.ip, back.port);
|
|
224
124
|
const ok = actual === expected;
|
|
225
125
|
process.stdout.write(` /health.code ${actual} ${ok ? "== expected" : `!= expected ${expected}`}\n`);
|
|
226
126
|
return ok;
|
|
227
|
-
}
|
|
228
|
-
|
|
229
|
-
// Put it back where it was found, the same contract the run's lease honours.
|
|
230
|
-
if (vm.state !== "started")
|
|
231
|
-
await ctl("stop", vm.name).catch(() => undefined);
|
|
127
|
+
} finally{
|
|
128
|
+
if ("started" !== vm.state) await ctl("stop", vm.name).catch(()=>void 0);
|
|
232
129
|
}
|
|
233
130
|
}
|
|
234
|
-
/**
|
|
235
|
-
* Refuse to deploy a CAPTURE_PROTOCOL_VERSION change unless it is asked for explicitly.
|
|
236
|
-
*
|
|
237
|
-
* This deploys the WORKING TREE, which is right for testing a change and dangerous for one specific
|
|
238
|
-
* change: the protocol version is a capture-cache key input, so shipping a bump invalidates every capture
|
|
239
|
-
* on disk and forces a full recapture (`RECAPTURE_COST`, in protocol-guard.mjs, says how big).
|
|
240
|
-
*
|
|
241
|
-
* The trap is real and was live in this repo. An uncommitted bump in a shared checkout makes
|
|
242
|
-
* `npm run worker:code` report every worker STALE, and the remedy it prints is "redeploy" — which would
|
|
243
|
-
* deploy the bump, wipe the cache, and give no clue why the next run recaptured everything.
|
|
244
|
-
*/
|
|
245
131
|
function guardProtocolChange() {
|
|
246
|
-
const inTree = String(
|
|
132
|
+
const inTree = String(CAPTURE_PROTOCOL_VERSION);
|
|
247
133
|
let committed;
|
|
248
134
|
try {
|
|
249
|
-
committed = /CAPTURE_PROTOCOL_VERSION = (\d+)/.exec(execFileSync("git", [
|
|
250
|
-
|
|
251
|
-
|
|
252
|
-
|
|
253
|
-
|
|
254
|
-
|
|
255
|
-
|
|
256
|
-
|
|
135
|
+
committed = /CAPTURE_PROTOCOL_VERSION = (\d+)/.exec(execFileSync("git", [
|
|
136
|
+
"-C",
|
|
137
|
+
NVDA_DIR,
|
|
138
|
+
"show",
|
|
139
|
+
"HEAD:./protocol-version.mjs"
|
|
140
|
+
], {
|
|
141
|
+
encoding: "utf8",
|
|
142
|
+
env: sandboxGitEnv()
|
|
143
|
+
}))?.[1];
|
|
144
|
+
} catch {
|
|
257
145
|
try {
|
|
258
|
-
execFileSync("git", [
|
|
259
|
-
|
|
260
|
-
|
|
261
|
-
|
|
262
|
-
|
|
263
|
-
|
|
264
|
-
|
|
265
|
-
|
|
266
|
-
|
|
267
|
-
|
|
268
|
-
if (process.argv.includes("--allow-protocol-change")) {
|
|
269
|
-
process.stdout.write(`\nDeploying CAPTURE_PROTOCOL_VERSION ${committed} -> ${inTree} as requested. ` +
|
|
270
|
-
"Every cached capture is now invalid and the next run will recapture all of them.\n");
|
|
146
|
+
execFileSync("git", [
|
|
147
|
+
"rev-parse",
|
|
148
|
+
"--verify",
|
|
149
|
+
"HEAD"
|
|
150
|
+
], {
|
|
151
|
+
stdio: "ignore",
|
|
152
|
+
env: sandboxGitEnv()
|
|
153
|
+
});
|
|
154
|
+
process.stdout.write(` note: cannot compare CAPTURE_PROTOCOL_VERSION against HEAD — ${external_node_path_resolve(NVDA_DIR, "protocol-version.mjs")} is not in HEAD.\n Expected for a brand-new or just-moved file; if the path moved, fix it here or this guard is off.\n`);
|
|
155
|
+
} catch {}
|
|
271
156
|
return;
|
|
272
157
|
}
|
|
273
|
-
|
|
274
|
-
|
|
275
|
-
|
|
276
|
-
`full recapture: ${RECAPTURE_COST}. If a \`worker:code\` STALE report sent you here, the stale\n` +
|
|
277
|
-
"hash is probably caused by this uncommitted bump rather than by the guests being out of date.\n\n" +
|
|
278
|
-
" git stash # deploy without the bump, or\n" +
|
|
279
|
-
" npm run worker:deploy -- --allow-protocol-change # deploy it deliberately\n");
|
|
158
|
+
if (!inTree || !committed || inTree === committed) return;
|
|
159
|
+
if (process.argv.includes("--allow-protocol-change")) return void process.stdout.write(`\nDeploying CAPTURE_PROTOCOL_VERSION ${committed} -> ${inTree} as requested. Every cached capture is now invalid and the next run will recapture all of them.\n`);
|
|
160
|
+
process.stderr.write(`\nREFUSING TO DEPLOY: the working tree has CAPTURE_PROTOCOL_VERSION = ${inTree}, but HEAD has ${committed}.\n\nThat value is a capture-cache key, so deploying it invalidates all cached captures and forces a\nfull recapture: ${RECAPTURE_COST}. If a \`worker:code\` STALE report sent you here, the stale\nhash is probably caused by this uncommitted bump rather than by the guests being out of date.\n\n git stash # deploy without the bump, or\n npm run worker:deploy -- --allow-protocol-change # deploy it deliberately\n`);
|
|
280
161
|
process.exit(3);
|
|
281
162
|
}
|
|
282
|
-
/**
|
|
283
|
-
* Nothing here runs on import.
|
|
284
|
-
*
|
|
285
|
-
* This module used to execute its whole deploy at module scope, so merely importing it — which I did, to read a
|
|
286
|
-
* path constant — ran `guardProtocolChange()` and began enumerating VMs. With guests running it would have
|
|
287
|
-
* pushed files and rebooted them. A program that reboots machines must be invoked, never merely mentioned.
|
|
288
|
-
*
|
|
289
|
-
* `check-worker-code.mjs` got the same treatment. The four remaining scripts in this repo that execute at
|
|
290
|
-
* module scope are pure programs nothing imports; they are left alone deliberately rather than restructured for
|
|
291
|
-
* symmetry.
|
|
292
|
-
*/
|
|
293
163
|
async function main() {
|
|
294
164
|
warnUtmDeprecated("npm run worker:deploy");
|
|
295
165
|
guardProtocolChange();
|
|
@@ -303,18 +173,13 @@ async function main() {
|
|
|
303
173
|
process.stdout.write(`Deploying ${files.length} file(s) to ${vms.length} worker(s)\n`);
|
|
304
174
|
process.stdout.write(`Files: ${files.join(", ")}\nExpected code: ${expected}\n`);
|
|
305
175
|
const failed = [];
|
|
306
|
-
for (const vm of vms)
|
|
176
|
+
for (const vm of vms){
|
|
307
177
|
try {
|
|
308
|
-
if (!await deployTo(vm, files, expected))
|
|
309
|
-
|
|
310
|
-
|
|
311
|
-
catch (error) {
|
|
312
|
-
process.stdout.write(` FAILED: ${ /** @type {Error} */(error).message}\n`);
|
|
178
|
+
if (!await deployTo(vm, files, expected)) failed.push(vm.name);
|
|
179
|
+
} catch (error) {
|
|
180
|
+
process.stdout.write(` FAILED: ${error.message}\n`);
|
|
313
181
|
failed.push(vm.name);
|
|
314
182
|
}
|
|
315
|
-
// A `stop` returns before the guest has actually released its memory — one was observed sitting in
|
|
316
|
-
// `stopping` for minutes. Starting the next VM into that overlap is the same over-commitment by a
|
|
317
|
-
// second route, so wait for the host to be quiet before moving on.
|
|
318
183
|
await waitUntilSettled(vm.name);
|
|
319
184
|
}
|
|
320
185
|
process.stdout.write(`\n${vms.length - failed.length}/${vms.length} worker(s) on ${expected}\n`);
|
|
@@ -324,10 +189,4 @@ async function main() {
|
|
|
324
189
|
}
|
|
325
190
|
process.exit(failed.length ? 1 : 0);
|
|
326
191
|
}
|
|
327
|
-
|
|
328
|
-
// is not, so a bin reached via its `.bin` symlink (which is how npm always installs one) mismatched here
|
|
329
|
-
// and this guard silently read false — the tool loaded, did nothing, and exited 0. `/var` and `/tmp` are
|
|
330
|
-
// themselves symlinks on macOS, so this fired every time. Same defect, same fix, as `cli.ts`'s `isProgram`.
|
|
331
|
-
if (import.meta.url === pathToFileURL(process.argv[1] ? realpathSync(process.argv[1]) : "").href)
|
|
332
|
-
await main();
|
|
333
|
-
//# sourceMappingURL=deploy-worker.mjs.map
|
|
192
|
+
if (import.meta.url === pathToFileURL(process.argv[1] ? realpathSync(process.argv[1]) : "").href) await main();
|
package/dist/doctor.d.mts
CHANGED