@celilo/e2e 0.11.3 → 0.13.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +5 -5
- package/bin/e2e-status +1 -1
- package/bin/e2e-up +2 -2
- package/config/dns/tangohost.com.zone +1 -1
- package/config/dns/templates/example.net.zone +3 -3
- package/config/dns/templates/iamtheinternet.org.zone +2 -2
- package/config/resolver/unbound.conf +4 -0
- package/config/routing/fw-ext-routes.sh +8 -8
- package/config/routing/fw-isp-routes.sh +2 -2
- package/config/routing/fw-main-routes.sh +5 -5
- package/config/routing/management-routes.sh +1 -1
- package/config/routing/observer-setup.sh +1 -1
- package/config/routing/public-resolver-routes.sh +33 -0
- package/config/routing/public-sim-entrypoint.sh +1 -1
- package/config/routing/resolver-internal-routes.sh +7 -2
- package/config/routing/resolver-routes.sh +1 -1
- package/config/routing/target-routes.sh +1 -1
- package/config/routing/target-setup.sh +1 -1
- package/config/socks/startup.sh +2 -2
- package/docker/Dockerfile.ip-echo +20 -0
- package/docker/Dockerfile.resolver +5 -1
- package/package.json +5 -4
- package/simulators/greenwave/server.ts +35 -0
- package/simulators/greenwave/state.ts +79 -7
- package/simulators/ip-echo/server.ts +76 -0
- package/src/address-plan.test.ts +4 -4
- package/src/cli/build.ts +30 -0
- package/src/cli/command-registry.ts +19 -0
- package/src/cli/completion.ts +16 -1
- package/src/cli/index.ts +85 -4
- package/src/container-manager.ts +75 -12
- package/src/docker-compose-generator.ts +97 -29
- package/src/doctor.test.ts +279 -0
- package/src/doctor.ts +421 -0
- package/src/extract-failure.ts +41 -0
- package/src/index.ts +12 -0
- package/src/last-run.test.ts +62 -0
- package/src/last-run.ts +54 -0
- package/src/network-builder.ts +11 -0
- package/src/observer.test.ts +2 -2
- package/src/observer.ts +1 -1
- package/src/router-swap.ts +121 -0
- package/src/run-lock.test.ts +73 -2
- package/src/run-lock.ts +110 -14
- package/src/runner.ts +99 -5
- package/src/simulator-ips.ts +21 -0
- package/src/socks-proxy.ts +2 -2
- package/src/types.ts +34 -4
- package/src/vantage.test.ts +1 -1
- package/src/zone-classifier.test.ts +11 -16
- package/src/zone-classifier.ts +16 -31
|
@@ -0,0 +1,121 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Swap the physical router the `fw-isp` simulator impersonates, on a RUNNING
|
|
3
|
+
* network — the harness half of the ISP-replaced-the-box scenario.
|
|
4
|
+
*
|
|
5
|
+
* `.axonRouter()` on the network builder cannot express this: it sets
|
|
6
|
+
* `ROUTER_VENDOR_PREFIX` in the generated compose file, which is read once when
|
|
7
|
+
* the simulator process starts. That is the right tool for "this network has an
|
|
8
|
+
* Axon in it"; it cannot say "this network had a GreenWave and now has an Axon",
|
|
9
|
+
* which is the event the whole module-pause change exists to survive.
|
|
10
|
+
*
|
|
11
|
+
* The swap models what actually happens to a fleet when an ISP replaces the
|
|
12
|
+
* hardware, and each part is load-bearing for what the migration test proves:
|
|
13
|
+
*
|
|
14
|
+
* - the TR-181 vendor prefix changes, so the OLD module's spelling is wrong;
|
|
15
|
+
* - the credentials change, which is why the fleet wedges rather than merely
|
|
16
|
+
* misbehaving — the old module cannot authenticate, so it can neither drive
|
|
17
|
+
* the new box nor clean up after itself;
|
|
18
|
+
* - every port forward is gone, because a new device arrives empty. That is
|
|
19
|
+
* the assertion with teeth: the forwards have to be recreated by unpausing
|
|
20
|
+
* the consumers that own them, and a simulator that kept its old table would
|
|
21
|
+
* let a completely broken unpause pass.
|
|
22
|
+
*/
|
|
23
|
+
|
|
24
|
+
import { greenwaveRouterIp } from './types';
|
|
25
|
+
import type { NetworkHandle } from './types';
|
|
26
|
+
|
|
27
|
+
/** The devices the shared simulator can stand in for. */
|
|
28
|
+
export const ROUTER_DEVICES = {
|
|
29
|
+
/** GreenWave C4000XG — the default, driven by `modules/greenwave`. */
|
|
30
|
+
greenwave: { vendorPrefix: 'X_GWS_', username: 'admin', password: 'admin' },
|
|
31
|
+
/** Axon Networks Q1000K — driven by `modules/axon`. */
|
|
32
|
+
axon: { vendorPrefix: 'X_AXON_', username: 'axonadmin', password: 'axonsecret' },
|
|
33
|
+
} as const;
|
|
34
|
+
|
|
35
|
+
export type RouterDevice = keyof typeof ROUTER_DEVICES;
|
|
36
|
+
|
|
37
|
+
export interface RouterSwapResult {
|
|
38
|
+
/** Port forwards the outgoing device was holding, all of which are now gone. */
|
|
39
|
+
forwardsDiscarded: number;
|
|
40
|
+
device: { vendorPrefix: string; username: string };
|
|
41
|
+
}
|
|
42
|
+
|
|
43
|
+
/**
|
|
44
|
+
* Replace the impersonated device. Runs the request from INSIDE the simulator
|
|
45
|
+
* container, so the control endpoint never has to be reachable from anywhere
|
|
46
|
+
* else on the simulated internet.
|
|
47
|
+
*/
|
|
48
|
+
export async function swapRouterDevice(
|
|
49
|
+
net: NetworkHandle,
|
|
50
|
+
device: RouterDevice,
|
|
51
|
+
): Promise<RouterSwapResult> {
|
|
52
|
+
const spec = ROUTER_DEVICES[device];
|
|
53
|
+
const result = await net.exec(
|
|
54
|
+
'fw-isp',
|
|
55
|
+
`curl -sk -X POST https://${greenwaveRouterIp()}/sim/swap-device ` +
|
|
56
|
+
`-d 'vendorPrefix=${spec.vendorPrefix}&username=${spec.username}&password=${spec.password}'`,
|
|
57
|
+
);
|
|
58
|
+
if (result.exitCode !== 0) {
|
|
59
|
+
throw new Error(`swapRouterDevice(${device}) failed: ${result.stderr || result.stdout}`);
|
|
60
|
+
}
|
|
61
|
+
return JSON.parse(result.stdout) as RouterSwapResult;
|
|
62
|
+
}
|
|
63
|
+
|
|
64
|
+
/**
|
|
65
|
+
* The port forwards the device is currently holding, read back OFF the device
|
|
66
|
+
* rather than out of celilo's own records.
|
|
67
|
+
*
|
|
68
|
+
* This exists because "verify the contract, not the verdict" is the whole point
|
|
69
|
+
* of the migration test: celilo reporting that it registered a forward is not
|
|
70
|
+
* evidence the router has one, and those are exactly the two things that came
|
|
71
|
+
* apart when the hardware changed underneath.
|
|
72
|
+
*/
|
|
73
|
+
export async function listRouterForwards(net: NetworkHandle, device: RouterDevice) {
|
|
74
|
+
const spec = ROUTER_DEVICES[device];
|
|
75
|
+
const login = await net.exec(
|
|
76
|
+
'fw-isp',
|
|
77
|
+
`curl -sk -X POST https://${greenwaveRouterIp()}/cgi/cgi_action ` +
|
|
78
|
+
`-d 'username=${spec.username}&password=${spec.password}' -c /tmp/swap-cookies`,
|
|
79
|
+
);
|
|
80
|
+
if (login.exitCode !== 0) {
|
|
81
|
+
throw new Error(`could not log in to the ${device} router: ${login.stderr || login.stdout}`);
|
|
82
|
+
}
|
|
83
|
+
const listed = await net.exec(
|
|
84
|
+
'fw-isp',
|
|
85
|
+
`curl -sk 'https://${greenwaveRouterIp()}/cgi/cgi_get?Object=Device.NAT.PortMapping.' -b /tmp/swap-cookies`,
|
|
86
|
+
);
|
|
87
|
+
return listed.stdout;
|
|
88
|
+
}
|
|
89
|
+
|
|
90
|
+
/**
|
|
91
|
+
* The external ports currently forwarded on the device, as a sorted list.
|
|
92
|
+
*
|
|
93
|
+
* Assert on THIS rather than on a consumer's container IP. A forward's
|
|
94
|
+
* `InternalClient` is the firewall's natIp DNAT ingress (e.g. `10.226.1.253`),
|
|
95
|
+
* not the consumer's zone-side address — the delegation chain hops through the
|
|
96
|
+
* firewall — so matching a module's own IP looks correct and never matches.
|
|
97
|
+
*/
|
|
98
|
+
export function forwardedPorts(listing: string): string[] {
|
|
99
|
+
const ports = [...listing.matchAll(/"ParamName":"ExternalPort","ParamValue":"(\d+)"/g)].map(
|
|
100
|
+
(m) => m[1],
|
|
101
|
+
);
|
|
102
|
+
return [...new Set(ports)].sort();
|
|
103
|
+
}
|
|
104
|
+
|
|
105
|
+
/** The DHCP pool's advertised DNS servers, read back off the device. */
|
|
106
|
+
export async function routerDhcpDnsServers(
|
|
107
|
+
net: NetworkHandle,
|
|
108
|
+
device: RouterDevice,
|
|
109
|
+
): Promise<string> {
|
|
110
|
+
const spec = ROUTER_DEVICES[device];
|
|
111
|
+
await net.exec(
|
|
112
|
+
'fw-isp',
|
|
113
|
+
`curl -sk -X POST https://${greenwaveRouterIp()}/cgi/cgi_action ` +
|
|
114
|
+
`-d 'username=${spec.username}&password=${spec.password}' -c /tmp/dhcp-cookies`,
|
|
115
|
+
);
|
|
116
|
+
const listed = await net.exec(
|
|
117
|
+
'fw-isp',
|
|
118
|
+
`curl -sk 'https://${greenwaveRouterIp()}/cgi/cgi_get?Object=Device.DHCPv4.Server.Pool.' -b /tmp/dhcp-cookies`,
|
|
119
|
+
);
|
|
120
|
+
return listed.stdout;
|
|
121
|
+
}
|
package/src/run-lock.test.ts
CHANGED
|
@@ -5,8 +5,12 @@ import { join } from 'node:path';
|
|
|
5
5
|
import {
|
|
6
6
|
E2eBusyError,
|
|
7
7
|
type LockHolder,
|
|
8
|
+
SUSPECT_HEARTBEAT_MS,
|
|
8
9
|
acquireRunLock,
|
|
9
10
|
clearLock,
|
|
11
|
+
formatBusy,
|
|
12
|
+
heartbeatAgeMs,
|
|
13
|
+
isSuspect,
|
|
10
14
|
lockStatus,
|
|
11
15
|
markKept,
|
|
12
16
|
readHolder,
|
|
@@ -25,6 +29,7 @@ afterEach(() => {
|
|
|
25
29
|
clearLock();
|
|
26
30
|
rmSync(dir, { recursive: true, force: true });
|
|
27
31
|
delete process.env.CELILO_E2E_LOCK_PATH;
|
|
32
|
+
delete process.env.CELILO_E2E_SESSION;
|
|
28
33
|
});
|
|
29
34
|
|
|
30
35
|
function writeRawHolder(h: Partial<LockHolder>): void {
|
|
@@ -71,7 +76,8 @@ test('a dead-pid holder on the same host is stale and reclaimed', () => {
|
|
|
71
76
|
releaseRunLock();
|
|
72
77
|
});
|
|
73
78
|
|
|
74
|
-
test('--keep leaves a kept lock that survives release and blocks
|
|
79
|
+
test('--keep leaves a kept lock that survives release and blocks other sessions', () => {
|
|
80
|
+
process.env.CELILO_E2E_SESSION = 'keeper (/keeper/worktree)';
|
|
75
81
|
acquireRunLock({ test: 'kept-test', runId: 'r' });
|
|
76
82
|
markKept();
|
|
77
83
|
releaseRunLock();
|
|
@@ -80,7 +86,8 @@ test('--keep leaves a kept lock that survives release and blocks the next run',
|
|
|
80
86
|
expect(h?.state).toBe('kept');
|
|
81
87
|
expect(lockStatus().free).toBe(false); // kept is never stale
|
|
82
88
|
|
|
83
|
-
// A plain run is refused...
|
|
89
|
+
// A plain run from ANOTHER session is refused...
|
|
90
|
+
process.env.CELILO_E2E_SESSION = 'other (/other/worktree)';
|
|
84
91
|
expect(() => acquireRunLock({ test: 'next', runId: 'r2' })).toThrow(E2eBusyError);
|
|
85
92
|
// ...but a --reuse run (allowKept) takes it over.
|
|
86
93
|
acquireRunLock({ test: 'reuse', runId: 'r3', allowKept: true });
|
|
@@ -96,6 +103,70 @@ test('clearLock frees a kept lock (the `cele2e release` path)', () => {
|
|
|
96
103
|
expect(lockStatus().free).toBe(true);
|
|
97
104
|
});
|
|
98
105
|
|
|
106
|
+
test('a kept lock left by THIS session is auto-released by the next run', () => {
|
|
107
|
+
// The friction case: `run --keep` then `run` from the same session. Refusing
|
|
108
|
+
// here protected nobody — the only stack at risk was the caller's own — and
|
|
109
|
+
// the refusal was routinely misread as a finished run, because the previous
|
|
110
|
+
// run's results dir is still sitting there looking like a clean pass.
|
|
111
|
+
process.env.CELILO_E2E_SESSION = 'my-branch (/my/worktree)';
|
|
112
|
+
acquireRunLock({ test: 'kept-test', runId: 'r' });
|
|
113
|
+
markKept();
|
|
114
|
+
releaseRunLock();
|
|
115
|
+
expect(readHolder()?.state).toBe('kept');
|
|
116
|
+
|
|
117
|
+
const outcome = acquireRunLock({ test: 'next', runId: 'r2' });
|
|
118
|
+
expect(outcome.autoReleasedOwnKept?.test).toBe('kept-test');
|
|
119
|
+
expect(readHolder()?.pid).toBe(process.pid);
|
|
120
|
+
expect(readHolder()?.state).toBe('running');
|
|
121
|
+
releaseRunLock();
|
|
122
|
+
});
|
|
123
|
+
|
|
124
|
+
test('a kept lock from a DIFFERENT session is still refused', () => {
|
|
125
|
+
process.env.CELILO_E2E_SESSION = 'my-branch (/my/worktree)';
|
|
126
|
+
// A live pid so staleness can't be what frees it — the point is the session.
|
|
127
|
+
writeRawHolder({ state: 'kept', session: 'someone-else (/their/worktree)', pid: process.pid });
|
|
128
|
+
expect(() => acquireRunLock({ test: 'mine', runId: 'r' })).toThrow(E2eBusyError);
|
|
129
|
+
});
|
|
130
|
+
|
|
131
|
+
test('a live pid with a silent heartbeat is SUSPECT but never auto-reclaimed', () => {
|
|
132
|
+
// build-infra wedged at step [16/27]: the process is alive (so PID-liveness
|
|
133
|
+
// says "healthy") while its heartbeat has not ticked for half an hour. This
|
|
134
|
+
// is the one hang the existing staleness check structurally cannot see.
|
|
135
|
+
writeRawHolder({ pid: process.pid, beatAt: Date.now() - 1_913_000 });
|
|
136
|
+
const h = readHolder() as LockHolder;
|
|
137
|
+
|
|
138
|
+
expect(isSuspect(h)).toBe(true);
|
|
139
|
+
expect(heartbeatAgeMs(h)).toBeGreaterThan(SUSPECT_HEARTBEAT_MS);
|
|
140
|
+
expect(formatBusy(h)).toContain('SUSPECT');
|
|
141
|
+
|
|
142
|
+
const status = lockStatus();
|
|
143
|
+
expect(status.suspect).toBe(true);
|
|
144
|
+
// Suspect surfaces the hang; it does NOT kill someone's build for them.
|
|
145
|
+
expect(status.free).toBe(false);
|
|
146
|
+
expect(() => acquireRunLock({ test: 'other', runId: 'r' })).toThrow(E2eBusyError);
|
|
147
|
+
});
|
|
148
|
+
|
|
149
|
+
test('a fresh heartbeat is not suspect', () => {
|
|
150
|
+
writeRawHolder({ pid: process.pid, beatAt: Date.now() - 5_000 });
|
|
151
|
+
const h = readHolder() as LockHolder;
|
|
152
|
+
expect(isSuspect(h)).toBe(false);
|
|
153
|
+
expect(lockStatus().suspect).toBe(false);
|
|
154
|
+
});
|
|
155
|
+
|
|
156
|
+
test('a healthy build-infra that blocks the event loop is NOT suspect', () => {
|
|
157
|
+
// spawnSync('docker', ['build', …]) blocks the event loop, so the heartbeat
|
|
158
|
+
// stops for the whole of each image build. A live holder was observed 67s
|
|
159
|
+
// stale while making normal progress; the threshold has to clear that or the
|
|
160
|
+
// flag fires on every large image and stops meaning anything.
|
|
161
|
+
writeRawHolder({ pid: process.pid, beatAt: Date.now() - 120_000 });
|
|
162
|
+
expect(isSuspect(readHolder() as LockHolder)).toBe(false);
|
|
163
|
+
});
|
|
164
|
+
|
|
165
|
+
test('a kept holder is never suspect — its beatAt is frozen on purpose', () => {
|
|
166
|
+
writeRawHolder({ state: 'kept', pid: process.pid, beatAt: Date.now() - 3_600_000 });
|
|
167
|
+
expect(isSuspect(readHolder() as LockHolder)).toBe(false);
|
|
168
|
+
});
|
|
169
|
+
|
|
99
170
|
test('a corrupt lock file is reclaimed, not fatal', () => {
|
|
100
171
|
writeFileSync(process.env.CELILO_E2E_LOCK_PATH as string, 'not json{');
|
|
101
172
|
expect(readHolder()).toBeNull();
|
package/src/run-lock.ts
CHANGED
|
@@ -18,9 +18,14 @@
|
|
|
18
18
|
* - Staleness: on the SAME host, PID-liveness is authoritative (process.kill
|
|
19
19
|
* (pid, 0)); a dead holder PID → reclaim. Cross-host we can't check the PID,
|
|
20
20
|
* so fall back to a heartbeat TTL (beatAt older than STALE_TTL_MS).
|
|
21
|
-
* - A `kept` lock (left by `--keep` / `up`) is
|
|
22
|
-
* guards a stack that outlives the process
|
|
23
|
-
*
|
|
21
|
+
* - A `kept` lock (left by `--keep` / `up`) is never reclaimed by ANOTHER
|
|
22
|
+
* actor — it guards a stack that outlives the process. Its OWN session
|
|
23
|
+
* reclaims it automatically (see isSameSession); everyone else clears it
|
|
24
|
+
* deliberately via `cele2e release` / `cele2e down`.
|
|
25
|
+
* - SUSPECT holders: a live PID whose heartbeat has gone quiet is the one
|
|
26
|
+
* hang the PID check cannot see (a build wedged mid-step keeps its process
|
|
27
|
+
* alive and sleeping). isSuspect() names it so `status`/`doctor` can flag it
|
|
28
|
+
* instead of leaving 20 silent minutes to be found by hand.
|
|
24
29
|
*/
|
|
25
30
|
|
|
26
31
|
import { execFileSync } from 'node:child_process';
|
|
@@ -31,6 +36,23 @@ import { basename, dirname, join } from 'node:path';
|
|
|
31
36
|
const STALE_TTL_MS = 90_000;
|
|
32
37
|
const HEARTBEAT_MS = 30_000;
|
|
33
38
|
|
|
39
|
+
/**
|
|
40
|
+
* How quiet a RUNNING holder's heartbeat may go before it is suspect.
|
|
41
|
+
*
|
|
42
|
+
* Calibrated against what a HEALTHY holder actually does. The heartbeat is a
|
|
43
|
+
* setInterval, and build-infra shells out with spawnSync — which blocks the
|
|
44
|
+
* event loop for the entire duration of each `docker build`. So a perfectly
|
|
45
|
+
* healthy build stops beating for as long as its slowest image takes: a live
|
|
46
|
+
* holder was observed 67s stale while making normal progress, and the largest
|
|
47
|
+
* images here are ~2GB. Anything near the 30s tick interval would flag those.
|
|
48
|
+
*
|
|
49
|
+
* Ten minutes sits above any single legitimate image build (the per-image
|
|
50
|
+
* watchdog caps one at 15) and far below the ~20 and ~32 minute hangs that
|
|
51
|
+
* went unnoticed. Deliberately NOT a reclaim threshold: the PID is alive, and
|
|
52
|
+
* killing someone's wedged build out from under them is the operator's call.
|
|
53
|
+
*/
|
|
54
|
+
export const SUSPECT_HEARTBEAT_MS = 600_000;
|
|
55
|
+
|
|
34
56
|
/**
|
|
35
57
|
* Machine-global lock path — NOT under any repo/worktree, because the Docker
|
|
36
58
|
* infra is global regardless of which checkout started it. Overridable via
|
|
@@ -75,14 +97,22 @@ export function formatBusy(h: LockHolder): string {
|
|
|
75
97
|
h.state === 'kept'
|
|
76
98
|
? '(run `cele2e release` to free it)'
|
|
77
99
|
: `${h.test}, started ${age} ago (pid ${h.pid})`;
|
|
78
|
-
|
|
100
|
+
const suspect = isSuspect(h)
|
|
101
|
+
? ` — SUSPECT: no heartbeat for ${formatAge(heartbeatAgeMs(h))} (process alive but not progressing)`
|
|
102
|
+
: '';
|
|
103
|
+
return `e2e busy: ${h.session} ${verb} ${what}${suspect}`;
|
|
79
104
|
}
|
|
80
105
|
|
|
81
106
|
function ageString(since: number): string {
|
|
82
|
-
|
|
107
|
+
return formatAge(Date.now() - since);
|
|
108
|
+
}
|
|
109
|
+
|
|
110
|
+
/** Human duration, e.g. "45s" / "32m" / "2h05m". */
|
|
111
|
+
export function formatAge(ms: number): string {
|
|
112
|
+
const s = Math.max(0, Math.floor(ms / 1000));
|
|
83
113
|
if (s < 60) return `${s}s`;
|
|
84
114
|
if (s < 3600) return `${Math.floor(s / 60)}m`;
|
|
85
|
-
return `${Math.floor(s / 3600)}h${Math.floor((s % 3600) / 60)}m`;
|
|
115
|
+
return `${Math.floor(s / 3600)}h${String(Math.floor((s % 3600) / 60)).padStart(2, '0')}m`;
|
|
86
116
|
}
|
|
87
117
|
|
|
88
118
|
/** Branch + worktree path so a holder maps back to a specific session/thread. */
|
|
@@ -107,6 +137,28 @@ function deriveSession(): string {
|
|
|
107
137
|
return branch || base;
|
|
108
138
|
}
|
|
109
139
|
|
|
140
|
+
/**
|
|
141
|
+
* Is this holder the caller's own session — same host, same branch+worktree?
|
|
142
|
+
* This is what separates "my own kept stack is in my way" (friction, auto-clear)
|
|
143
|
+
* from "someone else is mid-run" (contention, refuse).
|
|
144
|
+
*/
|
|
145
|
+
export function isSameSession(h: LockHolder): boolean {
|
|
146
|
+
return h.hostname === hostname() && h.session === deriveSession();
|
|
147
|
+
}
|
|
148
|
+
|
|
149
|
+
/** Milliseconds since the holder last refreshed its heartbeat. */
|
|
150
|
+
export function heartbeatAgeMs(h: LockHolder): number {
|
|
151
|
+
return Math.max(0, Date.now() - h.beatAt);
|
|
152
|
+
}
|
|
153
|
+
|
|
154
|
+
/**
|
|
155
|
+
* A live process whose heartbeat has gone quiet — the hang PID-liveness misses.
|
|
156
|
+
* `kept` holders are excluded: their beatAt is frozen on purpose at release.
|
|
157
|
+
*/
|
|
158
|
+
export function isSuspect(h: LockHolder): boolean {
|
|
159
|
+
return h.state === 'running' && heartbeatAgeMs(h) > SUSPECT_HEARTBEAT_MS;
|
|
160
|
+
}
|
|
161
|
+
|
|
110
162
|
function pidAlive(pid: number): boolean {
|
|
111
163
|
try {
|
|
112
164
|
process.kill(pid, 0);
|
|
@@ -140,15 +192,32 @@ function writeHolder(fd: number, h: LockHolder): void {
|
|
|
140
192
|
writeFileSync(fd, JSON.stringify(h, null, 2));
|
|
141
193
|
}
|
|
142
194
|
|
|
195
|
+
/** What the acquire had to clear on the way in, so callers can say so out loud. */
|
|
196
|
+
export interface AcquireOutcome {
|
|
197
|
+
/** A `kept` lock left by this same session that we auto-released, else null. */
|
|
198
|
+
autoReleasedOwnKept: LockHolder | null;
|
|
199
|
+
}
|
|
200
|
+
|
|
143
201
|
/**
|
|
144
202
|
* Acquire the run lock for this process. Throws E2eBusyError if another live
|
|
145
203
|
* (non-stale) holder owns it. Registers an exit handler so the lock is released
|
|
146
204
|
* on any process.exit() path (normal, exception, or a SIGINT handler that
|
|
147
205
|
* exits). A signal-killed process with no exit handler leaks the lock, but the
|
|
148
206
|
* next run reclaims it via PID-liveness — that's exactly what staleness is for.
|
|
207
|
+
*
|
|
208
|
+
* A `kept` lock left by THIS session (same host, same branch+worktree) is
|
|
209
|
+
* auto-released and reported in the outcome. Refusing there protected nobody:
|
|
210
|
+
* the only stack at risk was the caller's own, and the refusal was routinely
|
|
211
|
+
* misread as a finished run — the previous run's results dir is still sitting
|
|
212
|
+
* there looking like a clean pass. Another session's kept lock still refuses.
|
|
149
213
|
*/
|
|
150
|
-
export function acquireRunLock(opts: {
|
|
214
|
+
export function acquireRunLock(opts: {
|
|
215
|
+
test: string;
|
|
216
|
+
runId: string;
|
|
217
|
+
allowKept?: boolean;
|
|
218
|
+
}): AcquireOutcome {
|
|
151
219
|
mkdirSync(dirname(lockPath()), { recursive: true });
|
|
220
|
+
let autoReleasedOwnKept: LockHolder | null = null;
|
|
152
221
|
|
|
153
222
|
const holder: LockHolder = {
|
|
154
223
|
pid: process.pid,
|
|
@@ -168,9 +237,17 @@ export function acquireRunLock(opts: { test: string; runId: string; allowKept?:
|
|
|
168
237
|
} catch (err) {
|
|
169
238
|
if ((err as NodeJS.ErrnoException).code !== 'EEXIST') throw err;
|
|
170
239
|
const existing = readHolder();
|
|
171
|
-
// Reclaim when: corrupt/unreadable, the holder is stale,
|
|
172
|
-
// `--reuse` run taking over a `kept` stack it's deliberately reusing
|
|
173
|
-
|
|
240
|
+
// Reclaim when: corrupt/unreadable, the holder is stale, this is a
|
|
241
|
+
// `--reuse` run taking over a `kept` stack it's deliberately reusing, OR
|
|
242
|
+
// the `kept` stack belongs to this very session (see the doc comment).
|
|
243
|
+
const ownKept = !!existing && existing.state === 'kept' && isSameSession(existing);
|
|
244
|
+
if (
|
|
245
|
+
!existing ||
|
|
246
|
+
isStale(existing) ||
|
|
247
|
+
(opts.allowKept && existing.state === 'kept') ||
|
|
248
|
+
ownKept
|
|
249
|
+
) {
|
|
250
|
+
if (ownKept && existing) autoReleasedOwnKept = existing;
|
|
174
251
|
try {
|
|
175
252
|
unlinkSync(lockPath());
|
|
176
253
|
} catch {}
|
|
@@ -195,7 +272,7 @@ export function acquireRunLock(opts: { test: string; runId: string; allowKept?:
|
|
|
195
272
|
|
|
196
273
|
held = { keepOnRelease: false, heartbeat, released: false };
|
|
197
274
|
process.on('exit', releaseRunLock);
|
|
198
|
-
return;
|
|
275
|
+
return { autoReleasedOwnKept };
|
|
199
276
|
}
|
|
200
277
|
// Two reclaim attempts both lost the race → someone else won fair and square.
|
|
201
278
|
const existing = readHolder();
|
|
@@ -258,9 +335,28 @@ export function clearLock(): boolean {
|
|
|
258
335
|
}
|
|
259
336
|
}
|
|
260
337
|
|
|
338
|
+
export interface LockStatus {
|
|
339
|
+
free: boolean;
|
|
340
|
+
holder: LockHolder | null;
|
|
341
|
+
/** Milliseconds since the holder's last heartbeat; null when free. */
|
|
342
|
+
heartbeatAgeMs: number | null;
|
|
343
|
+
/** Live PID, quiet heartbeat — the hang the PID check cannot see. */
|
|
344
|
+
suspect: boolean;
|
|
345
|
+
/** The holder is this session's own kept stack, which the next run reclaims. */
|
|
346
|
+
ownKept: boolean;
|
|
347
|
+
}
|
|
348
|
+
|
|
261
349
|
/** Current lock state for `cele2e status` (and any poller). */
|
|
262
|
-
export function lockStatus():
|
|
350
|
+
export function lockStatus(): LockStatus {
|
|
263
351
|
const h = readHolder();
|
|
264
|
-
if (!h || isStale(h))
|
|
265
|
-
|
|
352
|
+
if (!h || isStale(h)) {
|
|
353
|
+
return { free: true, holder: null, heartbeatAgeMs: null, suspect: false, ownKept: false };
|
|
354
|
+
}
|
|
355
|
+
return {
|
|
356
|
+
free: false,
|
|
357
|
+
holder: h,
|
|
358
|
+
heartbeatAgeMs: heartbeatAgeMs(h),
|
|
359
|
+
suspect: isSuspect(h),
|
|
360
|
+
ownKept: h.state === 'kept' && isSameSession(h),
|
|
361
|
+
};
|
|
266
362
|
}
|
package/src/runner.ts
CHANGED
|
@@ -29,7 +29,9 @@ import {
|
|
|
29
29
|
emitTestCompleted,
|
|
30
30
|
emitTestStarted,
|
|
31
31
|
} from './bus-events';
|
|
32
|
-
import {
|
|
32
|
+
import { diagnose, formatReport } from './doctor';
|
|
33
|
+
import { type StageTally, extractFailureMessage, stripAnsi, tallyStages } from './extract-failure';
|
|
34
|
+
import { writeLastRun } from './last-run';
|
|
33
35
|
import { parseLine } from './parse-line';
|
|
34
36
|
import { E2eBusyError, acquireRunLock, markKept } from './run-lock';
|
|
35
37
|
import { SIMULATOR_IPS } from './simulator-ips';
|
|
@@ -200,6 +202,12 @@ interface TestResult {
|
|
|
200
202
|
rawTail?: string;
|
|
201
203
|
hadDebugPause?: boolean;
|
|
202
204
|
projectName?: string;
|
|
205
|
+
/**
|
|
206
|
+
* Real stage failures vs stages a failing earlier stage blocked. Without the
|
|
207
|
+
* split, one bad fixture line in stage 1 of a 10-stage suite reports as "9
|
|
208
|
+
* failed" — nine counts of a defect that does not exist.
|
|
209
|
+
*/
|
|
210
|
+
stages?: StageTally;
|
|
203
211
|
}
|
|
204
212
|
|
|
205
213
|
async function runTest(
|
|
@@ -399,12 +407,24 @@ async function runTest(
|
|
|
399
407
|
projectName: capturedProjectName,
|
|
400
408
|
error: extractFailureMessage(lines, code ?? 1),
|
|
401
409
|
rawTail,
|
|
410
|
+
stages: tallyStages(lines),
|
|
402
411
|
});
|
|
403
412
|
}
|
|
404
413
|
});
|
|
405
414
|
});
|
|
406
415
|
}
|
|
407
416
|
|
|
417
|
+
/**
|
|
418
|
+
* " — 1 stage failed, 8 skipped (blocked by an earlier stage)". Rendered only
|
|
419
|
+
* when a cascade actually happened, so an ordinary single-stage failure stays
|
|
420
|
+
* as terse as it was.
|
|
421
|
+
*/
|
|
422
|
+
function stageSuffix(stages: StageTally | undefined): string {
|
|
423
|
+
if (!stages || stages.skipped === 0) return '';
|
|
424
|
+
const failed = `${stages.failed} stage${stages.failed === 1 ? '' : 's'} failed`;
|
|
425
|
+
return ` ${dim}— ${failed}, ${stages.skipped} skipped (blocked by an earlier stage)${reset}`;
|
|
426
|
+
}
|
|
427
|
+
|
|
408
428
|
// ─── Shared Infrastructure ───────────────────────────────────────────
|
|
409
429
|
|
|
410
430
|
function dockerComposeShared(cmd: string): string {
|
|
@@ -475,11 +495,20 @@ async function main() {
|
|
|
475
495
|
// docker mutation so a competing run can't wipe our shared infra mid-setup.
|
|
476
496
|
const lockLabel = patterns.length ? patterns.join(',') : moduleDirs.length ? 'modules' : 'all';
|
|
477
497
|
try {
|
|
478
|
-
acquireRunLock({
|
|
498
|
+
const outcome = acquireRunLock({
|
|
479
499
|
test: lockLabel,
|
|
480
500
|
runId: process.env.CELE2E_RUN_ID ?? ambientRunId,
|
|
481
501
|
allowKept: flagReuse,
|
|
482
502
|
});
|
|
503
|
+
if (outcome.autoReleasedOwnKept) {
|
|
504
|
+
const own = outcome.autoReleasedOwnKept;
|
|
505
|
+
console.log(
|
|
506
|
+
`${dim}Auto-released your own kept stack (${own.test}, held since ${own.startedAt}) and continuing.${reset}`,
|
|
507
|
+
);
|
|
508
|
+
console.log(
|
|
509
|
+
`${dim}Use \`cele2e run --reuse\` instead if you meant to run against it.${reset}`,
|
|
510
|
+
);
|
|
511
|
+
}
|
|
483
512
|
// A --reuse run keeps using a kept stack that stays up afterwards — hold the
|
|
484
513
|
// lock in kept state so it isn't freed out from under the reused network.
|
|
485
514
|
if (flagReuse) markKept();
|
|
@@ -494,6 +523,25 @@ async function main() {
|
|
|
494
523
|
throw err;
|
|
495
524
|
}
|
|
496
525
|
|
|
526
|
+
// Preflight the environment BEFORE touching Docker. Every check here failed
|
|
527
|
+
// silently once and surfaced later somewhere unrelated — a missing bake as an
|
|
528
|
+
// SSH error against a firewall IP, a pruned base image as a TLS timeout 16
|
|
529
|
+
// images into a build. Refusing now costs seconds; not refusing cost a full
|
|
530
|
+
// run plus a debugging cycle each time. Lock state is excluded: we already
|
|
531
|
+
// hold the lock, so checkRunLock would report ourselves as contention.
|
|
532
|
+
const health = diagnose({ pkgDir: PKG_DIR, skipLock: true });
|
|
533
|
+
if (!health.ok) {
|
|
534
|
+
console.error(
|
|
535
|
+
`\n${red}✗ cele2e preflight failed — not starting a run that cannot succeed.${reset}`,
|
|
536
|
+
);
|
|
537
|
+
for (const line of formatReport(health)) console.error(line);
|
|
538
|
+
console.error(`\n${dim}Full environment report: cele2e doctor${reset}\n`);
|
|
539
|
+
process.exit(1);
|
|
540
|
+
}
|
|
541
|
+
for (const c of health.checks) {
|
|
542
|
+
if (c.status === 'warn') console.log(`${dim}! ${c.name}: ${c.detail}${reset}`);
|
|
543
|
+
}
|
|
544
|
+
|
|
497
545
|
let testFiles: string[] = [];
|
|
498
546
|
|
|
499
547
|
if (moduleDirs.length > 0) {
|
|
@@ -656,6 +704,22 @@ async function main() {
|
|
|
656
704
|
if (!process.env.CELE2E_RUN_ID) {
|
|
657
705
|
console.log(`${dim}runId: ${runId}${reset}`);
|
|
658
706
|
}
|
|
707
|
+
// Print the results dir HERE, not only at the end: a run that dies mid-way
|
|
708
|
+
// still wrote logs there, and `ls -t results | head -1` is not a safe way to
|
|
709
|
+
// find them — after a refused start it returns the PREVIOUS run's numbers,
|
|
710
|
+
// which look exactly like a clean pass. `cele2e last` reads the same record.
|
|
711
|
+
console.log(`${dim}results: ${resultsDir}${reset}`);
|
|
712
|
+
writeLastRun({
|
|
713
|
+
runId,
|
|
714
|
+
resultsDir,
|
|
715
|
+
startedAt: new Date(suiteStart).toISOString(),
|
|
716
|
+
status: 'running',
|
|
717
|
+
total: testFiles.length,
|
|
718
|
+
passed: 0,
|
|
719
|
+
failed: 0,
|
|
720
|
+
skipped: 0,
|
|
721
|
+
durationMs: 0,
|
|
722
|
+
});
|
|
659
723
|
emitRunStarted({
|
|
660
724
|
runId,
|
|
661
725
|
scenario: testFiles.length > 1 ? 'multi-test' : 'single-test',
|
|
@@ -784,7 +848,7 @@ async function main() {
|
|
|
784
848
|
console.log(` ${red} └─ ${result.error}${reset}`);
|
|
785
849
|
failed++;
|
|
786
850
|
} else {
|
|
787
|
-
console.log(` ${red}✗ failed (${result.duration}s)${reset}`);
|
|
851
|
+
console.log(` ${red}✗ failed (${result.duration}s)${reset}${stageSuffix(result.stages)}`);
|
|
788
852
|
if (result.error) {
|
|
789
853
|
console.log(` ${red} └─ ${result.error}${reset}`);
|
|
790
854
|
}
|
|
@@ -817,18 +881,28 @@ async function main() {
|
|
|
817
881
|
|
|
818
882
|
const suiteDuration = Math.floor((Date.now() - suiteStart) / 1000);
|
|
819
883
|
|
|
884
|
+
// Stages a failing earlier stage blocked are reported separately from real
|
|
885
|
+
// failures. Counting them together turns one bad fixture line into "9 failed"
|
|
886
|
+
// and buries the single defect that actually exists.
|
|
887
|
+
const stageSkipped = results.reduce((sum, r) => sum + (r.stages?.skipped ?? 0), 0);
|
|
888
|
+
|
|
820
889
|
console.log();
|
|
821
890
|
console.log(`${bold}╔══════════════════════════════════════════════════════════════${reset}`);
|
|
822
891
|
console.log(
|
|
823
892
|
`${bold}║ Results: ${green}${passed} passed${reset}${bold}, ${red}${failed} failed${reset}${bold} — ${formatDuration(suiteDuration)} total${reset}`,
|
|
824
893
|
);
|
|
894
|
+
if (stageSkipped > 0) {
|
|
895
|
+
console.log(
|
|
896
|
+
`${bold}║${reset} ${dim}${stageSkipped} stage(s) skipped — blocked by an earlier stage, not separate defects${reset}`,
|
|
897
|
+
);
|
|
898
|
+
}
|
|
825
899
|
console.log(`${bold}╠══════════════════════════════════════════════════════════════${reset}`);
|
|
826
900
|
|
|
827
901
|
for (const r of results) {
|
|
828
902
|
const icon = r.status === 'pass' ? '✓' : '✗';
|
|
829
903
|
const color = r.status === 'pass' ? green : red;
|
|
830
904
|
console.log(
|
|
831
|
-
`${bold}║${reset} ${color}${icon}${reset} ${r.name.padEnd(28)} ${dim}${(`${r.duration}s`).padStart(6)}${reset}`,
|
|
905
|
+
`${bold}║${reset} ${color}${icon}${reset} ${r.name.padEnd(28)} ${dim}${(`${r.duration}s`).padStart(6)}${reset}${stageSuffix(r.stages)}`,
|
|
832
906
|
);
|
|
833
907
|
if (r.error) {
|
|
834
908
|
console.log(`${bold}║${reset} ${red}└─ ${r.error}${reset}`);
|
|
@@ -838,16 +912,36 @@ async function main() {
|
|
|
838
912
|
console.log(`${bold}╚══════════════════════════════════════════════════════════════${reset}`);
|
|
839
913
|
console.log();
|
|
840
914
|
console.log(`${dim}Full logs: ${resultsDir}${reset}`);
|
|
915
|
+
console.log(`${dim}Machine-readable: cele2e last --json${reset}`);
|
|
841
916
|
|
|
842
917
|
writeFileSync(
|
|
843
918
|
join(resultsDir, 'summary.json'),
|
|
844
919
|
JSON.stringify(
|
|
845
|
-
{
|
|
920
|
+
{
|
|
921
|
+
timestamp,
|
|
922
|
+
total: testFiles.length,
|
|
923
|
+
passed,
|
|
924
|
+
failed,
|
|
925
|
+
stageSkipped,
|
|
926
|
+
duration: suiteDuration,
|
|
927
|
+
},
|
|
846
928
|
null,
|
|
847
929
|
2,
|
|
848
930
|
),
|
|
849
931
|
);
|
|
850
932
|
|
|
933
|
+
writeLastRun({
|
|
934
|
+
runId,
|
|
935
|
+
resultsDir,
|
|
936
|
+
startedAt: new Date(suiteStart).toISOString(),
|
|
937
|
+
status: failed > 0 ? 'failed' : 'completed',
|
|
938
|
+
total: testFiles.length,
|
|
939
|
+
passed,
|
|
940
|
+
failed,
|
|
941
|
+
skipped: stageSkipped,
|
|
942
|
+
durationMs: suiteDuration * 1000,
|
|
943
|
+
});
|
|
944
|
+
|
|
851
945
|
const gitignore = join(testDir, '.gitignore');
|
|
852
946
|
const content = existsSync(gitignore) ? readFileSync(gitignore, 'utf-8') : '';
|
|
853
947
|
const adds: string[] = [];
|
package/src/simulator-ips.ts
CHANGED
|
@@ -44,6 +44,27 @@ export const SIMULATOR_IPS = {
|
|
|
44
44
|
CPANEL_HOST: '100.64.0.63',
|
|
45
45
|
/** signal-cli release host — serves the tarball the signal module downloads at deploy time. */
|
|
46
46
|
SIGNAL_RELEASE: '100.64.0.62',
|
|
47
|
+
/**
|
|
48
|
+
* OFF-FLEET recursive resolver — the rig's stand-in for 1.1.1.1, and a peer
|
|
49
|
+
* of comcast-resolver rather than a replacement for it.
|
|
50
|
+
*
|
|
51
|
+
* The distinction is the whole point of celilo's `public_dns` check: the
|
|
52
|
+
* fleet's own resolver (its ISP's, or its internal split-horizon one)
|
|
53
|
+
* answers with whatever is correct for a client INSIDE, which is not
|
|
54
|
+
* evidence about what the internet sees. celilo REFUSES to use a resolver it
|
|
55
|
+
* is itself configured to use, so verifying public reachability requires a
|
|
56
|
+
* second, independent public resolver — and until this existed the topology
|
|
57
|
+
* had exactly one.
|
|
58
|
+
*/
|
|
59
|
+
PUBLIC_RESOLVER: '100.64.0.64',
|
|
60
|
+
/**
|
|
61
|
+
* IP echo service — the rig's stand-in for api.ipify.org. Reports the source
|
|
62
|
+
* address a request appears to come from, which for the customer fleet is
|
|
63
|
+
* the firewall's external address after SNAT. The `public_dns` check's
|
|
64
|
+
* expectation comes from here rather than from the registrar's own response,
|
|
65
|
+
* which is self-agreement (and, for a Namecheap `www` update, false).
|
|
66
|
+
*/
|
|
67
|
+
IP_ECHO: '100.64.0.65',
|
|
47
68
|
/** Pebble ACME server (replaces production Let's Encrypt). */
|
|
48
69
|
PEBBLE: '100.64.0.100',
|
|
49
70
|
} as const;
|
package/src/socks-proxy.ts
CHANGED
|
@@ -44,7 +44,7 @@ export interface SocksProxyHandle {
|
|
|
44
44
|
* Why custom: an off-the-shelf SOCKS5 image (e.g. serjs/go-socks5-proxy)
|
|
45
45
|
* starts fine on the test network but can't actually route to anything.
|
|
46
46
|
* Docker's IPAM-default gateway points at an IP with no listener (e.g.
|
|
47
|
-
*
|
|
47
|
+
* 203.0.113.250 on isp-external), and docker rewrites resolv.conf to
|
|
48
48
|
* its embedded resolver, which can't cleanly forward to the simulated
|
|
49
49
|
* network's DNS. The custom image's entrypoint replaces both.
|
|
50
50
|
*/
|
|
@@ -61,7 +61,7 @@ interface VantageConfig {
|
|
|
61
61
|
const VANTAGE_CONFIGS: Record<SocksProxyVantage, VantageConfig> = {
|
|
62
62
|
'isp-external': {
|
|
63
63
|
network: 'isp-external',
|
|
64
|
-
ip: '
|
|
64
|
+
ip: '203.0.113.150',
|
|
65
65
|
},
|
|
66
66
|
internal: {
|
|
67
67
|
network: 'internal',
|