@celilo/e2e 0.11.3 → 0.13.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (51) hide show
  1. package/README.md +5 -5
  2. package/bin/e2e-status +1 -1
  3. package/bin/e2e-up +2 -2
  4. package/config/dns/tangohost.com.zone +1 -1
  5. package/config/dns/templates/example.net.zone +3 -3
  6. package/config/dns/templates/iamtheinternet.org.zone +2 -2
  7. package/config/resolver/unbound.conf +4 -0
  8. package/config/routing/fw-ext-routes.sh +8 -8
  9. package/config/routing/fw-isp-routes.sh +2 -2
  10. package/config/routing/fw-main-routes.sh +5 -5
  11. package/config/routing/management-routes.sh +1 -1
  12. package/config/routing/observer-setup.sh +1 -1
  13. package/config/routing/public-resolver-routes.sh +33 -0
  14. package/config/routing/public-sim-entrypoint.sh +1 -1
  15. package/config/routing/resolver-internal-routes.sh +7 -2
  16. package/config/routing/resolver-routes.sh +1 -1
  17. package/config/routing/target-routes.sh +1 -1
  18. package/config/routing/target-setup.sh +1 -1
  19. package/config/socks/startup.sh +2 -2
  20. package/docker/Dockerfile.ip-echo +20 -0
  21. package/docker/Dockerfile.resolver +5 -1
  22. package/package.json +5 -4
  23. package/simulators/greenwave/server.ts +35 -0
  24. package/simulators/greenwave/state.ts +79 -7
  25. package/simulators/ip-echo/server.ts +76 -0
  26. package/src/address-plan.test.ts +4 -4
  27. package/src/cli/build.ts +30 -0
  28. package/src/cli/command-registry.ts +19 -0
  29. package/src/cli/completion.ts +16 -1
  30. package/src/cli/index.ts +85 -4
  31. package/src/container-manager.ts +75 -12
  32. package/src/docker-compose-generator.ts +97 -29
  33. package/src/doctor.test.ts +279 -0
  34. package/src/doctor.ts +421 -0
  35. package/src/extract-failure.ts +41 -0
  36. package/src/index.ts +12 -0
  37. package/src/last-run.test.ts +62 -0
  38. package/src/last-run.ts +54 -0
  39. package/src/network-builder.ts +11 -0
  40. package/src/observer.test.ts +2 -2
  41. package/src/observer.ts +1 -1
  42. package/src/router-swap.ts +121 -0
  43. package/src/run-lock.test.ts +73 -2
  44. package/src/run-lock.ts +110 -14
  45. package/src/runner.ts +99 -5
  46. package/src/simulator-ips.ts +21 -0
  47. package/src/socks-proxy.ts +2 -2
  48. package/src/types.ts +34 -4
  49. package/src/vantage.test.ts +1 -1
  50. package/src/zone-classifier.test.ts +11 -16
  51. package/src/zone-classifier.ts +16 -31
@@ -0,0 +1,121 @@
1
+ /**
2
+ * Swap the physical router the `fw-isp` simulator impersonates, on a RUNNING
3
+ * network — the harness half of the ISP-replaced-the-box scenario.
4
+ *
5
+ * `.axonRouter()` on the network builder cannot express this: it sets
6
+ * `ROUTER_VENDOR_PREFIX` in the generated compose file, which is read once when
7
+ * the simulator process starts. That is the right tool for "this network has an
8
+ * Axon in it"; it cannot say "this network had a GreenWave and now has an Axon",
9
+ * which is the event the whole module-pause change exists to survive.
10
+ *
11
+ * The swap models what actually happens to a fleet when an ISP replaces the
12
+ * hardware, and each part is load-bearing for what the migration test proves:
13
+ *
14
+ * - the TR-181 vendor prefix changes, so the OLD module's spelling is wrong;
15
+ * - the credentials change, which is why the fleet wedges rather than merely
16
+ * misbehaving — the old module cannot authenticate, so it can neither drive
17
+ * the new box nor clean up after itself;
18
+ * - every port forward is gone, because a new device arrives empty. That is
19
+ * the assertion with teeth: the forwards have to be recreated by unpausing
20
+ * the consumers that own them, and a simulator that kept its old table would
21
+ * let a completely broken unpause pass.
22
+ */
23
+
24
+ import { greenwaveRouterIp } from './types';
25
+ import type { NetworkHandle } from './types';
26
+
27
+ /** The devices the shared simulator can stand in for. */
28
+ export const ROUTER_DEVICES = {
29
+ /** GreenWave C4000XG — the default, driven by `modules/greenwave`. */
30
+ greenwave: { vendorPrefix: 'X_GWS_', username: 'admin', password: 'admin' },
31
+ /** Axon Networks Q1000K — driven by `modules/axon`. */
32
+ axon: { vendorPrefix: 'X_AXON_', username: 'axonadmin', password: 'axonsecret' },
33
+ } as const;
34
+
35
+ export type RouterDevice = keyof typeof ROUTER_DEVICES;
36
+
37
+ export interface RouterSwapResult {
38
+ /** Port forwards the outgoing device was holding, all of which are now gone. */
39
+ forwardsDiscarded: number;
40
+ device: { vendorPrefix: string; username: string };
41
+ }
42
+
43
+ /**
44
+ * Replace the impersonated device. Runs the request from INSIDE the simulator
45
+ * container, so the control endpoint never has to be reachable from anywhere
46
+ * else on the simulated internet.
47
+ */
48
+ export async function swapRouterDevice(
49
+ net: NetworkHandle,
50
+ device: RouterDevice,
51
+ ): Promise<RouterSwapResult> {
52
+ const spec = ROUTER_DEVICES[device];
53
+ const result = await net.exec(
54
+ 'fw-isp',
55
+ `curl -sk -X POST https://${greenwaveRouterIp()}/sim/swap-device ` +
56
+ `-d 'vendorPrefix=${spec.vendorPrefix}&username=${spec.username}&password=${spec.password}'`,
57
+ );
58
+ if (result.exitCode !== 0) {
59
+ throw new Error(`swapRouterDevice(${device}) failed: ${result.stderr || result.stdout}`);
60
+ }
61
+ return JSON.parse(result.stdout) as RouterSwapResult;
62
+ }
63
+
64
+ /**
65
+ * The port forwards the device is currently holding, read back OFF the device
66
+ * rather than out of celilo's own records.
67
+ *
68
+ * This exists because "verify the contract, not the verdict" is the whole point
69
+ * of the migration test: celilo reporting that it registered a forward is not
70
+ * evidence the router has one, and those are exactly the two things that came
71
+ * apart when the hardware changed underneath.
72
+ */
73
+ export async function listRouterForwards(net: NetworkHandle, device: RouterDevice) {
74
+ const spec = ROUTER_DEVICES[device];
75
+ const login = await net.exec(
76
+ 'fw-isp',
77
+ `curl -sk -X POST https://${greenwaveRouterIp()}/cgi/cgi_action ` +
78
+ `-d 'username=${spec.username}&password=${spec.password}' -c /tmp/swap-cookies`,
79
+ );
80
+ if (login.exitCode !== 0) {
81
+ throw new Error(`could not log in to the ${device} router: ${login.stderr || login.stdout}`);
82
+ }
83
+ const listed = await net.exec(
84
+ 'fw-isp',
85
+ `curl -sk 'https://${greenwaveRouterIp()}/cgi/cgi_get?Object=Device.NAT.PortMapping.' -b /tmp/swap-cookies`,
86
+ );
87
+ return listed.stdout;
88
+ }
89
+
90
+ /**
91
+ * The external ports currently forwarded on the device, as a sorted list.
92
+ *
93
+ * Assert on THIS rather than on a consumer's container IP. A forward's
94
+ * `InternalClient` is the firewall's natIp DNAT ingress (e.g. `10.226.1.253`),
95
+ * not the consumer's zone-side address — the delegation chain hops through the
96
+ * firewall — so matching a module's own IP looks correct and never matches.
97
+ */
98
+ export function forwardedPorts(listing: string): string[] {
99
+ const ports = [...listing.matchAll(/"ParamName":"ExternalPort","ParamValue":"(\d+)"/g)].map(
100
+ (m) => m[1],
101
+ );
102
+ return [...new Set(ports)].sort();
103
+ }
104
+
105
+ /** The DHCP pool's advertised DNS servers, read back off the device. */
106
+ export async function routerDhcpDnsServers(
107
+ net: NetworkHandle,
108
+ device: RouterDevice,
109
+ ): Promise<string> {
110
+ const spec = ROUTER_DEVICES[device];
111
+ await net.exec(
112
+ 'fw-isp',
113
+ `curl -sk -X POST https://${greenwaveRouterIp()}/cgi/cgi_action ` +
114
+ `-d 'username=${spec.username}&password=${spec.password}' -c /tmp/dhcp-cookies`,
115
+ );
116
+ const listed = await net.exec(
117
+ 'fw-isp',
118
+ `curl -sk 'https://${greenwaveRouterIp()}/cgi/cgi_get?Object=Device.DHCPv4.Server.Pool.' -b /tmp/dhcp-cookies`,
119
+ );
120
+ return listed.stdout;
121
+ }
@@ -5,8 +5,12 @@ import { join } from 'node:path';
5
5
  import {
6
6
  E2eBusyError,
7
7
  type LockHolder,
8
+ SUSPECT_HEARTBEAT_MS,
8
9
  acquireRunLock,
9
10
  clearLock,
11
+ formatBusy,
12
+ heartbeatAgeMs,
13
+ isSuspect,
10
14
  lockStatus,
11
15
  markKept,
12
16
  readHolder,
@@ -25,6 +29,7 @@ afterEach(() => {
25
29
  clearLock();
26
30
  rmSync(dir, { recursive: true, force: true });
27
31
  delete process.env.CELILO_E2E_LOCK_PATH;
32
+ delete process.env.CELILO_E2E_SESSION;
28
33
  });
29
34
 
30
35
  function writeRawHolder(h: Partial<LockHolder>): void {
@@ -71,7 +76,8 @@ test('a dead-pid holder on the same host is stale and reclaimed', () => {
71
76
  releaseRunLock();
72
77
  });
73
78
 
74
- test('--keep leaves a kept lock that survives release and blocks the next run', () => {
79
+ test('--keep leaves a kept lock that survives release and blocks other sessions', () => {
80
+ process.env.CELILO_E2E_SESSION = 'keeper (/keeper/worktree)';
75
81
  acquireRunLock({ test: 'kept-test', runId: 'r' });
76
82
  markKept();
77
83
  releaseRunLock();
@@ -80,7 +86,8 @@ test('--keep leaves a kept lock that survives release and blocks the next run',
80
86
  expect(h?.state).toBe('kept');
81
87
  expect(lockStatus().free).toBe(false); // kept is never stale
82
88
 
83
- // A plain run is refused...
89
+ // A plain run from ANOTHER session is refused...
90
+ process.env.CELILO_E2E_SESSION = 'other (/other/worktree)';
84
91
  expect(() => acquireRunLock({ test: 'next', runId: 'r2' })).toThrow(E2eBusyError);
85
92
  // ...but a --reuse run (allowKept) takes it over.
86
93
  acquireRunLock({ test: 'reuse', runId: 'r3', allowKept: true });
@@ -96,6 +103,70 @@ test('clearLock frees a kept lock (the `cele2e release` path)', () => {
96
103
  expect(lockStatus().free).toBe(true);
97
104
  });
98
105
 
106
+ test('a kept lock left by THIS session is auto-released by the next run', () => {
107
+ // The friction case: `run --keep` then `run` from the same session. Refusing
108
+ // here protected nobody — the only stack at risk was the caller's own — and
109
+ // the refusal was routinely misread as a finished run, because the previous
110
+ // run's results dir is still sitting there looking like a clean pass.
111
+ process.env.CELILO_E2E_SESSION = 'my-branch (/my/worktree)';
112
+ acquireRunLock({ test: 'kept-test', runId: 'r' });
113
+ markKept();
114
+ releaseRunLock();
115
+ expect(readHolder()?.state).toBe('kept');
116
+
117
+ const outcome = acquireRunLock({ test: 'next', runId: 'r2' });
118
+ expect(outcome.autoReleasedOwnKept?.test).toBe('kept-test');
119
+ expect(readHolder()?.pid).toBe(process.pid);
120
+ expect(readHolder()?.state).toBe('running');
121
+ releaseRunLock();
122
+ });
123
+
124
+ test('a kept lock from a DIFFERENT session is still refused', () => {
125
+ process.env.CELILO_E2E_SESSION = 'my-branch (/my/worktree)';
126
+ // A live pid so staleness can't be what frees it — the point is the session.
127
+ writeRawHolder({ state: 'kept', session: 'someone-else (/their/worktree)', pid: process.pid });
128
+ expect(() => acquireRunLock({ test: 'mine', runId: 'r' })).toThrow(E2eBusyError);
129
+ });
130
+
131
+ test('a live pid with a silent heartbeat is SUSPECT but never auto-reclaimed', () => {
132
+ // build-infra wedged at step [16/27]: the process is alive (so PID-liveness
133
+ // says "healthy") while its heartbeat has not ticked for half an hour. This
134
+ // is the one hang the existing staleness check structurally cannot see.
135
+ writeRawHolder({ pid: process.pid, beatAt: Date.now() - 1_913_000 });
136
+ const h = readHolder() as LockHolder;
137
+
138
+ expect(isSuspect(h)).toBe(true);
139
+ expect(heartbeatAgeMs(h)).toBeGreaterThan(SUSPECT_HEARTBEAT_MS);
140
+ expect(formatBusy(h)).toContain('SUSPECT');
141
+
142
+ const status = lockStatus();
143
+ expect(status.suspect).toBe(true);
144
+ // Suspect surfaces the hang; it does NOT kill someone's build for them.
145
+ expect(status.free).toBe(false);
146
+ expect(() => acquireRunLock({ test: 'other', runId: 'r' })).toThrow(E2eBusyError);
147
+ });
148
+
149
+ test('a fresh heartbeat is not suspect', () => {
150
+ writeRawHolder({ pid: process.pid, beatAt: Date.now() - 5_000 });
151
+ const h = readHolder() as LockHolder;
152
+ expect(isSuspect(h)).toBe(false);
153
+ expect(lockStatus().suspect).toBe(false);
154
+ });
155
+
156
+ test('a healthy build-infra that blocks the event loop is NOT suspect', () => {
157
+ // spawnSync('docker', ['build', …]) blocks the event loop, so the heartbeat
158
+ // stops for the whole of each image build. A live holder was observed 67s
159
+ // stale while making normal progress; the threshold has to clear that or the
160
+ // flag fires on every large image and stops meaning anything.
161
+ writeRawHolder({ pid: process.pid, beatAt: Date.now() - 120_000 });
162
+ expect(isSuspect(readHolder() as LockHolder)).toBe(false);
163
+ });
164
+
165
+ test('a kept holder is never suspect — its beatAt is frozen on purpose', () => {
166
+ writeRawHolder({ state: 'kept', pid: process.pid, beatAt: Date.now() - 3_600_000 });
167
+ expect(isSuspect(readHolder() as LockHolder)).toBe(false);
168
+ });
169
+
99
170
  test('a corrupt lock file is reclaimed, not fatal', () => {
100
171
  writeFileSync(process.env.CELILO_E2E_LOCK_PATH as string, 'not json{');
101
172
  expect(readHolder()).toBeNull();
package/src/run-lock.ts CHANGED
@@ -18,9 +18,14 @@
18
18
  * - Staleness: on the SAME host, PID-liveness is authoritative (process.kill
19
19
  * (pid, 0)); a dead holder PID → reclaim. Cross-host we can't check the PID,
20
20
  * so fall back to a heartbeat TTL (beatAt older than STALE_TTL_MS).
21
- * - A `kept` lock (left by `--keep` / `up`) is NEVER auto-reclaimed — it
22
- * guards a stack that outlives the process and is cleared only by
23
- * `cele2e release` / `cele2e down`. This enforces the honor-system rule.
21
+ * - A `kept` lock (left by `--keep` / `up`) is never reclaimed by ANOTHER
22
+ * actor — it guards a stack that outlives the process. Its OWN session
23
+ * reclaims it automatically (see isSameSession); everyone else clears it
24
+ * deliberately via `cele2e release` / `cele2e down`.
25
+ * - SUSPECT holders: a live PID whose heartbeat has gone quiet is the one
26
+ * hang the PID check cannot see (a build wedged mid-step keeps its process
27
+ * alive and sleeping). isSuspect() names it so `status`/`doctor` can flag it
28
+ * instead of leaving 20 silent minutes to be found by hand.
24
29
  */
25
30
 
26
31
  import { execFileSync } from 'node:child_process';
@@ -31,6 +36,23 @@ import { basename, dirname, join } from 'node:path';
31
36
  const STALE_TTL_MS = 90_000;
32
37
  const HEARTBEAT_MS = 30_000;
33
38
 
39
+ /**
40
+ * How quiet a RUNNING holder's heartbeat may go before it is suspect.
41
+ *
42
+ * Calibrated against what a HEALTHY holder actually does. The heartbeat is a
43
+ * setInterval, and build-infra shells out with spawnSync — which blocks the
44
+ * event loop for the entire duration of each `docker build`. So a perfectly
45
+ * healthy build stops beating for as long as its slowest image takes: a live
46
+ * holder was observed 67s stale while making normal progress, and the largest
47
+ * images here are ~2GB. Anything near the 30s tick interval would flag those.
48
+ *
49
+ * Ten minutes sits above any single legitimate image build (the per-image
50
+ * watchdog caps one at 15) and far below the ~20 and ~32 minute hangs that
51
+ * went unnoticed. Deliberately NOT a reclaim threshold: the PID is alive, and
52
+ * killing someone's wedged build out from under them is the operator's call.
53
+ */
54
+ export const SUSPECT_HEARTBEAT_MS = 600_000;
55
+
34
56
  /**
35
57
  * Machine-global lock path — NOT under any repo/worktree, because the Docker
36
58
  * infra is global regardless of which checkout started it. Overridable via
@@ -75,14 +97,22 @@ export function formatBusy(h: LockHolder): string {
75
97
  h.state === 'kept'
76
98
  ? '(run `cele2e release` to free it)'
77
99
  : `${h.test}, started ${age} ago (pid ${h.pid})`;
78
- return `e2e busy: ${h.session} ${verb} ${what}`;
100
+ const suspect = isSuspect(h)
101
+ ? ` — SUSPECT: no heartbeat for ${formatAge(heartbeatAgeMs(h))} (process alive but not progressing)`
102
+ : '';
103
+ return `e2e busy: ${h.session} ${verb} ${what}${suspect}`;
79
104
  }
80
105
 
81
106
  function ageString(since: number): string {
82
- const s = Math.max(0, Math.floor((Date.now() - since) / 1000));
107
+ return formatAge(Date.now() - since);
108
+ }
109
+
110
+ /** Human duration, e.g. "45s" / "32m" / "2h05m". */
111
+ export function formatAge(ms: number): string {
112
+ const s = Math.max(0, Math.floor(ms / 1000));
83
113
  if (s < 60) return `${s}s`;
84
114
  if (s < 3600) return `${Math.floor(s / 60)}m`;
85
- return `${Math.floor(s / 3600)}h${Math.floor((s % 3600) / 60)}m`;
115
+ return `${Math.floor(s / 3600)}h${String(Math.floor((s % 3600) / 60)).padStart(2, '0')}m`;
86
116
  }
87
117
 
88
118
  /** Branch + worktree path so a holder maps back to a specific session/thread. */
@@ -107,6 +137,28 @@ function deriveSession(): string {
107
137
  return branch || base;
108
138
  }
109
139
 
140
+ /**
141
+ * Is this holder the caller's own session — same host, same branch+worktree?
142
+ * This is what separates "my own kept stack is in my way" (friction, auto-clear)
143
+ * from "someone else is mid-run" (contention, refuse).
144
+ */
145
+ export function isSameSession(h: LockHolder): boolean {
146
+ return h.hostname === hostname() && h.session === deriveSession();
147
+ }
148
+
149
+ /** Milliseconds since the holder last refreshed its heartbeat. */
150
+ export function heartbeatAgeMs(h: LockHolder): number {
151
+ return Math.max(0, Date.now() - h.beatAt);
152
+ }
153
+
154
+ /**
155
+ * A live process whose heartbeat has gone quiet — the hang PID-liveness misses.
156
+ * `kept` holders are excluded: their beatAt is frozen on purpose at release.
157
+ */
158
+ export function isSuspect(h: LockHolder): boolean {
159
+ return h.state === 'running' && heartbeatAgeMs(h) > SUSPECT_HEARTBEAT_MS;
160
+ }
161
+
110
162
  function pidAlive(pid: number): boolean {
111
163
  try {
112
164
  process.kill(pid, 0);
@@ -140,15 +192,32 @@ function writeHolder(fd: number, h: LockHolder): void {
140
192
  writeFileSync(fd, JSON.stringify(h, null, 2));
141
193
  }
142
194
 
195
+ /** What the acquire had to clear on the way in, so callers can say so out loud. */
196
+ export interface AcquireOutcome {
197
+ /** A `kept` lock left by this same session that we auto-released, else null. */
198
+ autoReleasedOwnKept: LockHolder | null;
199
+ }
200
+
143
201
  /**
144
202
  * Acquire the run lock for this process. Throws E2eBusyError if another live
145
203
  * (non-stale) holder owns it. Registers an exit handler so the lock is released
146
204
  * on any process.exit() path (normal, exception, or a SIGINT handler that
147
205
  * exits). A signal-killed process with no exit handler leaks the lock, but the
148
206
  * next run reclaims it via PID-liveness — that's exactly what staleness is for.
207
+ *
208
+ * A `kept` lock left by THIS session (same host, same branch+worktree) is
209
+ * auto-released and reported in the outcome. Refusing there protected nobody:
210
+ * the only stack at risk was the caller's own, and the refusal was routinely
211
+ * misread as a finished run — the previous run's results dir is still sitting
212
+ * there looking like a clean pass. Another session's kept lock still refuses.
149
213
  */
150
- export function acquireRunLock(opts: { test: string; runId: string; allowKept?: boolean }): void {
214
+ export function acquireRunLock(opts: {
215
+ test: string;
216
+ runId: string;
217
+ allowKept?: boolean;
218
+ }): AcquireOutcome {
151
219
  mkdirSync(dirname(lockPath()), { recursive: true });
220
+ let autoReleasedOwnKept: LockHolder | null = null;
152
221
 
153
222
  const holder: LockHolder = {
154
223
  pid: process.pid,
@@ -168,9 +237,17 @@ export function acquireRunLock(opts: { test: string; runId: string; allowKept?:
168
237
  } catch (err) {
169
238
  if ((err as NodeJS.ErrnoException).code !== 'EEXIST') throw err;
170
239
  const existing = readHolder();
171
- // Reclaim when: corrupt/unreadable, the holder is stale, OR this is a
172
- // `--reuse` run taking over a `kept` stack it's deliberately reusing.
173
- if (!existing || isStale(existing) || (opts.allowKept && existing.state === 'kept')) {
240
+ // Reclaim when: corrupt/unreadable, the holder is stale, this is a
241
+ // `--reuse` run taking over a `kept` stack it's deliberately reusing, OR
242
+ // the `kept` stack belongs to this very session (see the doc comment).
243
+ const ownKept = !!existing && existing.state === 'kept' && isSameSession(existing);
244
+ if (
245
+ !existing ||
246
+ isStale(existing) ||
247
+ (opts.allowKept && existing.state === 'kept') ||
248
+ ownKept
249
+ ) {
250
+ if (ownKept && existing) autoReleasedOwnKept = existing;
174
251
  try {
175
252
  unlinkSync(lockPath());
176
253
  } catch {}
@@ -195,7 +272,7 @@ export function acquireRunLock(opts: { test: string; runId: string; allowKept?:
195
272
 
196
273
  held = { keepOnRelease: false, heartbeat, released: false };
197
274
  process.on('exit', releaseRunLock);
198
- return;
275
+ return { autoReleasedOwnKept };
199
276
  }
200
277
  // Two reclaim attempts both lost the race → someone else won fair and square.
201
278
  const existing = readHolder();
@@ -258,9 +335,28 @@ export function clearLock(): boolean {
258
335
  }
259
336
  }
260
337
 
338
+ export interface LockStatus {
339
+ free: boolean;
340
+ holder: LockHolder | null;
341
+ /** Milliseconds since the holder's last heartbeat; null when free. */
342
+ heartbeatAgeMs: number | null;
343
+ /** Live PID, quiet heartbeat — the hang the PID check cannot see. */
344
+ suspect: boolean;
345
+ /** The holder is this session's own kept stack, which the next run reclaims. */
346
+ ownKept: boolean;
347
+ }
348
+
261
349
  /** Current lock state for `cele2e status` (and any poller). */
262
- export function lockStatus(): { free: boolean; holder: LockHolder | null } {
350
+ export function lockStatus(): LockStatus {
263
351
  const h = readHolder();
264
- if (!h || isStale(h)) return { free: true, holder: null };
265
- return { free: false, holder: h };
352
+ if (!h || isStale(h)) {
353
+ return { free: true, holder: null, heartbeatAgeMs: null, suspect: false, ownKept: false };
354
+ }
355
+ return {
356
+ free: false,
357
+ holder: h,
358
+ heartbeatAgeMs: heartbeatAgeMs(h),
359
+ suspect: isSuspect(h),
360
+ ownKept: h.state === 'kept' && isSameSession(h),
361
+ };
266
362
  }
package/src/runner.ts CHANGED
@@ -29,7 +29,9 @@ import {
29
29
  emitTestCompleted,
30
30
  emitTestStarted,
31
31
  } from './bus-events';
32
- import { extractFailureMessage, stripAnsi } from './extract-failure';
32
+ import { diagnose, formatReport } from './doctor';
33
+ import { type StageTally, extractFailureMessage, stripAnsi, tallyStages } from './extract-failure';
34
+ import { writeLastRun } from './last-run';
33
35
  import { parseLine } from './parse-line';
34
36
  import { E2eBusyError, acquireRunLock, markKept } from './run-lock';
35
37
  import { SIMULATOR_IPS } from './simulator-ips';
@@ -200,6 +202,12 @@ interface TestResult {
200
202
  rawTail?: string;
201
203
  hadDebugPause?: boolean;
202
204
  projectName?: string;
205
+ /**
206
+ * Real stage failures vs stages a failing earlier stage blocked. Without the
207
+ * split, one bad fixture line in stage 1 of a 10-stage suite reports as "9
208
+ * failed" — nine counts of a defect that does not exist.
209
+ */
210
+ stages?: StageTally;
203
211
  }
204
212
 
205
213
  async function runTest(
@@ -399,12 +407,24 @@ async function runTest(
399
407
  projectName: capturedProjectName,
400
408
  error: extractFailureMessage(lines, code ?? 1),
401
409
  rawTail,
410
+ stages: tallyStages(lines),
402
411
  });
403
412
  }
404
413
  });
405
414
  });
406
415
  }
407
416
 
417
+ /**
418
+ * " — 1 stage failed, 8 skipped (blocked by an earlier stage)". Rendered only
419
+ * when a cascade actually happened, so an ordinary single-stage failure stays
420
+ * as terse as it was.
421
+ */
422
+ function stageSuffix(stages: StageTally | undefined): string {
423
+ if (!stages || stages.skipped === 0) return '';
424
+ const failed = `${stages.failed} stage${stages.failed === 1 ? '' : 's'} failed`;
425
+ return ` ${dim}— ${failed}, ${stages.skipped} skipped (blocked by an earlier stage)${reset}`;
426
+ }
427
+
408
428
  // ─── Shared Infrastructure ───────────────────────────────────────────
409
429
 
410
430
  function dockerComposeShared(cmd: string): string {
@@ -475,11 +495,20 @@ async function main() {
475
495
  // docker mutation so a competing run can't wipe our shared infra mid-setup.
476
496
  const lockLabel = patterns.length ? patterns.join(',') : moduleDirs.length ? 'modules' : 'all';
477
497
  try {
478
- acquireRunLock({
498
+ const outcome = acquireRunLock({
479
499
  test: lockLabel,
480
500
  runId: process.env.CELE2E_RUN_ID ?? ambientRunId,
481
501
  allowKept: flagReuse,
482
502
  });
503
+ if (outcome.autoReleasedOwnKept) {
504
+ const own = outcome.autoReleasedOwnKept;
505
+ console.log(
506
+ `${dim}Auto-released your own kept stack (${own.test}, held since ${own.startedAt}) and continuing.${reset}`,
507
+ );
508
+ console.log(
509
+ `${dim}Use \`cele2e run --reuse\` instead if you meant to run against it.${reset}`,
510
+ );
511
+ }
483
512
  // A --reuse run keeps using a kept stack that stays up afterwards — hold the
484
513
  // lock in kept state so it isn't freed out from under the reused network.
485
514
  if (flagReuse) markKept();
@@ -494,6 +523,25 @@ async function main() {
494
523
  throw err;
495
524
  }
496
525
 
526
+ // Preflight the environment BEFORE touching Docker. Every check here failed
527
+ // silently once and surfaced later somewhere unrelated — a missing bake as an
528
+ // SSH error against a firewall IP, a pruned base image as a TLS timeout 16
529
+ // images into a build. Refusing now costs seconds; not refusing cost a full
530
+ // run plus a debugging cycle each time. Lock state is excluded: we already
531
+ // hold the lock, so checkRunLock would report ourselves as contention.
532
+ const health = diagnose({ pkgDir: PKG_DIR, skipLock: true });
533
+ if (!health.ok) {
534
+ console.error(
535
+ `\n${red}✗ cele2e preflight failed — not starting a run that cannot succeed.${reset}`,
536
+ );
537
+ for (const line of formatReport(health)) console.error(line);
538
+ console.error(`\n${dim}Full environment report: cele2e doctor${reset}\n`);
539
+ process.exit(1);
540
+ }
541
+ for (const c of health.checks) {
542
+ if (c.status === 'warn') console.log(`${dim}! ${c.name}: ${c.detail}${reset}`);
543
+ }
544
+
497
545
  let testFiles: string[] = [];
498
546
 
499
547
  if (moduleDirs.length > 0) {
@@ -656,6 +704,22 @@ async function main() {
656
704
  if (!process.env.CELE2E_RUN_ID) {
657
705
  console.log(`${dim}runId: ${runId}${reset}`);
658
706
  }
707
+ // Print the results dir HERE, not only at the end: a run that dies mid-way
708
+ // still wrote logs there, and `ls -t results | head -1` is not a safe way to
709
+ // find them — after a refused start it returns the PREVIOUS run's numbers,
710
+ // which look exactly like a clean pass. `cele2e last` reads the same record.
711
+ console.log(`${dim}results: ${resultsDir}${reset}`);
712
+ writeLastRun({
713
+ runId,
714
+ resultsDir,
715
+ startedAt: new Date(suiteStart).toISOString(),
716
+ status: 'running',
717
+ total: testFiles.length,
718
+ passed: 0,
719
+ failed: 0,
720
+ skipped: 0,
721
+ durationMs: 0,
722
+ });
659
723
  emitRunStarted({
660
724
  runId,
661
725
  scenario: testFiles.length > 1 ? 'multi-test' : 'single-test',
@@ -784,7 +848,7 @@ async function main() {
784
848
  console.log(` ${red} └─ ${result.error}${reset}`);
785
849
  failed++;
786
850
  } else {
787
- console.log(` ${red}✗ failed (${result.duration}s)${reset}`);
851
+ console.log(` ${red}✗ failed (${result.duration}s)${reset}${stageSuffix(result.stages)}`);
788
852
  if (result.error) {
789
853
  console.log(` ${red} └─ ${result.error}${reset}`);
790
854
  }
@@ -817,18 +881,28 @@ async function main() {
817
881
 
818
882
  const suiteDuration = Math.floor((Date.now() - suiteStart) / 1000);
819
883
 
884
+ // Stages a failing earlier stage blocked are reported separately from real
885
+ // failures. Counting them together turns one bad fixture line into "9 failed"
886
+ // and buries the single defect that actually exists.
887
+ const stageSkipped = results.reduce((sum, r) => sum + (r.stages?.skipped ?? 0), 0);
888
+
820
889
  console.log();
821
890
  console.log(`${bold}╔══════════════════════════════════════════════════════════════${reset}`);
822
891
  console.log(
823
892
  `${bold}║ Results: ${green}${passed} passed${reset}${bold}, ${red}${failed} failed${reset}${bold} — ${formatDuration(suiteDuration)} total${reset}`,
824
893
  );
894
+ if (stageSkipped > 0) {
895
+ console.log(
896
+ `${bold}║${reset} ${dim}${stageSkipped} stage(s) skipped — blocked by an earlier stage, not separate defects${reset}`,
897
+ );
898
+ }
825
899
  console.log(`${bold}╠══════════════════════════════════════════════════════════════${reset}`);
826
900
 
827
901
  for (const r of results) {
828
902
  const icon = r.status === 'pass' ? '✓' : '✗';
829
903
  const color = r.status === 'pass' ? green : red;
830
904
  console.log(
831
- `${bold}║${reset} ${color}${icon}${reset} ${r.name.padEnd(28)} ${dim}${(`${r.duration}s`).padStart(6)}${reset}`,
905
+ `${bold}║${reset} ${color}${icon}${reset} ${r.name.padEnd(28)} ${dim}${(`${r.duration}s`).padStart(6)}${reset}${stageSuffix(r.stages)}`,
832
906
  );
833
907
  if (r.error) {
834
908
  console.log(`${bold}║${reset} ${red}└─ ${r.error}${reset}`);
@@ -838,16 +912,36 @@ async function main() {
838
912
  console.log(`${bold}╚══════════════════════════════════════════════════════════════${reset}`);
839
913
  console.log();
840
914
  console.log(`${dim}Full logs: ${resultsDir}${reset}`);
915
+ console.log(`${dim}Machine-readable: cele2e last --json${reset}`);
841
916
 
842
917
  writeFileSync(
843
918
  join(resultsDir, 'summary.json'),
844
919
  JSON.stringify(
845
- { timestamp, total: testFiles.length, passed, failed, duration: suiteDuration },
920
+ {
921
+ timestamp,
922
+ total: testFiles.length,
923
+ passed,
924
+ failed,
925
+ stageSkipped,
926
+ duration: suiteDuration,
927
+ },
846
928
  null,
847
929
  2,
848
930
  ),
849
931
  );
850
932
 
933
+ writeLastRun({
934
+ runId,
935
+ resultsDir,
936
+ startedAt: new Date(suiteStart).toISOString(),
937
+ status: failed > 0 ? 'failed' : 'completed',
938
+ total: testFiles.length,
939
+ passed,
940
+ failed,
941
+ skipped: stageSkipped,
942
+ durationMs: suiteDuration * 1000,
943
+ });
944
+
851
945
  const gitignore = join(testDir, '.gitignore');
852
946
  const content = existsSync(gitignore) ? readFileSync(gitignore, 'utf-8') : '';
853
947
  const adds: string[] = [];
@@ -44,6 +44,27 @@ export const SIMULATOR_IPS = {
44
44
  CPANEL_HOST: '100.64.0.63',
45
45
  /** signal-cli release host — serves the tarball the signal module downloads at deploy time. */
46
46
  SIGNAL_RELEASE: '100.64.0.62',
47
+ /**
48
+ * OFF-FLEET recursive resolver — the rig's stand-in for 1.1.1.1, and a peer
49
+ * of comcast-resolver rather than a replacement for it.
50
+ *
51
+ * The distinction is the whole point of celilo's `public_dns` check: the
52
+ * fleet's own resolver (its ISP's, or its internal split-horizon one)
53
+ * answers with whatever is correct for a client INSIDE, which is not
54
+ * evidence about what the internet sees. celilo REFUSES to use a resolver it
55
+ * is itself configured to use, so verifying public reachability requires a
56
+ * second, independent public resolver — and until this existed the topology
57
+ * had exactly one.
58
+ */
59
+ PUBLIC_RESOLVER: '100.64.0.64',
60
+ /**
61
+ * IP echo service — the rig's stand-in for api.ipify.org. Reports the source
62
+ * address a request appears to come from, which for the customer fleet is
63
+ * the firewall's external address after SNAT. The `public_dns` check's
64
+ * expectation comes from here rather than from the registrar's own response,
65
+ * which is self-agreement (and, for a Namecheap `www` update, false).
66
+ */
67
+ IP_ECHO: '100.64.0.65',
47
68
  /** Pebble ACME server (replaces production Let's Encrypt). */
48
69
  PEBBLE: '100.64.0.100',
49
70
  } as const;
@@ -44,7 +44,7 @@ export interface SocksProxyHandle {
44
44
  * Why custom: an off-the-shelf SOCKS5 image (e.g. serjs/go-socks5-proxy)
45
45
  * starts fine on the test network but can't actually route to anything.
46
46
  * Docker's IPAM-default gateway points at an IP with no listener (e.g.
47
- * 100.100.0.250 on isp-external), and docker rewrites resolv.conf to
47
+ * 203.0.113.250 on isp-external), and docker rewrites resolv.conf to
48
48
  * its embedded resolver, which can't cleanly forward to the simulated
49
49
  * network's DNS. The custom image's entrypoint replaces both.
50
50
  */
@@ -61,7 +61,7 @@ interface VantageConfig {
61
61
  const VANTAGE_CONFIGS: Record<SocksProxyVantage, VantageConfig> = {
62
62
  'isp-external': {
63
63
  network: 'isp-external',
64
- ip: '100.100.0.150',
64
+ ip: '203.0.113.150',
65
65
  },
66
66
  internal: {
67
67
  network: 'internal',