@celilo/e2e 0.12.0 → 0.13.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (46) hide show
  1. package/README.md +5 -5
  2. package/bin/e2e-status +1 -1
  3. package/bin/e2e-up +2 -2
  4. package/config/dns/tangohost.com.zone +1 -1
  5. package/config/dns/templates/example.net.zone +3 -3
  6. package/config/dns/templates/iamtheinternet.org.zone +2 -2
  7. package/config/routing/fw-ext-routes.sh +8 -8
  8. package/config/routing/fw-isp-routes.sh +2 -2
  9. package/config/routing/fw-main-routes.sh +5 -5
  10. package/config/routing/management-routes.sh +1 -1
  11. package/config/routing/observer-setup.sh +1 -1
  12. package/config/routing/public-sim-entrypoint.sh +1 -1
  13. package/config/routing/resolver-internal-routes.sh +2 -2
  14. package/config/routing/resolver-routes.sh +1 -1
  15. package/config/routing/target-routes.sh +1 -1
  16. package/config/routing/target-setup.sh +1 -1
  17. package/config/socks/startup.sh +2 -2
  18. package/docker/Dockerfile.firewall +12 -1
  19. package/package.json +5 -4
  20. package/simulators/greenwave/server.ts +35 -0
  21. package/simulators/greenwave/state.ts +64 -11
  22. package/simulators/ip-echo/server.ts +1 -1
  23. package/src/address-plan.test.ts +4 -4
  24. package/src/cli/build.ts +30 -0
  25. package/src/cli/command-registry.ts +19 -0
  26. package/src/cli/completion.ts +16 -1
  27. package/src/cli/index.ts +85 -4
  28. package/src/container-manager.ts +191 -14
  29. package/src/docker-compose-generator.ts +55 -26
  30. package/src/doctor.test.ts +279 -0
  31. package/src/doctor.ts +421 -0
  32. package/src/extract-failure.ts +41 -0
  33. package/src/index.ts +12 -0
  34. package/src/last-run.test.ts +62 -0
  35. package/src/last-run.ts +54 -0
  36. package/src/observer.test.ts +2 -2
  37. package/src/observer.ts +1 -1
  38. package/src/router-swap.ts +121 -0
  39. package/src/run-lock.test.ts +73 -2
  40. package/src/run-lock.ts +110 -14
  41. package/src/runner.ts +99 -5
  42. package/src/socks-proxy.ts +2 -2
  43. package/src/types.ts +48 -8
  44. package/src/vantage.test.ts +1 -1
  45. package/src/zone-classifier.test.ts +11 -16
  46. package/src/zone-classifier.ts +16 -31
@@ -12,6 +12,47 @@ export function stripAnsi(s: string): string {
12
12
  return s.replace(ANSI_RE, '');
13
13
  }
14
14
 
15
+ export interface StageTally {
16
+ /** Stages that failed on their own merits. */
17
+ failed: number;
18
+ /** Stages that never ran because an earlier stage had already failed. */
19
+ skipped: number;
20
+ }
21
+
22
+ /**
23
+ * Split a staged suite's failures into real ones and cascade-skips.
24
+ *
25
+ * Staged e2e suites guard every stage with `requireStage`, which throws
26
+ * `Skipped: <stage> — <earlier error>` once anything upstream has failed. bun
27
+ * counts each of those as a failure, so ONE bad fixture line in stage 1 reports
28
+ * as "9 failed" in a 10-stage suite — nine counts of a defect that does not
29
+ * exist, and a summary that buries the one that does.
30
+ *
31
+ * The `Skipped:` marker is the semantic signal, so that is what we key on; the
32
+ * sub-millisecond durations those stages show are a symptom, not the contract.
33
+ */
34
+ export function tallyStages(lines: string[]): StageTally {
35
+ const plain = lines.map(stripAnsi);
36
+ const isFailure = (l: string): boolean => /^\(fail\)\s+\S/.test(l);
37
+ let failed = 0;
38
+ let skipped = 0;
39
+ for (let i = 0; i < plain.length; i++) {
40
+ if (!isFailure(plain[i])) continue;
41
+ // bun prints the reason on the following lines, indented under the failure.
42
+ // The window MUST stop at the next failure: cascade-skips come in runs, and
43
+ // a window that reads past the boundary attributes the next stage's
44
+ // "Skipped:" to this one — which misreads the single real defect at the top
45
+ // of a cascade as just another skip, losing the only line worth reading.
46
+ const reason: string[] = [];
47
+ for (let j = i + 1; j < plain.length && j <= i + 4 && !isFailure(plain[j]); j++) {
48
+ reason.push(plain[j]);
49
+ }
50
+ if (/(?:^|\W)Skipped:\s/.test(reason.join('\n'))) skipped++;
51
+ else failed++;
52
+ }
53
+ return { failed, skipped };
54
+ }
55
+
15
56
  export function extractFailureMessage(lines: string[], exitCode: number): string {
16
57
  const plain = lines.map(stripAnsi);
17
58
 
package/src/index.ts CHANGED
@@ -102,3 +102,15 @@ export {
102
102
  SHARED_PROJECT_NAME,
103
103
  SHARED_NETWORKS,
104
104
  } from './docker-compose-generator';
105
+
106
+ // Runtime router-device swap — the harness half of "the ISP replaced the box"
107
+ // (openspec/changes/module-pause-lifecycle task 7.7).
108
+ export {
109
+ ROUTER_DEVICES,
110
+ type RouterDevice,
111
+ type RouterSwapResult,
112
+ forwardedPorts,
113
+ listRouterForwards,
114
+ routerDhcpDnsServers,
115
+ swapRouterDevice,
116
+ } from './router-swap';
@@ -0,0 +1,62 @@
1
+ import { afterEach, beforeEach, expect, test } from 'bun:test';
2
+ import { mkdtempSync, rmSync } from 'node:fs';
3
+ import { tmpdir } from 'node:os';
4
+ import { join } from 'node:path';
5
+ import { type LastRun, readLastRun, writeLastRun } from './last-run';
6
+
7
+ let dir: string;
8
+
9
+ beforeEach(() => {
10
+ dir = mkdtempSync(join(tmpdir(), 'e2e-last-'));
11
+ process.env.CELILO_E2E_LAST_RUN_PATH = join(dir, 'last-run.json');
12
+ });
13
+
14
+ afterEach(() => {
15
+ rmSync(dir, { recursive: true, force: true });
16
+ delete process.env.CELILO_E2E_LAST_RUN_PATH;
17
+ });
18
+
19
+ const RUN: LastRun = {
20
+ runId: 'abc-123',
21
+ resultsDir: '/repo/e2e/results/2026-08-12T10-00-00',
22
+ startedAt: '2026-08-12T17:00:00.000Z',
23
+ status: 'completed',
24
+ total: 3,
25
+ passed: 3,
26
+ failed: 0,
27
+ skipped: 0,
28
+ durationMs: 600_000,
29
+ };
30
+
31
+ test('with no run recorded, readLastRun is null rather than throwing', () => {
32
+ expect(readLastRun()).toBeNull();
33
+ });
34
+
35
+ test('a recorded run round-trips every field `cele2e last --json` promises', () => {
36
+ writeLastRun(RUN);
37
+ expect(readLastRun()).toEqual(RUN);
38
+ });
39
+
40
+ test('the record is overwritten, so `last` never returns a stale earlier run', () => {
41
+ // The whole point: inferring "my results dir" from `ls -t results | head -1`
42
+ // hands back the PREVIOUS run's numbers when a run refuses to start, and a
43
+ // suite that never executed then reads as a clean pass.
44
+ writeLastRun(RUN);
45
+ writeLastRun({ ...RUN, runId: 'def-456', resultsDir: '/repo/e2e/results/later', failed: 2 });
46
+ const last = readLastRun();
47
+ expect(last?.runId).toBe('def-456');
48
+ expect(last?.resultsDir).toBe('/repo/e2e/results/later');
49
+ expect(last?.failed).toBe(2);
50
+ });
51
+
52
+ test('a run still in flight is recorded as running, so its logs are findable', () => {
53
+ writeLastRun({ ...RUN, status: 'running', passed: 0, total: 3 });
54
+ expect(readLastRun()?.status).toBe('running');
55
+ expect(readLastRun()?.resultsDir).toBe(RUN.resultsDir);
56
+ });
57
+
58
+ test('a corrupt record is null, not fatal', () => {
59
+ writeLastRun(RUN);
60
+ Bun.write(process.env.CELILO_E2E_LAST_RUN_PATH as string, 'not json{');
61
+ expect(readLastRun()).toBeNull();
62
+ });
@@ -0,0 +1,54 @@
1
+ /**
2
+ * A machine-global pointer to the most recent cele2e run.
3
+ *
4
+ * Without one, "how did my run go?" is answered by `ls -t e2e/results | head -1`
5
+ * — which returns the newest directory, not YOUR run. When a run refuses to
6
+ * start (a held lock, a failed preflight) that inference silently hands back the
7
+ * PREVIOUS run's numbers, and a suite that never executed reads as a clean pass.
8
+ *
9
+ * Written at run start (so a run that dies mid-way still has a findable results
10
+ * dir) and again at the end with the counts. Lives beside the run-lock rather
11
+ * than in a worktree, because which checkout started the run is not something
12
+ * the reader knows.
13
+ */
14
+
15
+ import { mkdirSync, readFileSync, writeFileSync } from 'node:fs';
16
+ import { homedir } from 'node:os';
17
+ import { dirname, join } from 'node:path';
18
+
19
+ export interface LastRun {
20
+ runId: string;
21
+ resultsDir: string;
22
+ startedAt: string;
23
+ /** 'running' until the suite finishes; then 'completed' or 'failed'. */
24
+ status: 'running' | 'completed' | 'failed';
25
+ total: number;
26
+ passed: number;
27
+ failed: number;
28
+ /** Stages a failing earlier stage blocked. Distinct from real failures. */
29
+ skipped: number;
30
+ durationMs: number;
31
+ }
32
+
33
+ export function lastRunPath(): string {
34
+ return (
35
+ process.env.CELILO_E2E_LAST_RUN_PATH || join(homedir(), '.cache', 'celilo-e2e', 'last-run.json')
36
+ );
37
+ }
38
+
39
+ export function writeLastRun(run: LastRun): void {
40
+ try {
41
+ mkdirSync(dirname(lastRunPath()), { recursive: true });
42
+ writeFileSync(lastRunPath(), `${JSON.stringify(run, null, 2)}\n`);
43
+ } catch {
44
+ // ponytail: a bookkeeping pointer must never be the thing that fails a run.
45
+ }
46
+ }
47
+
48
+ export function readLastRun(): LastRun | null {
49
+ try {
50
+ return JSON.parse(readFileSync(lastRunPath(), 'utf-8')) as LastRun;
51
+ } catch {
52
+ return null;
53
+ }
54
+ }
@@ -44,14 +44,14 @@ describe('placement faithfulness (the load-bearing invariants)', () => {
44
44
  });
45
45
 
46
46
  test('publicInternet uses only the public resolver (never the internal split-horizon view)', () => {
47
- expect(OBSERVER_PLACEMENTS.publicInternet.resolvers).toEqual(['100.100.0.1']);
47
+ expect(OBSERVER_PLACEMENTS.publicInternet.resolvers).toEqual(['203.0.113.1']);
48
48
  });
49
49
 
50
50
  test('observerEnv serializes the routing profile for the setup script', () => {
51
51
  const env = observerEnv(OBSERVER_PLACEMENTS.internalDevice);
52
52
  expect(env.OBSERVER_INTERZONE).toBe('0');
53
53
  expect(env.OBSERVER_GATEWAY).toBe(OBSERVER_PLACEMENTS.internalDevice.gateway);
54
- expect(env.OBSERVER_RESOLVERS).toContain('100.100.0.1');
54
+ expect(env.OBSERVER_RESOLVERS).toContain('203.0.113.1');
55
55
  });
56
56
  });
57
57
 
package/src/observer.ts CHANGED
@@ -47,7 +47,7 @@ export interface ObserverPlacement {
47
47
  // NO route to the segmented zones, so a dmz/app/secure container IP is unreachable.
48
48
  const HOME_ROUTER = '10.226.1.1';
49
49
  const INTERNAL_RESOLVER = '10.226.1.10';
50
- const PUBLIC_RESOLVER = '100.100.0.1';
50
+ const PUBLIC_RESOLVER = '203.0.113.1';
51
51
  const INTERNET_GATEWAY = '100.64.0.1'; // fw-ext, on internet-external
52
52
 
53
53
  /** Derive an observer host IP in a zone from its gateway (no new literal subnets). */
@@ -0,0 +1,121 @@
1
+ /**
2
+ * Swap the physical router the `fw-isp` simulator impersonates, on a RUNNING
3
+ * network — the harness half of the ISP-replaced-the-box scenario.
4
+ *
5
+ * `.axonRouter()` on the network builder cannot express this: it sets
6
+ * `ROUTER_VENDOR_PREFIX` in the generated compose file, which is read once when
7
+ * the simulator process starts. That is the right tool for "this network has an
8
+ * Axon in it"; it cannot say "this network had a GreenWave and now has an Axon",
9
+ * which is the event the whole module-pause change exists to survive.
10
+ *
11
+ * The swap models what actually happens to a fleet when an ISP replaces the
12
+ * hardware, and each part is load-bearing for what the migration test proves:
13
+ *
14
+ * - the TR-181 vendor prefix changes, so the OLD module's spelling is wrong;
15
+ * - the credentials change, which is why the fleet wedges rather than merely
16
+ * misbehaving — the old module cannot authenticate, so it can neither drive
17
+ * the new box nor clean up after itself;
18
+ * - every port forward is gone, because a new device arrives empty. That is
19
+ * the assertion with teeth: the forwards have to be recreated by unpausing
20
+ * the consumers that own them, and a simulator that kept its old table would
21
+ * let a completely broken unpause pass.
22
+ */
23
+
24
+ import { greenwaveRouterIp } from './types';
25
+ import type { NetworkHandle } from './types';
26
+
27
+ /** The devices the shared simulator can stand in for. */
28
+ export const ROUTER_DEVICES = {
29
+ /** GreenWave C4000XG — the default, driven by `modules/greenwave`. */
30
+ greenwave: { vendorPrefix: 'X_GWS_', username: 'admin', password: 'admin' },
31
+ /** Axon Networks Q1000K — driven by `modules/axon`. */
32
+ axon: { vendorPrefix: 'X_AXON_', username: 'axonadmin', password: 'axonsecret' },
33
+ } as const;
34
+
35
+ export type RouterDevice = keyof typeof ROUTER_DEVICES;
36
+
37
+ export interface RouterSwapResult {
38
+ /** Port forwards the outgoing device was holding, all of which are now gone. */
39
+ forwardsDiscarded: number;
40
+ device: { vendorPrefix: string; username: string };
41
+ }
42
+
43
+ /**
44
+ * Replace the impersonated device. Runs the request from INSIDE the simulator
45
+ * container, so the control endpoint never has to be reachable from anywhere
46
+ * else on the simulated internet.
47
+ */
48
+ export async function swapRouterDevice(
49
+ net: NetworkHandle,
50
+ device: RouterDevice,
51
+ ): Promise<RouterSwapResult> {
52
+ const spec = ROUTER_DEVICES[device];
53
+ const result = await net.exec(
54
+ 'fw-isp',
55
+ `curl -sk -X POST https://${greenwaveRouterIp()}/sim/swap-device ` +
56
+ `-d 'vendorPrefix=${spec.vendorPrefix}&username=${spec.username}&password=${spec.password}'`,
57
+ );
58
+ if (result.exitCode !== 0) {
59
+ throw new Error(`swapRouterDevice(${device}) failed: ${result.stderr || result.stdout}`);
60
+ }
61
+ return JSON.parse(result.stdout) as RouterSwapResult;
62
+ }
63
+
64
+ /**
65
+ * The port forwards the device is currently holding, read back OFF the device
66
+ * rather than out of celilo's own records.
67
+ *
68
+ * This exists because "verify the contract, not the verdict" is the whole point
69
+ * of the migration test: celilo reporting that it registered a forward is not
70
+ * evidence the router has one, and those are exactly the two things that came
71
+ * apart when the hardware changed underneath.
72
+ */
73
+ export async function listRouterForwards(net: NetworkHandle, device: RouterDevice) {
74
+ const spec = ROUTER_DEVICES[device];
75
+ const login = await net.exec(
76
+ 'fw-isp',
77
+ `curl -sk -X POST https://${greenwaveRouterIp()}/cgi/cgi_action ` +
78
+ `-d 'username=${spec.username}&password=${spec.password}' -c /tmp/swap-cookies`,
79
+ );
80
+ if (login.exitCode !== 0) {
81
+ throw new Error(`could not log in to the ${device} router: ${login.stderr || login.stdout}`);
82
+ }
83
+ const listed = await net.exec(
84
+ 'fw-isp',
85
+ `curl -sk 'https://${greenwaveRouterIp()}/cgi/cgi_get?Object=Device.NAT.PortMapping.' -b /tmp/swap-cookies`,
86
+ );
87
+ return listed.stdout;
88
+ }
89
+
90
+ /**
91
+ * The external ports currently forwarded on the device, as a sorted list.
92
+ *
93
+ * Assert on THIS rather than on a consumer's container IP. A forward's
94
+ * `InternalClient` is the firewall's natIp DNAT ingress (e.g. `10.226.1.253`),
95
+ * not the consumer's zone-side address — the delegation chain hops through the
96
+ * firewall — so matching a module's own IP looks correct and never matches.
97
+ */
98
+ export function forwardedPorts(listing: string): string[] {
99
+ const ports = [...listing.matchAll(/"ParamName":"ExternalPort","ParamValue":"(\d+)"/g)].map(
100
+ (m) => m[1],
101
+ );
102
+ return [...new Set(ports)].sort();
103
+ }
104
+
105
+ /** The DHCP pool's advertised DNS servers, read back off the device. */
106
+ export async function routerDhcpDnsServers(
107
+ net: NetworkHandle,
108
+ device: RouterDevice,
109
+ ): Promise<string> {
110
+ const spec = ROUTER_DEVICES[device];
111
+ await net.exec(
112
+ 'fw-isp',
113
+ `curl -sk -X POST https://${greenwaveRouterIp()}/cgi/cgi_action ` +
114
+ `-d 'username=${spec.username}&password=${spec.password}' -c /tmp/dhcp-cookies`,
115
+ );
116
+ const listed = await net.exec(
117
+ 'fw-isp',
118
+ `curl -sk 'https://${greenwaveRouterIp()}/cgi/cgi_get?Object=Device.DHCPv4.Server.Pool.' -b /tmp/dhcp-cookies`,
119
+ );
120
+ return listed.stdout;
121
+ }
@@ -5,8 +5,12 @@ import { join } from 'node:path';
5
5
  import {
6
6
  E2eBusyError,
7
7
  type LockHolder,
8
+ SUSPECT_HEARTBEAT_MS,
8
9
  acquireRunLock,
9
10
  clearLock,
11
+ formatBusy,
12
+ heartbeatAgeMs,
13
+ isSuspect,
10
14
  lockStatus,
11
15
  markKept,
12
16
  readHolder,
@@ -25,6 +29,7 @@ afterEach(() => {
25
29
  clearLock();
26
30
  rmSync(dir, { recursive: true, force: true });
27
31
  delete process.env.CELILO_E2E_LOCK_PATH;
32
+ delete process.env.CELILO_E2E_SESSION;
28
33
  });
29
34
 
30
35
  function writeRawHolder(h: Partial<LockHolder>): void {
@@ -71,7 +76,8 @@ test('a dead-pid holder on the same host is stale and reclaimed', () => {
71
76
  releaseRunLock();
72
77
  });
73
78
 
74
- test('--keep leaves a kept lock that survives release and blocks the next run', () => {
79
+ test('--keep leaves a kept lock that survives release and blocks other sessions', () => {
80
+ process.env.CELILO_E2E_SESSION = 'keeper (/keeper/worktree)';
75
81
  acquireRunLock({ test: 'kept-test', runId: 'r' });
76
82
  markKept();
77
83
  releaseRunLock();
@@ -80,7 +86,8 @@ test('--keep leaves a kept lock that survives release and blocks the next run',
80
86
  expect(h?.state).toBe('kept');
81
87
  expect(lockStatus().free).toBe(false); // kept is never stale
82
88
 
83
- // A plain run is refused...
89
+ // A plain run from ANOTHER session is refused...
90
+ process.env.CELILO_E2E_SESSION = 'other (/other/worktree)';
84
91
  expect(() => acquireRunLock({ test: 'next', runId: 'r2' })).toThrow(E2eBusyError);
85
92
  // ...but a --reuse run (allowKept) takes it over.
86
93
  acquireRunLock({ test: 'reuse', runId: 'r3', allowKept: true });
@@ -96,6 +103,70 @@ test('clearLock frees a kept lock (the `cele2e release` path)', () => {
96
103
  expect(lockStatus().free).toBe(true);
97
104
  });
98
105
 
106
+ test('a kept lock left by THIS session is auto-released by the next run', () => {
107
+ // The friction case: `run --keep` then `run` from the same session. Refusing
108
+ // here protected nobody — the only stack at risk was the caller's own — and
109
+ // the refusal was routinely misread as a finished run, because the previous
110
+ // run's results dir is still sitting there looking like a clean pass.
111
+ process.env.CELILO_E2E_SESSION = 'my-branch (/my/worktree)';
112
+ acquireRunLock({ test: 'kept-test', runId: 'r' });
113
+ markKept();
114
+ releaseRunLock();
115
+ expect(readHolder()?.state).toBe('kept');
116
+
117
+ const outcome = acquireRunLock({ test: 'next', runId: 'r2' });
118
+ expect(outcome.autoReleasedOwnKept?.test).toBe('kept-test');
119
+ expect(readHolder()?.pid).toBe(process.pid);
120
+ expect(readHolder()?.state).toBe('running');
121
+ releaseRunLock();
122
+ });
123
+
124
+ test('a kept lock from a DIFFERENT session is still refused', () => {
125
+ process.env.CELILO_E2E_SESSION = 'my-branch (/my/worktree)';
126
+ // A live pid so staleness can't be what frees it — the point is the session.
127
+ writeRawHolder({ state: 'kept', session: 'someone-else (/their/worktree)', pid: process.pid });
128
+ expect(() => acquireRunLock({ test: 'mine', runId: 'r' })).toThrow(E2eBusyError);
129
+ });
130
+
131
+ test('a live pid with a silent heartbeat is SUSPECT but never auto-reclaimed', () => {
132
+ // build-infra wedged at step [16/27]: the process is alive (so PID-liveness
133
+ // says "healthy") while its heartbeat has not ticked for half an hour. This
134
+ // is the one hang the existing staleness check structurally cannot see.
135
+ writeRawHolder({ pid: process.pid, beatAt: Date.now() - 1_913_000 });
136
+ const h = readHolder() as LockHolder;
137
+
138
+ expect(isSuspect(h)).toBe(true);
139
+ expect(heartbeatAgeMs(h)).toBeGreaterThan(SUSPECT_HEARTBEAT_MS);
140
+ expect(formatBusy(h)).toContain('SUSPECT');
141
+
142
+ const status = lockStatus();
143
+ expect(status.suspect).toBe(true);
144
+ // Suspect surfaces the hang; it does NOT kill someone's build for them.
145
+ expect(status.free).toBe(false);
146
+ expect(() => acquireRunLock({ test: 'other', runId: 'r' })).toThrow(E2eBusyError);
147
+ });
148
+
149
+ test('a fresh heartbeat is not suspect', () => {
150
+ writeRawHolder({ pid: process.pid, beatAt: Date.now() - 5_000 });
151
+ const h = readHolder() as LockHolder;
152
+ expect(isSuspect(h)).toBe(false);
153
+ expect(lockStatus().suspect).toBe(false);
154
+ });
155
+
156
+ test('a healthy build-infra that blocks the event loop is NOT suspect', () => {
157
+ // spawnSync('docker', ['build', …]) blocks the event loop, so the heartbeat
158
+ // stops for the whole of each image build. A live holder was observed 67s
159
+ // stale while making normal progress; the threshold has to clear that or the
160
+ // flag fires on every large image and stops meaning anything.
161
+ writeRawHolder({ pid: process.pid, beatAt: Date.now() - 120_000 });
162
+ expect(isSuspect(readHolder() as LockHolder)).toBe(false);
163
+ });
164
+
165
+ test('a kept holder is never suspect — its beatAt is frozen on purpose', () => {
166
+ writeRawHolder({ state: 'kept', pid: process.pid, beatAt: Date.now() - 3_600_000 });
167
+ expect(isSuspect(readHolder() as LockHolder)).toBe(false);
168
+ });
169
+
99
170
  test('a corrupt lock file is reclaimed, not fatal', () => {
100
171
  writeFileSync(process.env.CELILO_E2E_LOCK_PATH as string, 'not json{');
101
172
  expect(readHolder()).toBeNull();
package/src/run-lock.ts CHANGED
@@ -18,9 +18,14 @@
18
18
  * - Staleness: on the SAME host, PID-liveness is authoritative (process.kill
19
19
  * (pid, 0)); a dead holder PID → reclaim. Cross-host we can't check the PID,
20
20
  * so fall back to a heartbeat TTL (beatAt older than STALE_TTL_MS).
21
- * - A `kept` lock (left by `--keep` / `up`) is NEVER auto-reclaimed — it
22
- * guards a stack that outlives the process and is cleared only by
23
- * `cele2e release` / `cele2e down`. This enforces the honor-system rule.
21
+ * - A `kept` lock (left by `--keep` / `up`) is never reclaimed by ANOTHER
22
+ * actor — it guards a stack that outlives the process. Its OWN session
23
+ * reclaims it automatically (see isSameSession); everyone else clears it
24
+ * deliberately via `cele2e release` / `cele2e down`.
25
+ * - SUSPECT holders: a live PID whose heartbeat has gone quiet is the one
26
+ * hang the PID check cannot see (a build wedged mid-step keeps its process
27
+ * alive and sleeping). isSuspect() names it so `status`/`doctor` can flag it
28
+ * instead of leaving 20 silent minutes to be found by hand.
24
29
  */
25
30
 
26
31
  import { execFileSync } from 'node:child_process';
@@ -31,6 +36,23 @@ import { basename, dirname, join } from 'node:path';
31
36
  const STALE_TTL_MS = 90_000;
32
37
  const HEARTBEAT_MS = 30_000;
33
38
 
39
+ /**
40
+ * How quiet a RUNNING holder's heartbeat may go before it is suspect.
41
+ *
42
+ * Calibrated against what a HEALTHY holder actually does. The heartbeat is a
43
+ * setInterval, and build-infra shells out with spawnSync — which blocks the
44
+ * event loop for the entire duration of each `docker build`. So a perfectly
45
+ * healthy build stops beating for as long as its slowest image takes: a live
46
+ * holder was observed 67s stale while making normal progress, and the largest
47
+ * images here are ~2GB. Anything near the 30s tick interval would flag those.
48
+ *
49
+ * Ten minutes sits above any single legitimate image build (the per-image
50
+ * watchdog caps one at 15) and far below the ~20 and ~32 minute hangs that
51
+ * went unnoticed. Deliberately NOT a reclaim threshold: the PID is alive, and
52
+ * killing someone's wedged build out from under them is the operator's call.
53
+ */
54
+ export const SUSPECT_HEARTBEAT_MS = 600_000;
55
+
34
56
  /**
35
57
  * Machine-global lock path — NOT under any repo/worktree, because the Docker
36
58
  * infra is global regardless of which checkout started it. Overridable via
@@ -75,14 +97,22 @@ export function formatBusy(h: LockHolder): string {
75
97
  h.state === 'kept'
76
98
  ? '(run `cele2e release` to free it)'
77
99
  : `${h.test}, started ${age} ago (pid ${h.pid})`;
78
- return `e2e busy: ${h.session} ${verb} ${what}`;
100
+ const suspect = isSuspect(h)
101
+ ? ` — SUSPECT: no heartbeat for ${formatAge(heartbeatAgeMs(h))} (process alive but not progressing)`
102
+ : '';
103
+ return `e2e busy: ${h.session} ${verb} ${what}${suspect}`;
79
104
  }
80
105
 
81
106
  function ageString(since: number): string {
82
- const s = Math.max(0, Math.floor((Date.now() - since) / 1000));
107
+ return formatAge(Date.now() - since);
108
+ }
109
+
110
+ /** Human duration, e.g. "45s" / "32m" / "2h05m". */
111
+ export function formatAge(ms: number): string {
112
+ const s = Math.max(0, Math.floor(ms / 1000));
83
113
  if (s < 60) return `${s}s`;
84
114
  if (s < 3600) return `${Math.floor(s / 60)}m`;
85
- return `${Math.floor(s / 3600)}h${Math.floor((s % 3600) / 60)}m`;
115
+ return `${Math.floor(s / 3600)}h${String(Math.floor((s % 3600) / 60)).padStart(2, '0')}m`;
86
116
  }
87
117
 
88
118
  /** Branch + worktree path so a holder maps back to a specific session/thread. */
@@ -107,6 +137,28 @@ function deriveSession(): string {
107
137
  return branch || base;
108
138
  }
109
139
 
140
+ /**
141
+ * Is this holder the caller's own session — same host, same branch+worktree?
142
+ * This is what separates "my own kept stack is in my way" (friction, auto-clear)
143
+ * from "someone else is mid-run" (contention, refuse).
144
+ */
145
+ export function isSameSession(h: LockHolder): boolean {
146
+ return h.hostname === hostname() && h.session === deriveSession();
147
+ }
148
+
149
+ /** Milliseconds since the holder last refreshed its heartbeat. */
150
+ export function heartbeatAgeMs(h: LockHolder): number {
151
+ return Math.max(0, Date.now() - h.beatAt);
152
+ }
153
+
154
+ /**
155
+ * A live process whose heartbeat has gone quiet — the hang PID-liveness misses.
156
+ * `kept` holders are excluded: their beatAt is frozen on purpose at release.
157
+ */
158
+ export function isSuspect(h: LockHolder): boolean {
159
+ return h.state === 'running' && heartbeatAgeMs(h) > SUSPECT_HEARTBEAT_MS;
160
+ }
161
+
110
162
  function pidAlive(pid: number): boolean {
111
163
  try {
112
164
  process.kill(pid, 0);
@@ -140,15 +192,32 @@ function writeHolder(fd: number, h: LockHolder): void {
140
192
  writeFileSync(fd, JSON.stringify(h, null, 2));
141
193
  }
142
194
 
195
+ /** What the acquire had to clear on the way in, so callers can say so out loud. */
196
+ export interface AcquireOutcome {
197
+ /** A `kept` lock left by this same session that we auto-released, else null. */
198
+ autoReleasedOwnKept: LockHolder | null;
199
+ }
200
+
143
201
  /**
144
202
  * Acquire the run lock for this process. Throws E2eBusyError if another live
145
203
  * (non-stale) holder owns it. Registers an exit handler so the lock is released
146
204
  * on any process.exit() path (normal, exception, or a SIGINT handler that
147
205
  * exits). A signal-killed process with no exit handler leaks the lock, but the
148
206
  * next run reclaims it via PID-liveness — that's exactly what staleness is for.
207
+ *
208
+ * A `kept` lock left by THIS session (same host, same branch+worktree) is
209
+ * auto-released and reported in the outcome. Refusing there protected nobody:
210
+ * the only stack at risk was the caller's own, and the refusal was routinely
211
+ * misread as a finished run — the previous run's results dir is still sitting
212
+ * there looking like a clean pass. Another session's kept lock still refuses.
149
213
  */
150
- export function acquireRunLock(opts: { test: string; runId: string; allowKept?: boolean }): void {
214
+ export function acquireRunLock(opts: {
215
+ test: string;
216
+ runId: string;
217
+ allowKept?: boolean;
218
+ }): AcquireOutcome {
151
219
  mkdirSync(dirname(lockPath()), { recursive: true });
220
+ let autoReleasedOwnKept: LockHolder | null = null;
152
221
 
153
222
  const holder: LockHolder = {
154
223
  pid: process.pid,
@@ -168,9 +237,17 @@ export function acquireRunLock(opts: { test: string; runId: string; allowKept?:
168
237
  } catch (err) {
169
238
  if ((err as NodeJS.ErrnoException).code !== 'EEXIST') throw err;
170
239
  const existing = readHolder();
171
- // Reclaim when: corrupt/unreadable, the holder is stale, OR this is a
172
- // `--reuse` run taking over a `kept` stack it's deliberately reusing.
173
- if (!existing || isStale(existing) || (opts.allowKept && existing.state === 'kept')) {
240
+ // Reclaim when: corrupt/unreadable, the holder is stale, this is a
241
+ // `--reuse` run taking over a `kept` stack it's deliberately reusing, OR
242
+ // the `kept` stack belongs to this very session (see the doc comment).
243
+ const ownKept = !!existing && existing.state === 'kept' && isSameSession(existing);
244
+ if (
245
+ !existing ||
246
+ isStale(existing) ||
247
+ (opts.allowKept && existing.state === 'kept') ||
248
+ ownKept
249
+ ) {
250
+ if (ownKept && existing) autoReleasedOwnKept = existing;
174
251
  try {
175
252
  unlinkSync(lockPath());
176
253
  } catch {}
@@ -195,7 +272,7 @@ export function acquireRunLock(opts: { test: string; runId: string; allowKept?:
195
272
 
196
273
  held = { keepOnRelease: false, heartbeat, released: false };
197
274
  process.on('exit', releaseRunLock);
198
- return;
275
+ return { autoReleasedOwnKept };
199
276
  }
200
277
  // Two reclaim attempts both lost the race → someone else won fair and square.
201
278
  const existing = readHolder();
@@ -258,9 +335,28 @@ export function clearLock(): boolean {
258
335
  }
259
336
  }
260
337
 
338
+ export interface LockStatus {
339
+ free: boolean;
340
+ holder: LockHolder | null;
341
+ /** Milliseconds since the holder's last heartbeat; null when free. */
342
+ heartbeatAgeMs: number | null;
343
+ /** Live PID, quiet heartbeat — the hang the PID check cannot see. */
344
+ suspect: boolean;
345
+ /** The holder is this session's own kept stack, which the next run reclaims. */
346
+ ownKept: boolean;
347
+ }
348
+
261
349
  /** Current lock state for `cele2e status` (and any poller). */
262
- export function lockStatus(): { free: boolean; holder: LockHolder | null } {
350
+ export function lockStatus(): LockStatus {
263
351
  const h = readHolder();
264
- if (!h || isStale(h)) return { free: true, holder: null };
265
- return { free: false, holder: h };
352
+ if (!h || isStale(h)) {
353
+ return { free: true, holder: null, heartbeatAgeMs: null, suspect: false, ownKept: false };
354
+ }
355
+ return {
356
+ free: false,
357
+ holder: h,
358
+ heartbeatAgeMs: heartbeatAgeMs(h),
359
+ suspect: isSuspect(h),
360
+ ownKept: h.state === 'kept' && isSameSession(h),
361
+ };
266
362
  }