@celilo/e2e 0.12.0 → 0.13.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +5 -5
- package/bin/e2e-status +1 -1
- package/bin/e2e-up +2 -2
- package/config/dns/tangohost.com.zone +1 -1
- package/config/dns/templates/example.net.zone +3 -3
- package/config/dns/templates/iamtheinternet.org.zone +2 -2
- package/config/routing/fw-ext-routes.sh +8 -8
- package/config/routing/fw-isp-routes.sh +2 -2
- package/config/routing/fw-main-routes.sh +5 -5
- package/config/routing/management-routes.sh +1 -1
- package/config/routing/observer-setup.sh +1 -1
- package/config/routing/public-sim-entrypoint.sh +1 -1
- package/config/routing/resolver-internal-routes.sh +2 -2
- package/config/routing/resolver-routes.sh +1 -1
- package/config/routing/target-routes.sh +1 -1
- package/config/routing/target-setup.sh +1 -1
- package/config/socks/startup.sh +2 -2
- package/package.json +4 -3
- package/simulators/greenwave/server.ts +35 -0
- package/simulators/greenwave/state.ts +64 -11
- package/simulators/ip-echo/server.ts +1 -1
- package/src/address-plan.test.ts +4 -4
- package/src/cli/build.ts +30 -0
- package/src/cli/command-registry.ts +19 -0
- package/src/cli/completion.ts +16 -1
- package/src/cli/index.ts +85 -4
- package/src/container-manager.ts +75 -12
- package/src/docker-compose-generator.ts +55 -26
- package/src/doctor.test.ts +279 -0
- package/src/doctor.ts +421 -0
- package/src/extract-failure.ts +41 -0
- package/src/index.ts +12 -0
- package/src/last-run.test.ts +62 -0
- package/src/last-run.ts +54 -0
- package/src/observer.test.ts +2 -2
- package/src/observer.ts +1 -1
- package/src/router-swap.ts +121 -0
- package/src/run-lock.test.ts +73 -2
- package/src/run-lock.ts +110 -14
- package/src/runner.ts +99 -5
- package/src/socks-proxy.ts +2 -2
- package/src/types.ts +25 -4
- package/src/vantage.test.ts +1 -1
- package/src/zone-classifier.test.ts +11 -16
- package/src/zone-classifier.ts +16 -31
package/src/extract-failure.ts
CHANGED
|
@@ -12,6 +12,47 @@ export function stripAnsi(s: string): string {
|
|
|
12
12
|
return s.replace(ANSI_RE, '');
|
|
13
13
|
}
|
|
14
14
|
|
|
15
|
+
export interface StageTally {
|
|
16
|
+
/** Stages that failed on their own merits. */
|
|
17
|
+
failed: number;
|
|
18
|
+
/** Stages that never ran because an earlier stage had already failed. */
|
|
19
|
+
skipped: number;
|
|
20
|
+
}
|
|
21
|
+
|
|
22
|
+
/**
|
|
23
|
+
* Split a staged suite's failures into real ones and cascade-skips.
|
|
24
|
+
*
|
|
25
|
+
* Staged e2e suites guard every stage with `requireStage`, which throws
|
|
26
|
+
* `Skipped: <stage> — <earlier error>` once anything upstream has failed. bun
|
|
27
|
+
* counts each of those as a failure, so ONE bad fixture line in stage 1 reports
|
|
28
|
+
* as "9 failed" in a 10-stage suite — nine counts of a defect that does not
|
|
29
|
+
* exist, and a summary that buries the one that does.
|
|
30
|
+
*
|
|
31
|
+
* The `Skipped:` marker is the semantic signal, so that is what we key on; the
|
|
32
|
+
* sub-millisecond durations those stages show are a symptom, not the contract.
|
|
33
|
+
*/
|
|
34
|
+
export function tallyStages(lines: string[]): StageTally {
|
|
35
|
+
const plain = lines.map(stripAnsi);
|
|
36
|
+
const isFailure = (l: string): boolean => /^\(fail\)\s+\S/.test(l);
|
|
37
|
+
let failed = 0;
|
|
38
|
+
let skipped = 0;
|
|
39
|
+
for (let i = 0; i < plain.length; i++) {
|
|
40
|
+
if (!isFailure(plain[i])) continue;
|
|
41
|
+
// bun prints the reason on the following lines, indented under the failure.
|
|
42
|
+
// The window MUST stop at the next failure: cascade-skips come in runs, and
|
|
43
|
+
// a window that reads past the boundary attributes the next stage's
|
|
44
|
+
// "Skipped:" to this one — which misreads the single real defect at the top
|
|
45
|
+
// of a cascade as just another skip, losing the only line worth reading.
|
|
46
|
+
const reason: string[] = [];
|
|
47
|
+
for (let j = i + 1; j < plain.length && j <= i + 4 && !isFailure(plain[j]); j++) {
|
|
48
|
+
reason.push(plain[j]);
|
|
49
|
+
}
|
|
50
|
+
if (/(?:^|\W)Skipped:\s/.test(reason.join('\n'))) skipped++;
|
|
51
|
+
else failed++;
|
|
52
|
+
}
|
|
53
|
+
return { failed, skipped };
|
|
54
|
+
}
|
|
55
|
+
|
|
15
56
|
export function extractFailureMessage(lines: string[], exitCode: number): string {
|
|
16
57
|
const plain = lines.map(stripAnsi);
|
|
17
58
|
|
package/src/index.ts
CHANGED
|
@@ -102,3 +102,15 @@ export {
|
|
|
102
102
|
SHARED_PROJECT_NAME,
|
|
103
103
|
SHARED_NETWORKS,
|
|
104
104
|
} from './docker-compose-generator';
|
|
105
|
+
|
|
106
|
+
// Runtime router-device swap — the harness half of "the ISP replaced the box"
|
|
107
|
+
// (openspec/changes/module-pause-lifecycle task 7.7).
|
|
108
|
+
export {
|
|
109
|
+
ROUTER_DEVICES,
|
|
110
|
+
type RouterDevice,
|
|
111
|
+
type RouterSwapResult,
|
|
112
|
+
forwardedPorts,
|
|
113
|
+
listRouterForwards,
|
|
114
|
+
routerDhcpDnsServers,
|
|
115
|
+
swapRouterDevice,
|
|
116
|
+
} from './router-swap';
|
|
@@ -0,0 +1,62 @@
|
|
|
1
|
+
import { afterEach, beforeEach, expect, test } from 'bun:test';
|
|
2
|
+
import { mkdtempSync, rmSync } from 'node:fs';
|
|
3
|
+
import { tmpdir } from 'node:os';
|
|
4
|
+
import { join } from 'node:path';
|
|
5
|
+
import { type LastRun, readLastRun, writeLastRun } from './last-run';
|
|
6
|
+
|
|
7
|
+
let dir: string;
|
|
8
|
+
|
|
9
|
+
beforeEach(() => {
|
|
10
|
+
dir = mkdtempSync(join(tmpdir(), 'e2e-last-'));
|
|
11
|
+
process.env.CELILO_E2E_LAST_RUN_PATH = join(dir, 'last-run.json');
|
|
12
|
+
});
|
|
13
|
+
|
|
14
|
+
afterEach(() => {
|
|
15
|
+
rmSync(dir, { recursive: true, force: true });
|
|
16
|
+
delete process.env.CELILO_E2E_LAST_RUN_PATH;
|
|
17
|
+
});
|
|
18
|
+
|
|
19
|
+
const RUN: LastRun = {
|
|
20
|
+
runId: 'abc-123',
|
|
21
|
+
resultsDir: '/repo/e2e/results/2026-08-12T10-00-00',
|
|
22
|
+
startedAt: '2026-08-12T17:00:00.000Z',
|
|
23
|
+
status: 'completed',
|
|
24
|
+
total: 3,
|
|
25
|
+
passed: 3,
|
|
26
|
+
failed: 0,
|
|
27
|
+
skipped: 0,
|
|
28
|
+
durationMs: 600_000,
|
|
29
|
+
};
|
|
30
|
+
|
|
31
|
+
test('with no run recorded, readLastRun is null rather than throwing', () => {
|
|
32
|
+
expect(readLastRun()).toBeNull();
|
|
33
|
+
});
|
|
34
|
+
|
|
35
|
+
test('a recorded run round-trips every field `cele2e last --json` promises', () => {
|
|
36
|
+
writeLastRun(RUN);
|
|
37
|
+
expect(readLastRun()).toEqual(RUN);
|
|
38
|
+
});
|
|
39
|
+
|
|
40
|
+
test('the record is overwritten, so `last` never returns a stale earlier run', () => {
|
|
41
|
+
// The whole point: inferring "my results dir" from `ls -t results | head -1`
|
|
42
|
+
// hands back the PREVIOUS run's numbers when a run refuses to start, and a
|
|
43
|
+
// suite that never executed then reads as a clean pass.
|
|
44
|
+
writeLastRun(RUN);
|
|
45
|
+
writeLastRun({ ...RUN, runId: 'def-456', resultsDir: '/repo/e2e/results/later', failed: 2 });
|
|
46
|
+
const last = readLastRun();
|
|
47
|
+
expect(last?.runId).toBe('def-456');
|
|
48
|
+
expect(last?.resultsDir).toBe('/repo/e2e/results/later');
|
|
49
|
+
expect(last?.failed).toBe(2);
|
|
50
|
+
});
|
|
51
|
+
|
|
52
|
+
test('a run still in flight is recorded as running, so its logs are findable', () => {
|
|
53
|
+
writeLastRun({ ...RUN, status: 'running', passed: 0, total: 3 });
|
|
54
|
+
expect(readLastRun()?.status).toBe('running');
|
|
55
|
+
expect(readLastRun()?.resultsDir).toBe(RUN.resultsDir);
|
|
56
|
+
});
|
|
57
|
+
|
|
58
|
+
test('a corrupt record is null, not fatal', () => {
|
|
59
|
+
writeLastRun(RUN);
|
|
60
|
+
Bun.write(process.env.CELILO_E2E_LAST_RUN_PATH as string, 'not json{');
|
|
61
|
+
expect(readLastRun()).toBeNull();
|
|
62
|
+
});
|
package/src/last-run.ts
ADDED
|
@@ -0,0 +1,54 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* A machine-global pointer to the most recent cele2e run.
|
|
3
|
+
*
|
|
4
|
+
* Without one, "how did my run go?" is answered by `ls -t e2e/results | head -1`
|
|
5
|
+
* — which returns the newest directory, not YOUR run. When a run refuses to
|
|
6
|
+
* start (a held lock, a failed preflight) that inference silently hands back the
|
|
7
|
+
* PREVIOUS run's numbers, and a suite that never executed reads as a clean pass.
|
|
8
|
+
*
|
|
9
|
+
* Written at run start (so a run that dies mid-way still has a findable results
|
|
10
|
+
* dir) and again at the end with the counts. Lives beside the run-lock rather
|
|
11
|
+
* than in a worktree, because which checkout started the run is not something
|
|
12
|
+
* the reader knows.
|
|
13
|
+
*/
|
|
14
|
+
|
|
15
|
+
import { mkdirSync, readFileSync, writeFileSync } from 'node:fs';
|
|
16
|
+
import { homedir } from 'node:os';
|
|
17
|
+
import { dirname, join } from 'node:path';
|
|
18
|
+
|
|
19
|
+
export interface LastRun {
|
|
20
|
+
runId: string;
|
|
21
|
+
resultsDir: string;
|
|
22
|
+
startedAt: string;
|
|
23
|
+
/** 'running' until the suite finishes; then 'completed' or 'failed'. */
|
|
24
|
+
status: 'running' | 'completed' | 'failed';
|
|
25
|
+
total: number;
|
|
26
|
+
passed: number;
|
|
27
|
+
failed: number;
|
|
28
|
+
/** Stages a failing earlier stage blocked. Distinct from real failures. */
|
|
29
|
+
skipped: number;
|
|
30
|
+
durationMs: number;
|
|
31
|
+
}
|
|
32
|
+
|
|
33
|
+
export function lastRunPath(): string {
|
|
34
|
+
return (
|
|
35
|
+
process.env.CELILO_E2E_LAST_RUN_PATH || join(homedir(), '.cache', 'celilo-e2e', 'last-run.json')
|
|
36
|
+
);
|
|
37
|
+
}
|
|
38
|
+
|
|
39
|
+
export function writeLastRun(run: LastRun): void {
|
|
40
|
+
try {
|
|
41
|
+
mkdirSync(dirname(lastRunPath()), { recursive: true });
|
|
42
|
+
writeFileSync(lastRunPath(), `${JSON.stringify(run, null, 2)}\n`);
|
|
43
|
+
} catch {
|
|
44
|
+
// ponytail: a bookkeeping pointer must never be the thing that fails a run.
|
|
45
|
+
}
|
|
46
|
+
}
|
|
47
|
+
|
|
48
|
+
export function readLastRun(): LastRun | null {
|
|
49
|
+
try {
|
|
50
|
+
return JSON.parse(readFileSync(lastRunPath(), 'utf-8')) as LastRun;
|
|
51
|
+
} catch {
|
|
52
|
+
return null;
|
|
53
|
+
}
|
|
54
|
+
}
|
package/src/observer.test.ts
CHANGED
|
@@ -44,14 +44,14 @@ describe('placement faithfulness (the load-bearing invariants)', () => {
|
|
|
44
44
|
});
|
|
45
45
|
|
|
46
46
|
test('publicInternet uses only the public resolver (never the internal split-horizon view)', () => {
|
|
47
|
-
expect(OBSERVER_PLACEMENTS.publicInternet.resolvers).toEqual(['
|
|
47
|
+
expect(OBSERVER_PLACEMENTS.publicInternet.resolvers).toEqual(['203.0.113.1']);
|
|
48
48
|
});
|
|
49
49
|
|
|
50
50
|
test('observerEnv serializes the routing profile for the setup script', () => {
|
|
51
51
|
const env = observerEnv(OBSERVER_PLACEMENTS.internalDevice);
|
|
52
52
|
expect(env.OBSERVER_INTERZONE).toBe('0');
|
|
53
53
|
expect(env.OBSERVER_GATEWAY).toBe(OBSERVER_PLACEMENTS.internalDevice.gateway);
|
|
54
|
-
expect(env.OBSERVER_RESOLVERS).toContain('
|
|
54
|
+
expect(env.OBSERVER_RESOLVERS).toContain('203.0.113.1');
|
|
55
55
|
});
|
|
56
56
|
});
|
|
57
57
|
|
package/src/observer.ts
CHANGED
|
@@ -47,7 +47,7 @@ export interface ObserverPlacement {
|
|
|
47
47
|
// NO route to the segmented zones, so a dmz/app/secure container IP is unreachable.
|
|
48
48
|
const HOME_ROUTER = '10.226.1.1';
|
|
49
49
|
const INTERNAL_RESOLVER = '10.226.1.10';
|
|
50
|
-
const PUBLIC_RESOLVER = '
|
|
50
|
+
const PUBLIC_RESOLVER = '203.0.113.1';
|
|
51
51
|
const INTERNET_GATEWAY = '100.64.0.1'; // fw-ext, on internet-external
|
|
52
52
|
|
|
53
53
|
/** Derive an observer host IP in a zone from its gateway (no new literal subnets). */
|
|
@@ -0,0 +1,121 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Swap the physical router the `fw-isp` simulator impersonates, on a RUNNING
|
|
3
|
+
* network — the harness half of the ISP-replaced-the-box scenario.
|
|
4
|
+
*
|
|
5
|
+
* `.axonRouter()` on the network builder cannot express this: it sets
|
|
6
|
+
* `ROUTER_VENDOR_PREFIX` in the generated compose file, which is read once when
|
|
7
|
+
* the simulator process starts. That is the right tool for "this network has an
|
|
8
|
+
* Axon in it"; it cannot say "this network had a GreenWave and now has an Axon",
|
|
9
|
+
* which is the event the whole module-pause change exists to survive.
|
|
10
|
+
*
|
|
11
|
+
* The swap models what actually happens to a fleet when an ISP replaces the
|
|
12
|
+
* hardware, and each part is load-bearing for what the migration test proves:
|
|
13
|
+
*
|
|
14
|
+
* - the TR-181 vendor prefix changes, so the OLD module's spelling is wrong;
|
|
15
|
+
* - the credentials change, which is why the fleet wedges rather than merely
|
|
16
|
+
* misbehaving — the old module cannot authenticate, so it can neither drive
|
|
17
|
+
* the new box nor clean up after itself;
|
|
18
|
+
* - every port forward is gone, because a new device arrives empty. That is
|
|
19
|
+
* the assertion with teeth: the forwards have to be recreated by unpausing
|
|
20
|
+
* the consumers that own them, and a simulator that kept its old table would
|
|
21
|
+
* let a completely broken unpause pass.
|
|
22
|
+
*/
|
|
23
|
+
|
|
24
|
+
import { greenwaveRouterIp } from './types';
|
|
25
|
+
import type { NetworkHandle } from './types';
|
|
26
|
+
|
|
27
|
+
/** The devices the shared simulator can stand in for. */
|
|
28
|
+
export const ROUTER_DEVICES = {
|
|
29
|
+
/** GreenWave C4000XG — the default, driven by `modules/greenwave`. */
|
|
30
|
+
greenwave: { vendorPrefix: 'X_GWS_', username: 'admin', password: 'admin' },
|
|
31
|
+
/** Axon Networks Q1000K — driven by `modules/axon`. */
|
|
32
|
+
axon: { vendorPrefix: 'X_AXON_', username: 'axonadmin', password: 'axonsecret' },
|
|
33
|
+
} as const;
|
|
34
|
+
|
|
35
|
+
export type RouterDevice = keyof typeof ROUTER_DEVICES;
|
|
36
|
+
|
|
37
|
+
export interface RouterSwapResult {
|
|
38
|
+
/** Port forwards the outgoing device was holding, all of which are now gone. */
|
|
39
|
+
forwardsDiscarded: number;
|
|
40
|
+
device: { vendorPrefix: string; username: string };
|
|
41
|
+
}
|
|
42
|
+
|
|
43
|
+
/**
|
|
44
|
+
* Replace the impersonated device. Runs the request from INSIDE the simulator
|
|
45
|
+
* container, so the control endpoint never has to be reachable from anywhere
|
|
46
|
+
* else on the simulated internet.
|
|
47
|
+
*/
|
|
48
|
+
export async function swapRouterDevice(
|
|
49
|
+
net: NetworkHandle,
|
|
50
|
+
device: RouterDevice,
|
|
51
|
+
): Promise<RouterSwapResult> {
|
|
52
|
+
const spec = ROUTER_DEVICES[device];
|
|
53
|
+
const result = await net.exec(
|
|
54
|
+
'fw-isp',
|
|
55
|
+
`curl -sk -X POST https://${greenwaveRouterIp()}/sim/swap-device ` +
|
|
56
|
+
`-d 'vendorPrefix=${spec.vendorPrefix}&username=${spec.username}&password=${spec.password}'`,
|
|
57
|
+
);
|
|
58
|
+
if (result.exitCode !== 0) {
|
|
59
|
+
throw new Error(`swapRouterDevice(${device}) failed: ${result.stderr || result.stdout}`);
|
|
60
|
+
}
|
|
61
|
+
return JSON.parse(result.stdout) as RouterSwapResult;
|
|
62
|
+
}
|
|
63
|
+
|
|
64
|
+
/**
|
|
65
|
+
* The port forwards the device is currently holding, read back OFF the device
|
|
66
|
+
* rather than out of celilo's own records.
|
|
67
|
+
*
|
|
68
|
+
* This exists because "verify the contract, not the verdict" is the whole point
|
|
69
|
+
* of the migration test: celilo reporting that it registered a forward is not
|
|
70
|
+
* evidence the router has one, and those are exactly the two things that came
|
|
71
|
+
* apart when the hardware changed underneath.
|
|
72
|
+
*/
|
|
73
|
+
export async function listRouterForwards(net: NetworkHandle, device: RouterDevice) {
|
|
74
|
+
const spec = ROUTER_DEVICES[device];
|
|
75
|
+
const login = await net.exec(
|
|
76
|
+
'fw-isp',
|
|
77
|
+
`curl -sk -X POST https://${greenwaveRouterIp()}/cgi/cgi_action ` +
|
|
78
|
+
`-d 'username=${spec.username}&password=${spec.password}' -c /tmp/swap-cookies`,
|
|
79
|
+
);
|
|
80
|
+
if (login.exitCode !== 0) {
|
|
81
|
+
throw new Error(`could not log in to the ${device} router: ${login.stderr || login.stdout}`);
|
|
82
|
+
}
|
|
83
|
+
const listed = await net.exec(
|
|
84
|
+
'fw-isp',
|
|
85
|
+
`curl -sk 'https://${greenwaveRouterIp()}/cgi/cgi_get?Object=Device.NAT.PortMapping.' -b /tmp/swap-cookies`,
|
|
86
|
+
);
|
|
87
|
+
return listed.stdout;
|
|
88
|
+
}
|
|
89
|
+
|
|
90
|
+
/**
|
|
91
|
+
* The external ports currently forwarded on the device, as a sorted list.
|
|
92
|
+
*
|
|
93
|
+
* Assert on THIS rather than on a consumer's container IP. A forward's
|
|
94
|
+
* `InternalClient` is the firewall's natIp DNAT ingress (e.g. `10.226.1.253`),
|
|
95
|
+
* not the consumer's zone-side address — the delegation chain hops through the
|
|
96
|
+
* firewall — so matching a module's own IP looks correct and never matches.
|
|
97
|
+
*/
|
|
98
|
+
export function forwardedPorts(listing: string): string[] {
|
|
99
|
+
const ports = [...listing.matchAll(/"ParamName":"ExternalPort","ParamValue":"(\d+)"/g)].map(
|
|
100
|
+
(m) => m[1],
|
|
101
|
+
);
|
|
102
|
+
return [...new Set(ports)].sort();
|
|
103
|
+
}
|
|
104
|
+
|
|
105
|
+
/** The DHCP pool's advertised DNS servers, read back off the device. */
|
|
106
|
+
export async function routerDhcpDnsServers(
|
|
107
|
+
net: NetworkHandle,
|
|
108
|
+
device: RouterDevice,
|
|
109
|
+
): Promise<string> {
|
|
110
|
+
const spec = ROUTER_DEVICES[device];
|
|
111
|
+
await net.exec(
|
|
112
|
+
'fw-isp',
|
|
113
|
+
`curl -sk -X POST https://${greenwaveRouterIp()}/cgi/cgi_action ` +
|
|
114
|
+
`-d 'username=${spec.username}&password=${spec.password}' -c /tmp/dhcp-cookies`,
|
|
115
|
+
);
|
|
116
|
+
const listed = await net.exec(
|
|
117
|
+
'fw-isp',
|
|
118
|
+
`curl -sk 'https://${greenwaveRouterIp()}/cgi/cgi_get?Object=Device.DHCPv4.Server.Pool.' -b /tmp/dhcp-cookies`,
|
|
119
|
+
);
|
|
120
|
+
return listed.stdout;
|
|
121
|
+
}
|
package/src/run-lock.test.ts
CHANGED
|
@@ -5,8 +5,12 @@ import { join } from 'node:path';
|
|
|
5
5
|
import {
|
|
6
6
|
E2eBusyError,
|
|
7
7
|
type LockHolder,
|
|
8
|
+
SUSPECT_HEARTBEAT_MS,
|
|
8
9
|
acquireRunLock,
|
|
9
10
|
clearLock,
|
|
11
|
+
formatBusy,
|
|
12
|
+
heartbeatAgeMs,
|
|
13
|
+
isSuspect,
|
|
10
14
|
lockStatus,
|
|
11
15
|
markKept,
|
|
12
16
|
readHolder,
|
|
@@ -25,6 +29,7 @@ afterEach(() => {
|
|
|
25
29
|
clearLock();
|
|
26
30
|
rmSync(dir, { recursive: true, force: true });
|
|
27
31
|
delete process.env.CELILO_E2E_LOCK_PATH;
|
|
32
|
+
delete process.env.CELILO_E2E_SESSION;
|
|
28
33
|
});
|
|
29
34
|
|
|
30
35
|
function writeRawHolder(h: Partial<LockHolder>): void {
|
|
@@ -71,7 +76,8 @@ test('a dead-pid holder on the same host is stale and reclaimed', () => {
|
|
|
71
76
|
releaseRunLock();
|
|
72
77
|
});
|
|
73
78
|
|
|
74
|
-
test('--keep leaves a kept lock that survives release and blocks
|
|
79
|
+
test('--keep leaves a kept lock that survives release and blocks other sessions', () => {
|
|
80
|
+
process.env.CELILO_E2E_SESSION = 'keeper (/keeper/worktree)';
|
|
75
81
|
acquireRunLock({ test: 'kept-test', runId: 'r' });
|
|
76
82
|
markKept();
|
|
77
83
|
releaseRunLock();
|
|
@@ -80,7 +86,8 @@ test('--keep leaves a kept lock that survives release and blocks the next run',
|
|
|
80
86
|
expect(h?.state).toBe('kept');
|
|
81
87
|
expect(lockStatus().free).toBe(false); // kept is never stale
|
|
82
88
|
|
|
83
|
-
// A plain run is refused...
|
|
89
|
+
// A plain run from ANOTHER session is refused...
|
|
90
|
+
process.env.CELILO_E2E_SESSION = 'other (/other/worktree)';
|
|
84
91
|
expect(() => acquireRunLock({ test: 'next', runId: 'r2' })).toThrow(E2eBusyError);
|
|
85
92
|
// ...but a --reuse run (allowKept) takes it over.
|
|
86
93
|
acquireRunLock({ test: 'reuse', runId: 'r3', allowKept: true });
|
|
@@ -96,6 +103,70 @@ test('clearLock frees a kept lock (the `cele2e release` path)', () => {
|
|
|
96
103
|
expect(lockStatus().free).toBe(true);
|
|
97
104
|
});
|
|
98
105
|
|
|
106
|
+
test('a kept lock left by THIS session is auto-released by the next run', () => {
|
|
107
|
+
// The friction case: `run --keep` then `run` from the same session. Refusing
|
|
108
|
+
// here protected nobody — the only stack at risk was the caller's own — and
|
|
109
|
+
// the refusal was routinely misread as a finished run, because the previous
|
|
110
|
+
// run's results dir is still sitting there looking like a clean pass.
|
|
111
|
+
process.env.CELILO_E2E_SESSION = 'my-branch (/my/worktree)';
|
|
112
|
+
acquireRunLock({ test: 'kept-test', runId: 'r' });
|
|
113
|
+
markKept();
|
|
114
|
+
releaseRunLock();
|
|
115
|
+
expect(readHolder()?.state).toBe('kept');
|
|
116
|
+
|
|
117
|
+
const outcome = acquireRunLock({ test: 'next', runId: 'r2' });
|
|
118
|
+
expect(outcome.autoReleasedOwnKept?.test).toBe('kept-test');
|
|
119
|
+
expect(readHolder()?.pid).toBe(process.pid);
|
|
120
|
+
expect(readHolder()?.state).toBe('running');
|
|
121
|
+
releaseRunLock();
|
|
122
|
+
});
|
|
123
|
+
|
|
124
|
+
test('a kept lock from a DIFFERENT session is still refused', () => {
|
|
125
|
+
process.env.CELILO_E2E_SESSION = 'my-branch (/my/worktree)';
|
|
126
|
+
// A live pid so staleness can't be what frees it — the point is the session.
|
|
127
|
+
writeRawHolder({ state: 'kept', session: 'someone-else (/their/worktree)', pid: process.pid });
|
|
128
|
+
expect(() => acquireRunLock({ test: 'mine', runId: 'r' })).toThrow(E2eBusyError);
|
|
129
|
+
});
|
|
130
|
+
|
|
131
|
+
test('a live pid with a silent heartbeat is SUSPECT but never auto-reclaimed', () => {
|
|
132
|
+
// build-infra wedged at step [16/27]: the process is alive (so PID-liveness
|
|
133
|
+
// says "healthy") while its heartbeat has not ticked for half an hour. This
|
|
134
|
+
// is the one hang the existing staleness check structurally cannot see.
|
|
135
|
+
writeRawHolder({ pid: process.pid, beatAt: Date.now() - 1_913_000 });
|
|
136
|
+
const h = readHolder() as LockHolder;
|
|
137
|
+
|
|
138
|
+
expect(isSuspect(h)).toBe(true);
|
|
139
|
+
expect(heartbeatAgeMs(h)).toBeGreaterThan(SUSPECT_HEARTBEAT_MS);
|
|
140
|
+
expect(formatBusy(h)).toContain('SUSPECT');
|
|
141
|
+
|
|
142
|
+
const status = lockStatus();
|
|
143
|
+
expect(status.suspect).toBe(true);
|
|
144
|
+
// Suspect surfaces the hang; it does NOT kill someone's build for them.
|
|
145
|
+
expect(status.free).toBe(false);
|
|
146
|
+
expect(() => acquireRunLock({ test: 'other', runId: 'r' })).toThrow(E2eBusyError);
|
|
147
|
+
});
|
|
148
|
+
|
|
149
|
+
test('a fresh heartbeat is not suspect', () => {
|
|
150
|
+
writeRawHolder({ pid: process.pid, beatAt: Date.now() - 5_000 });
|
|
151
|
+
const h = readHolder() as LockHolder;
|
|
152
|
+
expect(isSuspect(h)).toBe(false);
|
|
153
|
+
expect(lockStatus().suspect).toBe(false);
|
|
154
|
+
});
|
|
155
|
+
|
|
156
|
+
test('a healthy build-infra that blocks the event loop is NOT suspect', () => {
|
|
157
|
+
// spawnSync('docker', ['build', …]) blocks the event loop, so the heartbeat
|
|
158
|
+
// stops for the whole of each image build. A live holder was observed 67s
|
|
159
|
+
// stale while making normal progress; the threshold has to clear that or the
|
|
160
|
+
// flag fires on every large image and stops meaning anything.
|
|
161
|
+
writeRawHolder({ pid: process.pid, beatAt: Date.now() - 120_000 });
|
|
162
|
+
expect(isSuspect(readHolder() as LockHolder)).toBe(false);
|
|
163
|
+
});
|
|
164
|
+
|
|
165
|
+
test('a kept holder is never suspect — its beatAt is frozen on purpose', () => {
|
|
166
|
+
writeRawHolder({ state: 'kept', pid: process.pid, beatAt: Date.now() - 3_600_000 });
|
|
167
|
+
expect(isSuspect(readHolder() as LockHolder)).toBe(false);
|
|
168
|
+
});
|
|
169
|
+
|
|
99
170
|
test('a corrupt lock file is reclaimed, not fatal', () => {
|
|
100
171
|
writeFileSync(process.env.CELILO_E2E_LOCK_PATH as string, 'not json{');
|
|
101
172
|
expect(readHolder()).toBeNull();
|
package/src/run-lock.ts
CHANGED
|
@@ -18,9 +18,14 @@
|
|
|
18
18
|
* - Staleness: on the SAME host, PID-liveness is authoritative (process.kill
|
|
19
19
|
* (pid, 0)); a dead holder PID → reclaim. Cross-host we can't check the PID,
|
|
20
20
|
* so fall back to a heartbeat TTL (beatAt older than STALE_TTL_MS).
|
|
21
|
-
* - A `kept` lock (left by `--keep` / `up`) is
|
|
22
|
-
* guards a stack that outlives the process
|
|
23
|
-
*
|
|
21
|
+
* - A `kept` lock (left by `--keep` / `up`) is never reclaimed by ANOTHER
|
|
22
|
+
* actor — it guards a stack that outlives the process. Its OWN session
|
|
23
|
+
* reclaims it automatically (see isSameSession); everyone else clears it
|
|
24
|
+
* deliberately via `cele2e release` / `cele2e down`.
|
|
25
|
+
* - SUSPECT holders: a live PID whose heartbeat has gone quiet is the one
|
|
26
|
+
* hang the PID check cannot see (a build wedged mid-step keeps its process
|
|
27
|
+
* alive and sleeping). isSuspect() names it so `status`/`doctor` can flag it
|
|
28
|
+
* instead of leaving 20 silent minutes to be found by hand.
|
|
24
29
|
*/
|
|
25
30
|
|
|
26
31
|
import { execFileSync } from 'node:child_process';
|
|
@@ -31,6 +36,23 @@ import { basename, dirname, join } from 'node:path';
|
|
|
31
36
|
const STALE_TTL_MS = 90_000;
|
|
32
37
|
const HEARTBEAT_MS = 30_000;
|
|
33
38
|
|
|
39
|
+
/**
|
|
40
|
+
* How quiet a RUNNING holder's heartbeat may go before it is suspect.
|
|
41
|
+
*
|
|
42
|
+
* Calibrated against what a HEALTHY holder actually does. The heartbeat is a
|
|
43
|
+
* setInterval, and build-infra shells out with spawnSync — which blocks the
|
|
44
|
+
* event loop for the entire duration of each `docker build`. So a perfectly
|
|
45
|
+
* healthy build stops beating for as long as its slowest image takes: a live
|
|
46
|
+
* holder was observed 67s stale while making normal progress, and the largest
|
|
47
|
+
* images here are ~2GB. Anything near the 30s tick interval would flag those.
|
|
48
|
+
*
|
|
49
|
+
* Ten minutes sits above any single legitimate image build (the per-image
|
|
50
|
+
* watchdog caps one at 15) and far below the ~20 and ~32 minute hangs that
|
|
51
|
+
* went unnoticed. Deliberately NOT a reclaim threshold: the PID is alive, and
|
|
52
|
+
* killing someone's wedged build out from under them is the operator's call.
|
|
53
|
+
*/
|
|
54
|
+
export const SUSPECT_HEARTBEAT_MS = 600_000;
|
|
55
|
+
|
|
34
56
|
/**
|
|
35
57
|
* Machine-global lock path — NOT under any repo/worktree, because the Docker
|
|
36
58
|
* infra is global regardless of which checkout started it. Overridable via
|
|
@@ -75,14 +97,22 @@ export function formatBusy(h: LockHolder): string {
|
|
|
75
97
|
h.state === 'kept'
|
|
76
98
|
? '(run `cele2e release` to free it)'
|
|
77
99
|
: `${h.test}, started ${age} ago (pid ${h.pid})`;
|
|
78
|
-
|
|
100
|
+
const suspect = isSuspect(h)
|
|
101
|
+
? ` — SUSPECT: no heartbeat for ${formatAge(heartbeatAgeMs(h))} (process alive but not progressing)`
|
|
102
|
+
: '';
|
|
103
|
+
return `e2e busy: ${h.session} ${verb} ${what}${suspect}`;
|
|
79
104
|
}
|
|
80
105
|
|
|
81
106
|
function ageString(since: number): string {
|
|
82
|
-
|
|
107
|
+
return formatAge(Date.now() - since);
|
|
108
|
+
}
|
|
109
|
+
|
|
110
|
+
/** Human duration, e.g. "45s" / "32m" / "2h05m". */
|
|
111
|
+
export function formatAge(ms: number): string {
|
|
112
|
+
const s = Math.max(0, Math.floor(ms / 1000));
|
|
83
113
|
if (s < 60) return `${s}s`;
|
|
84
114
|
if (s < 3600) return `${Math.floor(s / 60)}m`;
|
|
85
|
-
return `${Math.floor(s / 3600)}h${Math.floor((s % 3600) / 60)}m`;
|
|
115
|
+
return `${Math.floor(s / 3600)}h${String(Math.floor((s % 3600) / 60)).padStart(2, '0')}m`;
|
|
86
116
|
}
|
|
87
117
|
|
|
88
118
|
/** Branch + worktree path so a holder maps back to a specific session/thread. */
|
|
@@ -107,6 +137,28 @@ function deriveSession(): string {
|
|
|
107
137
|
return branch || base;
|
|
108
138
|
}
|
|
109
139
|
|
|
140
|
+
/**
|
|
141
|
+
* Is this holder the caller's own session — same host, same branch+worktree?
|
|
142
|
+
* This is what separates "my own kept stack is in my way" (friction, auto-clear)
|
|
143
|
+
* from "someone else is mid-run" (contention, refuse).
|
|
144
|
+
*/
|
|
145
|
+
export function isSameSession(h: LockHolder): boolean {
|
|
146
|
+
return h.hostname === hostname() && h.session === deriveSession();
|
|
147
|
+
}
|
|
148
|
+
|
|
149
|
+
/** Milliseconds since the holder last refreshed its heartbeat. */
|
|
150
|
+
export function heartbeatAgeMs(h: LockHolder): number {
|
|
151
|
+
return Math.max(0, Date.now() - h.beatAt);
|
|
152
|
+
}
|
|
153
|
+
|
|
154
|
+
/**
|
|
155
|
+
* A live process whose heartbeat has gone quiet — the hang PID-liveness misses.
|
|
156
|
+
* `kept` holders are excluded: their beatAt is frozen on purpose at release.
|
|
157
|
+
*/
|
|
158
|
+
export function isSuspect(h: LockHolder): boolean {
|
|
159
|
+
return h.state === 'running' && heartbeatAgeMs(h) > SUSPECT_HEARTBEAT_MS;
|
|
160
|
+
}
|
|
161
|
+
|
|
110
162
|
function pidAlive(pid: number): boolean {
|
|
111
163
|
try {
|
|
112
164
|
process.kill(pid, 0);
|
|
@@ -140,15 +192,32 @@ function writeHolder(fd: number, h: LockHolder): void {
|
|
|
140
192
|
writeFileSync(fd, JSON.stringify(h, null, 2));
|
|
141
193
|
}
|
|
142
194
|
|
|
195
|
+
/** What the acquire had to clear on the way in, so callers can say so out loud. */
|
|
196
|
+
export interface AcquireOutcome {
|
|
197
|
+
/** A `kept` lock left by this same session that we auto-released, else null. */
|
|
198
|
+
autoReleasedOwnKept: LockHolder | null;
|
|
199
|
+
}
|
|
200
|
+
|
|
143
201
|
/**
|
|
144
202
|
* Acquire the run lock for this process. Throws E2eBusyError if another live
|
|
145
203
|
* (non-stale) holder owns it. Registers an exit handler so the lock is released
|
|
146
204
|
* on any process.exit() path (normal, exception, or a SIGINT handler that
|
|
147
205
|
* exits). A signal-killed process with no exit handler leaks the lock, but the
|
|
148
206
|
* next run reclaims it via PID-liveness — that's exactly what staleness is for.
|
|
207
|
+
*
|
|
208
|
+
* A `kept` lock left by THIS session (same host, same branch+worktree) is
|
|
209
|
+
* auto-released and reported in the outcome. Refusing there protected nobody:
|
|
210
|
+
* the only stack at risk was the caller's own, and the refusal was routinely
|
|
211
|
+
* misread as a finished run — the previous run's results dir is still sitting
|
|
212
|
+
* there looking like a clean pass. Another session's kept lock still refuses.
|
|
149
213
|
*/
|
|
150
|
-
export function acquireRunLock(opts: {
|
|
214
|
+
export function acquireRunLock(opts: {
|
|
215
|
+
test: string;
|
|
216
|
+
runId: string;
|
|
217
|
+
allowKept?: boolean;
|
|
218
|
+
}): AcquireOutcome {
|
|
151
219
|
mkdirSync(dirname(lockPath()), { recursive: true });
|
|
220
|
+
let autoReleasedOwnKept: LockHolder | null = null;
|
|
152
221
|
|
|
153
222
|
const holder: LockHolder = {
|
|
154
223
|
pid: process.pid,
|
|
@@ -168,9 +237,17 @@ export function acquireRunLock(opts: { test: string; runId: string; allowKept?:
|
|
|
168
237
|
} catch (err) {
|
|
169
238
|
if ((err as NodeJS.ErrnoException).code !== 'EEXIST') throw err;
|
|
170
239
|
const existing = readHolder();
|
|
171
|
-
// Reclaim when: corrupt/unreadable, the holder is stale,
|
|
172
|
-
// `--reuse` run taking over a `kept` stack it's deliberately reusing
|
|
173
|
-
|
|
240
|
+
// Reclaim when: corrupt/unreadable, the holder is stale, this is a
|
|
241
|
+
// `--reuse` run taking over a `kept` stack it's deliberately reusing, OR
|
|
242
|
+
// the `kept` stack belongs to this very session (see the doc comment).
|
|
243
|
+
const ownKept = !!existing && existing.state === 'kept' && isSameSession(existing);
|
|
244
|
+
if (
|
|
245
|
+
!existing ||
|
|
246
|
+
isStale(existing) ||
|
|
247
|
+
(opts.allowKept && existing.state === 'kept') ||
|
|
248
|
+
ownKept
|
|
249
|
+
) {
|
|
250
|
+
if (ownKept && existing) autoReleasedOwnKept = existing;
|
|
174
251
|
try {
|
|
175
252
|
unlinkSync(lockPath());
|
|
176
253
|
} catch {}
|
|
@@ -195,7 +272,7 @@ export function acquireRunLock(opts: { test: string; runId: string; allowKept?:
|
|
|
195
272
|
|
|
196
273
|
held = { keepOnRelease: false, heartbeat, released: false };
|
|
197
274
|
process.on('exit', releaseRunLock);
|
|
198
|
-
return;
|
|
275
|
+
return { autoReleasedOwnKept };
|
|
199
276
|
}
|
|
200
277
|
// Two reclaim attempts both lost the race → someone else won fair and square.
|
|
201
278
|
const existing = readHolder();
|
|
@@ -258,9 +335,28 @@ export function clearLock(): boolean {
|
|
|
258
335
|
}
|
|
259
336
|
}
|
|
260
337
|
|
|
338
|
+
export interface LockStatus {
|
|
339
|
+
free: boolean;
|
|
340
|
+
holder: LockHolder | null;
|
|
341
|
+
/** Milliseconds since the holder's last heartbeat; null when free. */
|
|
342
|
+
heartbeatAgeMs: number | null;
|
|
343
|
+
/** Live PID, quiet heartbeat — the hang the PID check cannot see. */
|
|
344
|
+
suspect: boolean;
|
|
345
|
+
/** The holder is this session's own kept stack, which the next run reclaims. */
|
|
346
|
+
ownKept: boolean;
|
|
347
|
+
}
|
|
348
|
+
|
|
261
349
|
/** Current lock state for `cele2e status` (and any poller). */
|
|
262
|
-
export function lockStatus():
|
|
350
|
+
export function lockStatus(): LockStatus {
|
|
263
351
|
const h = readHolder();
|
|
264
|
-
if (!h || isStale(h))
|
|
265
|
-
|
|
352
|
+
if (!h || isStale(h)) {
|
|
353
|
+
return { free: true, holder: null, heartbeatAgeMs: null, suspect: false, ownKept: false };
|
|
354
|
+
}
|
|
355
|
+
return {
|
|
356
|
+
free: false,
|
|
357
|
+
holder: h,
|
|
358
|
+
heartbeatAgeMs: heartbeatAgeMs(h),
|
|
359
|
+
suspect: isSuspect(h),
|
|
360
|
+
ownKept: h.state === 'kept' && isSameSession(h),
|
|
361
|
+
};
|
|
266
362
|
}
|