@celilo/e2e 0.10.2 → 0.10.4

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@celilo/e2e",
3
- "version": "0.10.2",
3
+ "version": "0.10.4",
4
4
  "description": "E2E test infrastructure for Celilo-deployed applications. Provides a simulated internet with DNS hierarchy, ACME server, firewalls, and target machines in Docker.",
5
5
  "type": "module",
6
6
  "main": "./src/index.ts",
@@ -38,7 +38,7 @@
38
38
  ],
39
39
  "dependencies": {
40
40
  "@celilo/cli-display": "^0.1.9",
41
- "@celilo/event-bus": "^0.1.9",
41
+ "@celilo/event-bus": "^0.1.10",
42
42
  "yaml": "^2.8.0",
43
43
  "zod": "^3.24.1"
44
44
  },
@@ -113,6 +113,14 @@ function handleSend(id: unknown, params: Record<string, unknown>): Response {
113
113
  if (recipients.length === 0) return rpcError(id, -32602, 'No recipient given');
114
114
 
115
115
  for (const recipient of recipients) {
116
+ // The linked account is ALWAYS a valid recipient: messaging yourself is
117
+ // note-to-self, a first-class Signal feature and the default
118
+ // single-operator setup (design R2) — the route points at the very number
119
+ // the transport is a secondary device of. Rejecting it as "unregistered"
120
+ // is something no real daemon does, and it made the note-to-self ack path
121
+ // untestable end to end: the page could not be sent, so no token ever
122
+ // reached the operator to reply with.
123
+ if (recipient === ACCOUNT) continue;
116
124
  // Push back like Signal: an unregistered number is a hard failure, not a
117
125
  // silent no-op.
118
126
  if (KNOWN_RECIPIENTS.size > 0 && !KNOWN_RECIPIENTS.has(recipient)) {
@@ -75,6 +75,78 @@ const activeProjects: Set<string> = new Set();
75
75
  * MUST stay fully synchronous: it runs from an `exit` handler, where the
76
76
  * event loop is already closed and any async work is silently dropped.
77
77
  */
78
+ /**
79
+ * The commands that remove a per-test project WITHOUT needing its compose file.
80
+ * Pure (Rule 10.1) so the no-compose-dependency property is testable.
81
+ *
82
+ * `docker compose -p <project> down` — what this used to run — resolves the
83
+ * project from a compose file in the CURRENT DIRECTORY. An exit handler has no
84
+ * reliable cwd, so compose exited "no configuration file provided", the error
85
+ * was swallowed by the surrounding catch, and nothing was removed. The handler
86
+ * fired correctly and tore down nothing.
87
+ *
88
+ * Removing by name needs only the project name, which we already have.
89
+ * Per-test projects are `celilo-e2e-<timestamp>`; the shared project is
90
+ * `celilo-e2e-shared`, which contains no timestamp and so is never matched —
91
+ * shared infra stays up, as it must.
92
+ */
93
+ export function projectTeardownCommands(project: string): {
94
+ listContainers: string;
95
+ listNetworks: string;
96
+ listVolumes: string;
97
+ } {
98
+ return {
99
+ listContainers: `docker ps -aq --filter name=${project}`,
100
+ listNetworks: `docker network ls --format {{.Name}} --filter name=${project}`,
101
+ // Volumes were missing here, and the compose path that would have removed
102
+ // them (`down --volumes`) is the same best-effort call that silently does
103
+ // nothing without a resolvable compose file. So every run leaked its
104
+ // volumes: 58 `celilo-e2e-<ts>_ssh-keys` volumes were found on the builder,
105
+ // the oldest months old. Nothing breaks from a leaked volume the way a
106
+ // leaked network breaks routing — it just accumulates unbounded, and the
107
+ // pile is the evidence that teardown has not been running.
108
+ listVolumes: `docker volume ls -q --filter name=${project}`,
109
+ };
110
+ }
111
+
112
+ /** Force-remove a project's containers, networks and volumes by name. Never throws. */
113
+ function forceRemoveProject(project: string): void {
114
+ const cmds = projectTeardownCommands(project);
115
+ try {
116
+ const ids = execSync(cmds.listContainers, { timeout: 15_000, stdio: 'pipe' })
117
+ .toString()
118
+ .split('\n')
119
+ .filter(Boolean);
120
+ if (ids.length > 0) {
121
+ execSync(`docker rm -f ${ids.join(' ')}`, { timeout: 30_000, stdio: 'pipe' });
122
+ }
123
+ } catch {}
124
+ try {
125
+ const nets = execSync(cmds.listNetworks, { timeout: 15_000, stdio: 'pipe' })
126
+ .toString()
127
+ .split('\n')
128
+ .filter(Boolean);
129
+ for (const net of nets) {
130
+ try {
131
+ execSync(`docker network rm ${net}`, { timeout: 10_000, stdio: 'pipe' });
132
+ } catch {}
133
+ }
134
+ } catch {}
135
+ // Volumes last: a volume still attached to a container cannot be removed, so
136
+ // this must follow the container sweep above.
137
+ try {
138
+ const vols = execSync(cmds.listVolumes, { timeout: 15_000, stdio: 'pipe' })
139
+ .toString()
140
+ .split('\n')
141
+ .filter(Boolean);
142
+ for (const vol of vols) {
143
+ try {
144
+ execSync(`docker volume rm ${vol}`, { timeout: 10_000, stdio: 'pipe' });
145
+ } catch {}
146
+ }
147
+ } catch {}
148
+ }
149
+
78
150
  function cleanupOnExit() {
79
151
  // `--keep` / `--reuse` mean the operator wants the stack to survive for
80
152
  // debugging — and they want it MOST after a failure, which is exactly the
@@ -82,12 +154,17 @@ function cleanupOnExit() {
82
154
  if (process.env.CELILO_E2E_KEEP === '1' || process.env.CELILO_E2E_REUSE === '1') return;
83
155
 
84
156
  for (const project of activeProjects) {
157
+ // Compose first when it can work — it also drops volumes and orphans — but
158
+ // it is best-effort, so never rely on it having done anything.
85
159
  try {
86
- execSync(`docker compose -p ${project} down --volumes --remove-orphans`, {
160
+ execSync(`docker compose -f ${COMPOSE_FILE} -p ${project} down --volumes --remove-orphans`, {
161
+ cwd: PACKAGE_ROOT,
87
162
  timeout: 30_000,
88
163
  stdio: 'pipe',
89
164
  });
90
165
  } catch {}
166
+ // Authoritative: needs no compose file, no cwd, no working directory state.
167
+ forceRemoveProject(project);
91
168
  }
92
169
  activeProjects.clear();
93
170
  try {
@@ -216,9 +293,7 @@ export async function scrubDnsZones(): Promise<void> {
216
293
  }
217
294
  if (served !== null && !served.split(/\s+/).includes(expected)) {
218
295
  throw new Error(
219
- `[dns-scrub] post-reset verification FAILED: celilo.computer apex should serve ${expected} ` +
220
- `(website-sim) after scrub, but namecheap-dns returns "${served || '(empty)'}". ` +
221
- `The DNS reset did not take effect — shared DNS would bleed across tests.`,
296
+ `[dns-scrub] post-reset verification FAILED: celilo.computer apex should serve ${expected} (website-sim) after scrub, but namecheap-dns returns "${served || '(empty)'}". The DNS reset did not take effect — shared DNS would bleed across tests.`,
222
297
  );
223
298
  }
224
299
  }
@@ -275,9 +350,7 @@ function assertCliVersion(projectName: string, composeDir: string): void {
275
350
  const result = dockerExec(projectName, composeDir, 'management', 'celilo --version', 8_000);
276
351
  if (result.exitCode !== 0) {
277
352
  throw new Error(
278
- `Could not read celilo version from the management image ` +
279
- `(harness requires >=${MIN_CLI_VERSION}). ` +
280
- `\`celilo --version\` exited ${result.exitCode}: ${result.stderr || result.stdout || '(no output)'}`,
353
+ `Could not read celilo version from the management image (harness requires >=${MIN_CLI_VERSION}). \`celilo --version\` exited ${result.exitCode}: ${result.stderr || result.stdout || '(no output)'}`,
281
354
  );
282
355
  }
283
356
  const problem = checkCliVersion(result.stdout);
@@ -509,11 +582,7 @@ function buildNetworkHandle(
509
582
  );
510
583
  // Spawn detached. nohup + & + disown cleanly survives the
511
584
  // dockerExec session ending.
512
- const spawnCmd =
513
- `nohup celilo events respond --values ${valuesPath} ` +
514
- '--idle-timeout 1h --max-duration 2h ' +
515
- '--emittedBy cele2e-responder ' +
516
- '> /tmp/cele2e-responder.log 2>&1 < /dev/null & disown';
585
+ const spawnCmd = `nohup celilo events respond --values ${valuesPath} --idle-timeout 1h --max-duration 2h --emittedBy cele2e-responder > /tmp/cele2e-responder.log 2>&1 < /dev/null & disown`;
517
586
  const spawnResult = dockerExec(projectName, composeDir, 'management', spawnCmd, 5_000);
518
587
  if (spawnResult.exitCode !== 0) {
519
588
  throw new Error(
@@ -717,9 +786,7 @@ export async function startNetwork(config: NetworkConfig): Promise<NetworkHandle
717
786
  if (heavy.length > 0) {
718
787
  const names = heavy.map((c) => ` - ${c}`).join('\n');
719
788
  throw new Error(
720
- `\nCannot start e2e tests: ${heavy.length} live container(s) are running:\n${names}\n\n` +
721
- 'Live and e2e environments are mutually exclusive.\n' +
722
- 'Stop live containers first: cele2e down --all\n',
789
+ `\nCannot start e2e tests: ${heavy.length} live container(s) are running:\n${names}\n\nLive and e2e environments are mutually exclusive.\nStop live containers first: cele2e down --all\n`,
723
790
  );
724
791
  }
725
792
  } catch (e) {
@@ -748,25 +815,33 @@ export async function startNetwork(config: NetworkConfig): Promise<NetworkHandle
748
815
  }
749
816
  } catch {}
750
817
 
751
- // Remove stale per-test networks
818
+ // Remove stale per-test networks AND their volumes, via the same
819
+ // by-name teardown the exit handler uses. This used to lean on
820
+ // `docker compose -p <project> down --volumes` for the volume half, which
821
+ // has no `-f` and no reliable cwd here either — compose exits "no
822
+ // configuration file provided" and the catch swallows it, so volumes were
823
+ // never removed on this path either. Reusing forceRemoveProject keeps the
824
+ // crash-recovery sweep and the exit handler from drifting apart.
752
825
  try {
753
826
  const networks = run('docker network ls --format "{{.Name}}" 2>/dev/null')
754
827
  .split('\n')
755
828
  .filter((n) => n.startsWith('celilo-e2e-1')); // per-test projects start with timestamp
756
- if (networks.length > 0) {
757
- const projects = [...new Set(networks.map((n) => n.replace(/_[^_]+$/, '')))];
758
- for (const project of projects) {
759
- try {
760
- run(`docker compose -p ${project} down --volumes --remove-orphans`, {
761
- timeout: 30_000,
762
- });
763
- } catch {}
764
- }
765
- for (const net of networks) {
766
- try {
767
- run(`docker network rm ${net}`, { timeout: 5_000 });
768
- } catch {}
769
- }
829
+ const projects = [...new Set(networks.map((n) => n.replace(/_[^_]+$/, '')))];
830
+ for (const project of projects) {
831
+ forceRemoveProject(project);
832
+ }
833
+ } catch {}
834
+ // Volumes can outlive every container and network of their project — that
835
+ // is exactly the 58-volume pile — so sweep by prefix too, not only for
836
+ // projects that still have a network to be discovered by.
837
+ try {
838
+ const vols = run('docker volume ls -q --filter name=celilo-e2e-1 2>/dev/null')
839
+ .split('\n')
840
+ .filter(Boolean);
841
+ for (const vol of vols) {
842
+ try {
843
+ run(`docker volume rm ${vol}`, { timeout: 5_000 });
844
+ } catch {}
770
845
  }
771
846
  } catch {}
772
847
  } catch {}
@@ -968,8 +1043,7 @@ export async function startNetwork(config: NetworkConfig): Promise<NetworkHandle
968
1043
  // The control plane needs its subnet declared if celilo-mgr lives
969
1044
  // there OR if a machine does — infrastructure selection for a module
970
1045
  // declaring `zone: secure-mgmt` reads it either way (#436).
971
- config.managementZone === 'secure-mgmt' ||
972
- (config.secureMgmtMachines ?? []).length > 0
1046
+ config.managementZone === 'secure-mgmt' || (config.secureMgmtMachines ?? []).length > 0
973
1047
  ? `network.secure-mgmt.subnet=${ZONE_SUBNETS['secure-mgmt']} network.secure-mgmt.gateway=${ZONE_GATEWAYS['secure-mgmt']} `
974
1048
  : ''
975
1049
  }dns.primary=100.100.0.1 \
@@ -1029,8 +1103,7 @@ export function reconnectNetwork(projectName: string): NetworkHandle {
1029
1103
  }
1030
1104
  } catch (err) {
1031
1105
  throw new Error(
1032
- `Cannot reconnect to network '${projectName}': ${err instanceof Error ? err.message : String(err)}\n` +
1033
- `Fix: Run without --reuse to create a fresh network, or check if containers were torn down.`,
1106
+ `Cannot reconnect to network '${projectName}': ${err instanceof Error ? err.message : String(err)}\nFix: Run without --reuse to create a fresh network, or check if containers were torn down.`,
1034
1107
  );
1035
1108
  }
1036
1109
 
@@ -11,10 +11,11 @@
11
11
  * The leak's mechanism was a wiring bug, so the gate is on the wiring.
12
12
  */
13
13
 
14
+ import { expect, test } from 'bun:test';
14
15
  import { readFileSync } from 'node:fs';
15
16
  import { dirname, join } from 'node:path';
16
17
  import { fileURLToPath } from 'node:url';
17
- import { expect, test } from 'bun:test';
18
+ import { projectTeardownCommands } from './container-manager';
18
19
 
19
20
  const SRC = readFileSync(
20
21
  join(dirname(fileURLToPath(import.meta.url)), 'container-manager.ts'),
@@ -33,12 +34,51 @@ test('cleanup honors --keep / --reuse so a failed run stays debuggable', () => {
33
34
  // an operator asked to keep, since --keep matters most after a failure.
34
35
  const body = SRC.slice(
35
36
  SRC.indexOf('function cleanupOnExit'),
36
- SRC.indexOf('process.on(\'SIGTERM\''),
37
+ SRC.indexOf("process.on('SIGTERM'"),
37
38
  );
38
39
  expect(body).toContain('CELILO_E2E_KEEP');
39
40
  expect(body).toContain('CELILO_E2E_REUSE');
40
41
  });
41
42
 
43
+ // The first version of this fix wired the handler correctly and still leaked,
44
+ // because the teardown it ran was `docker compose -p <project> down` with no
45
+ // `-f` and no cwd. Compose resolves the project from a compose file in the
46
+ // CURRENT DIRECTORY; an exit handler has no reliable cwd, so it exited "no
47
+ // configuration file provided", the surrounding catch swallowed it, and nothing
48
+ // was removed. Asserting the hook exists was not enough — assert it can work.
49
+ test('teardown does not depend on a compose file being findable', () => {
50
+ const cmds = projectTeardownCommands('celilo-e2e-1785546029645');
51
+ for (const cmd of Object.values(cmds)) {
52
+ expect(cmd).not.toContain('docker compose');
53
+ expect(cmd).toContain('celilo-e2e-1785546029645');
54
+ }
55
+ });
56
+
57
+ // Containers and networks were removed; volumes were not. The only thing that
58
+ // would have removed them was `docker compose down --volumes`, which is the
59
+ // same best-effort call that silently does nothing without a resolvable compose
60
+ // file — so in practice they were never removed at all. 58 leaked
61
+ // `celilo-e2e-<ts>_ssh-keys` volumes were found on the builder, the oldest
62
+ // months old. A leaked volume breaks nothing on its own, which is why it went
63
+ // unnoticed for months; the pile is the evidence teardown was not running.
64
+ test('teardown removes volumes, not just containers and networks', () => {
65
+ const cmds = projectTeardownCommands('celilo-e2e-1785546029645');
66
+ expect(cmds.listVolumes).toContain('docker volume ls');
67
+ expect(cmds.listVolumes).toContain('celilo-e2e-1785546029645');
68
+ expect(cmds.listVolumes).not.toContain('docker compose');
69
+ });
70
+
71
+ test('teardown cannot match the shared project', () => {
72
+ // Per-test projects carry a timestamp; `celilo-e2e-shared` does not, so a
73
+ // name filter for one can never match the other. Shared infra must survive —
74
+ // it is torn down separately, and removing it mid-suite breaks every
75
+ // subsequent test.
76
+ const cmds = projectTeardownCommands('celilo-e2e-1785546029645');
77
+ for (const cmd of Object.values(cmds)) {
78
+ expect(cmd).not.toContain('celilo-e2e-shared');
79
+ }
80
+ });
81
+
42
82
  test("the premise holds: process.exit() fires 'exit' but NOT 'beforeExit'", async () => {
43
83
  // The whole fix rests on this runtime behavior. If it ever changes, the
44
84
  // reasoning above is void and this should fail loudly rather than silently