@specific.dev/spectest 0.69.0 → 0.71.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/src/daemon.ts CHANGED
@@ -92,6 +92,7 @@ import {
92
92
  import { pollUntilReady } from "./harness/ready-poll.js";
93
93
  import { runWrapperRules } from "./harness/wrapper-rules.js";
94
94
  import type { WrapperDiagnostic } from "./harness/wrapper-rules.js";
95
+ import { cpus } from "node:os";
95
96
  import { APP_DIR, WORKSPACE, resolveProjectPath } from "./project-files.js";
96
97
  import {
97
98
  isTextualContentType,
@@ -116,6 +117,7 @@ import {
116
117
  certCovers as hostmatchCertCovers,
117
118
  hostWithoutPort,
118
119
  matchRoute,
120
+ selectCertName,
119
121
  wildcardCoversHost,
120
122
  wildcardSuffix,
121
123
  } from "./harness/hostmatch.js";
@@ -123,7 +125,6 @@ import {
123
125
  INGRESS_HTTPS_PORT,
124
126
  INGRESS_HTTP_PORT,
125
127
  bindRoute,
126
- certEntries,
127
128
  clearTables,
128
129
  emptyTables,
129
130
  planBind,
@@ -132,6 +133,7 @@ import {
132
133
  unbindRoute,
133
134
  type Route as IngressRoute,
134
135
  } from "./harness/ingress-table.js";
136
+ import { startTlsTerminator, type TlsTerminator } from "./harness/tls-terminator.js";
135
137
  import { runContainerArgs } from "./harness/container-run.js";
136
138
  import {
137
139
  assertAbsolute,
@@ -239,30 +241,49 @@ const DEFAULT_TEST_TIMEOUT_MS = 60_000;
239
241
  const NETWORK_NAME = process.env.SPECTEST_NETWORK ?? "spectest-net";
240
242
 
241
243
  // Stable hostname every service container resolves to the host (the
242
- // `spectest-br0` gateway) — so apps that build or pull images at runtime
243
- // can point a builder at `spectest-host:5000` (the zot Docker Hub mirror)
244
- // or `spectest-host:1234` (the shared buildkitd) without hard-coding the
245
- // gateway IP. Injected into each container's /etc/hosts in runContainer.
244
+ // `spectest-br0` gateway) — so an app that runs its OWN BuildKit inside a
245
+ // test can point its cache export at the host's build-cache registry,
246
+ // `spectest-host:5007`, without hard-coding the gateway IP. Nothing else
247
+ // lives behind it any more: images are pulled and built inside the VM
248
+ // against the container store (CONTAINER_STORE.md). Injected into each
249
+ // container's /etc/hosts in runContainer.
246
250
  const SPECTEST_HOST_NAME = "spectest-host";
247
251
 
248
- // The host image-cache gateway, discovered once from the same
249
- // `registry-mirrors` entry the in-VM dockerd already uses (baked into the
250
- // local provider's golden /etc/docker/daemon.json). `null` when there's
251
- // no host cache, so nothing is injected.
252
+ // The host gateway, read once from the guest's own default route
253
+ // (/proc/net/route: destination 0, gateway as a little-endian hex word).
254
+ // `null` when the guest has no default route, in which case nothing is
255
+ // injected.
252
256
  let _hostCacheGateway: string | null | undefined;
253
257
  function hostCacheGateway(): string | null {
254
258
  if (_hostCacheGateway !== undefined) return _hostCacheGateway;
259
+ _hostCacheGateway = null;
255
260
  try {
256
- const cfg = JSON.parse(
257
- readFileSync("/etc/docker/daemon.json", "utf8"),
258
- ) as { "registry-mirrors"?: string[] };
259
- const first = cfg["registry-mirrors"]?.[0];
260
- _hostCacheGateway = first ? new URL(first).hostname || null : null;
261
+ for (const line of readFileSync("/proc/net/route", "utf8").split("\n").slice(1)) {
262
+ const f = line.trim().split(/\s+/);
263
+ if (f.length < 3 || f[1] !== "00000000") continue;
264
+ const hex = f[2];
265
+ const octets = [6, 4, 2, 0].map((i) => parseInt(hex.slice(i, i + 2), 16));
266
+ if (octets.every((o) => Number.isFinite(o))) {
267
+ _hostCacheGateway = octets.join(".");
268
+ break;
269
+ }
270
+ }
261
271
  } catch {
262
272
  _hostCacheGateway = null;
263
273
  }
264
274
  return _hostCacheGateway;
265
275
  }
276
+ /** vCPUs this guest has, for BuildKit's `max-parallelism` — the number
277
+ * Blacksmith sets too, and the one that matters now that the build
278
+ * competes for the guest's own cores rather than the host's. */
279
+ function cpuCount(): number {
280
+ try {
281
+ return Math.max(1, cpus().length);
282
+ } catch {
283
+ return 4;
284
+ }
285
+ }
286
+
266
287
  // WORKSPACE (/workspace) and APP_DIR (/opt/spectest/app) both live in
267
288
  // project-files.ts, next to the rule that decides which copy of a project
268
289
  // file is the current one.
@@ -687,46 +708,110 @@ async function hasBuildx(): Promise<boolean> {
687
708
  return _buildxAvailable;
688
709
  }
689
710
 
690
- // A single buildkitd runs on the host (see scripts/install-buildkitd.sh),
691
- // reachable from every VM at the bridge gateway. Building against it as a
692
- // `remote` buildx builder gives a persistent, shared layer/mount cache that
693
- // survives forks and warm-template misses — a fresh VM no longer rebuilds
694
- // from scratch. The build runs on the host (runc-isolated); `--load` pulls
695
- // the finished image back into the in-VM dockerd. Detected once; if the
696
- // builder can't be created or buildkitd is unreachable we fall back to the
697
- // in-VM builder, so a missing/dead buildkitd just means slower builds.
698
- const REMOTE_BUILDER_ADDR = process.env.SPECTEST_BUILDKIT_ADDR ?? "tcp://10.42.0.1:1234";
699
- const REMOTE_BUILDER_NAME = "spectest-remote";
700
- /** Parent of the per-build buildx config dirs (see isolatedBuildxConfig).
701
- * On tmpfs: each holds a builder stub and an 8-byte node id. */
702
- let _remoteBuilder: boolean | undefined;
703
- async function ensureRemoteBuilder(): Promise<boolean> {
704
- if (_remoteBuilder !== undefined) return _remoteBuilder;
705
- if (!(await hasBuildx())) {
706
- _remoteBuilder = false;
711
+ // Every VM carries its project's container store (CONTAINER_STORE.md):
712
+ // the disk every pull and every build lands on, mounted by the control
713
+ // plane before this harness starts. The manifest below says where; the
714
+ // in-VM buildkitd keeps its exported cache there too, so a fresh VM finds
715
+ // every layer it built before. Detected once; if the daemon will not
716
+ // start, dockerd's own BuildKit builds instead.
717
+ const IMAGE_CACHE_MANIFEST = "/run/spectest-image-cache.json";
718
+ const LOCAL_BUILDER_NAME = "spectest-local";
719
+ const LOCAL_BUILDKIT_ADDR = "tcp://127.0.0.1:1234";
720
+ /** The bring-up script the cache base bakes (`base.rs::BUILDKITD_UP_SH`). */
721
+ const BUILDKITD_UP_PATH = "/usr/local/bin/spectest-buildkitd-up";
722
+
723
+ /** Where the control plane mounted this VM's image cache: containerd's
724
+ * root (read-write, this VM's own) and the layers disk (read-only, shared
725
+ * by every VM of a generation). `null` when the VM carries no cache. */
726
+ async function imageCachePaths(): Promise<{ root: string; layers: string } | null> {
727
+ try {
728
+ const raw = await fs.readFile(IMAGE_CACHE_MANIFEST, "utf8");
729
+ const parsed = JSON.parse(raw) as { disks?: { role?: string; path?: string }[] };
730
+ const root = (parsed.disks ?? []).find((d) => d.role === "root" && d.path)?.path;
731
+ const layers = (parsed.disks ?? []).find((d) => d.role === "layers" && d.path)?.path;
732
+ return root && layers ? { root, layers } : null;
733
+ } catch {
734
+ return null;
735
+ }
736
+ }
737
+
738
+ let _localBuilder: boolean | undefined;
739
+ /**
740
+ * Start buildkitd inside this VM with its state on the cache disk, and
741
+ * register it as a buildx `remote` builder.
742
+ *
743
+ * Lazy on purpose: it runs on the first build a project actually does,
744
+ * so a project with no dockerfile service never pays for a builder. And
745
+ * best-effort: a daemon that will not start falls back to the shared host
746
+ * one, which is a slower build and not a failed one.
747
+ */
748
+ async function ensureLocalBuildkitd(): Promise<boolean> {
749
+ if (_localBuilder !== undefined) return _localBuilder;
750
+ _localBuilder = false;
751
+ const paths = await imageCachePaths();
752
+ if (!paths) return false;
753
+ const state = paths.root;
754
+ if (!(await hasBuildx())) return false;
755
+
756
+ // The bring-up itself is a script baked into the image cache's base
757
+ // snapshot (`base.rs::BUILDKITD_UP_SH`), so production and the real-VM
758
+ // test drive exactly the same daemon with exactly the same config. All
759
+ // it takes from us is where the state lives — the registry routing is
760
+ // its own (Docker Hub via `mirror.gcr.io`, deliberately not the host
761
+ // zot instance).
762
+ const started = await shx("/bin/bash", [BUILDKITD_UP_PATH, state], 900_000);
763
+ if (started.code !== 0) {
764
+ // eslint-disable-next-line no-console
765
+ console.warn(
766
+ `[build] in-VM buildkitd would not start; building with dockerd's own BuildKit:\n${(started.stderr || started.stdout).trim()}`,
767
+ );
707
768
  return false;
708
769
  }
709
- // Idempotent: a repeat create with the same name errors ("existing
710
- // instance"), which we treat as already-present.
711
770
  const create = await docker(
712
- ["buildx", "create", "--name", REMOTE_BUILDER_NAME, "--driver", "remote", REMOTE_BUILDER_ADDR],
771
+ ["buildx", "create", "--name", LOCAL_BUILDER_NAME, "--driver", "remote", LOCAL_BUILDKIT_ADDR],
713
772
  30_000,
714
773
  );
715
774
  if (create.code !== 0 && !/existing instance|already exists/i.test(create.stderr)) {
716
- _remoteBuilder = false;
775
+ // eslint-disable-next-line no-console
776
+ console.warn(`[build] could not register the in-VM builder:\n${create.stderr.trim()}`);
717
777
  return false;
718
778
  }
719
- // `inspect --bootstrap` actually dials buildkitd, so it's our reachability
720
- // probe. If buildkitd is down this fails and we fall back.
721
- const boot = await docker(["buildx", "inspect", "--bootstrap", REMOTE_BUILDER_NAME], 60_000);
722
- _remoteBuilder = boot.code === 0;
723
- if (!_remoteBuilder) {
779
+ // `inspect --bootstrap` dials the daemon, so it is both the readiness
780
+ // wait and the proof it is really answering.
781
+ const boot = await docker(["buildx", "inspect", "--bootstrap", LOCAL_BUILDER_NAME], 120_000);
782
+ if (boot.code !== 0) {
724
783
  // eslint-disable-next-line no-console
725
- console.warn(
726
- `[build] remote buildkitd at ${REMOTE_BUILDER_ADDR} unreachable; using in-VM builder:\n${boot.stderr.trim()}`,
727
- );
784
+ console.warn(`[build] in-VM buildkitd never answered; building with dockerd's own BuildKit:\n${boot.stderr.trim()}`);
785
+ return false;
728
786
  }
729
- return _remoteBuilder;
787
+ // eslint-disable-next-line no-console
788
+ console.log(`[build] building in this VM against the image cache (root ${paths.root}, layers ${paths.layers})`);
789
+ // Imports come from the merged cache on the read-only layers disk and
790
+ // from this lineage's own exports on the root; exports go to the root,
791
+ // where the merge picks them up (image_cache/merge.rs).
792
+ _localCacheDir = `${state}/spectest-buildkit-cache`;
793
+ _localCacheImports = [`${paths.layers}/spectest-buildkit-cache`, `${state}/spectest-buildkit-cache`];
794
+ _localBuilder = true;
795
+ return true;
796
+ }
797
+ /** The exported-cache directory on the cache disk, once the in-VM builder is up. */
798
+ let _localCacheDir: string | null = null;
799
+ /** The cache directories a build imports from: the merged one on the
800
+ * layers disk, then this lineage's own exports. */
801
+ let _localCacheImports: string[] = [];
802
+ /**
803
+ * A reference as containerd names it. buildx's `-t` on a remote builder
804
+ * stores an unqualified name (`probe:bx`) that dockerd then cannot
805
+ * resolve (measured, CONTAINER_STORE.md), so the in-VM build names its
806
+ * output in full.
807
+ */
808
+ function qualifyImageRef(ref: string): string {
809
+ const slash = ref.indexOf("/");
810
+ const first = slash < 0 ? "" : ref.slice(0, slash);
811
+ const isRegistry = first.includes(".") || first.includes(":") || first === "localhost";
812
+ if (slash < 0) return `docker.io/library/${ref}`;
813
+ if (!isRegistry) return `docker.io/${ref}`;
814
+ return ref;
730
815
  }
731
816
 
732
817
  async function ensureNetwork(): Promise<void> {
@@ -787,7 +872,12 @@ async function ensureVolumes(svc: NamedService): Promise<string[]> {
787
872
  // created here nor recorded for the delta-restore wipe.
788
873
  if (!(await isExistingNonDirectory(host))) {
789
874
  await fs.mkdir(host, { recursive: true });
790
- if (vol.source?.startsWith("/") && !host.startsWith("/var/cache/spectest/")) {
875
+ // …and neither is the pre-disks cache tree, kept for a server still
876
+ // serving older SDKs. Leaving it out of the manifest is what
877
+ // protects a project running this SDK against a server whose
878
+ // teardown guard predates it.
879
+ const durable = host.startsWith("/var/cache/spectest/");
880
+ if (vol.source?.startsWith("/") && !durable) {
791
881
  await recordAbsoluteVolumeDir(host);
792
882
  }
793
883
  }
@@ -1347,24 +1437,46 @@ async function runServiceBuild(
1347
1437
  // predates per-Dockerfile ignores — and is written ONLY when the
1348
1438
  // project ships none of its own (see readProjectDockerignore).
1349
1439
  await fs.writeFile(`${dfPath}.dockerignore`, serviceDockerignore(image.exclude));
1350
- const useRemote = await ensureRemoteBuilder();
1351
- // Both the remote builder and a local buildx are BuildKit, so both emit
1352
- // per-step timing on stderr under `--progress=plain` (parsed below). Only
1353
- // the legacy in-VM builder takes no progress flag.
1354
- const useBuildKit = useRemote || (await hasBuildx());
1440
+ // The in-VM buildkitd on the container store first: the build runs
1441
+ // inside the guest's own isolation boundary, against this project's
1442
+ // own layer cache, and the finished image is already in the store
1443
+ // dockerd reads. dockerd's built-in BuildKit is the fallback (a guest
1444
+ // with no store, or a daemon that would not start).
1445
+ const useLocal = await ensureLocalBuildkitd();
1446
+ // Both the in-VM daemon and dockerd's buildx are BuildKit, so both
1447
+ // emit per-step timing on stderr under `--progress=plain` (parsed
1448
+ // below). Only the legacy builder takes no progress flag.
1449
+ const useBuildKit = useLocal || (await hasBuildx());
1355
1450
  const buildEnv: Record<string, string> = {};
1356
1451
  let buildArgs: string[];
1357
1452
  // The user's `buildArgs`, as `--build-arg` flags; a plain client flag,
1358
1453
  // so every builder — host buildkitd, in-VM BuildKit, legacy — takes it.
1359
1454
  const argFlags = buildArgFlags(image.buildArgs);
1360
- if (useRemote) {
1361
- // Build on the host-side shared buildkitd (persistent cross-VM cache);
1362
- // `--load` brings the finished image back into the in-VM dockerd so
1363
- // runContainer can `docker run` it. The build context (WORKSPACE, minus
1364
- // .dockerignore) streams to buildkitd over the bridge.
1455
+ if (useLocal && _localCacheDir) {
1456
+ // The in-VM builder is BuildKit's containerd worker on this VM's
1457
+ // own image store (CONTAINER_STORE.md): the output is an image
1458
+ // record in dockerd's namespace, unpacked, so there is no `--load`
1459
+ // and nothing crosses a socket. The cache directory on the image cache
1460
+ // disk is what outlives the VM; `mode=max` keeps every
1461
+ // intermediate layer, uncompressed so an import never inflates,
1462
+ // and one tag per service so exports do not replace each other.
1463
+ const cacheTag = name.replace(/[^a-z0-9-]/gi, "-").toLowerCase();
1365
1464
  buildArgs = [
1366
1465
  "buildx", "build",
1367
- "--builder", REMOTE_BUILDER_NAME,
1466
+ "--builder", LOCAL_BUILDER_NAME,
1467
+ "--progress=plain",
1468
+ "--output", `type=image,name=${qualifyImageRef(tag)},unpack=true`,
1469
+ ..._localCacheImports.flatMap((src) => ["--cache-from", `type=local,src=${src}`]),
1470
+ "--cache-to", `type=local,dest=${_localCacheDir},mode=max,compression=uncompressed,force-compression=true,tag=${cacheTag}`,
1471
+ ...argFlags,
1472
+ "-f", dfPath, WORKSPACE,
1473
+ ];
1474
+ } else if (useLocal) {
1475
+ // The daemon came up but reported no cache directory: build on it
1476
+ // and `--load` the result into dockerd.
1477
+ buildArgs = [
1478
+ "buildx", "build",
1479
+ "--builder", LOCAL_BUILDER_NAME,
1368
1480
  "--load",
1369
1481
  "--progress=plain",
1370
1482
  ...argFlags,
@@ -1673,6 +1785,45 @@ async function waitForReady(svc: NamedService): Promise<void> {
1673
1785
  throw new Error(msg);
1674
1786
  }
1675
1787
 
1788
+ /**
1789
+ * Add the container's own account of its death to an error raised by its
1790
+ * `setup` hook.
1791
+ *
1792
+ * `waitForReady` already does this, because a container that never becomes
1793
+ * ready is obviously the container's fault. A `setup` hook is the case that
1794
+ * was missing, and it is the one that reads most misleadingly: the hook
1795
+ * talks to the service over the network, so when the container dies
1796
+ * mid-hook what surfaces is a name that no longer resolves or a rollout
1797
+ * that never finished — a symptom from the far end of a connection to
1798
+ * something that is not there any more. The reason is in a log that goes
1799
+ * with the VM at teardown.
1800
+ *
1801
+ * Only for a container that is **gone**: one that is still up did not cause
1802
+ * this, and its log would bury the real error. Best-effort throughout — a
1803
+ * diagnostic must never replace the failure it explains.
1804
+ */
1805
+ async function withContainerPostMortem(name: string, err: unknown): Promise<unknown> {
1806
+ try {
1807
+ const state = await docker(
1808
+ ["inspect", "-f", "{{.State.Running}} {{.State.ExitCode}} {{.State.OOMKilled}}", name],
1809
+ 15_000,
1810
+ );
1811
+ const [running, code, oom] = state.stdout.trim().split(/\s+/);
1812
+ if (state.code !== 0 || running !== "false") return err;
1813
+ const logs = await docker(["logs", "--tail=120", name], 30_000);
1814
+ const output = `${logs.stdout}\n${logs.stderr}`.trim();
1815
+ const base = err instanceof Error ? err : new Error(String(err));
1816
+ base.message +=
1817
+ `\n\nThe "${name}" container exited (code ${code}` +
1818
+ `${oom === "true" ? ", OOM-killed" : ""}) while its setup hook was running, ` +
1819
+ `which is why the hook could not reach it.` +
1820
+ (output ? `\nIts last output:\n${output}` : `\nIt logged nothing.`);
1821
+ return base;
1822
+ } catch {
1823
+ return err;
1824
+ }
1825
+ }
1826
+
1676
1827
  /** Validate the `dependsOn` graph and return the name→service map used to
1677
1828
  * walk it. Rules live in `harness/service-graph.ts`. */
1678
1829
  function validateServiceGraph(services: NamedService[]): Map<string, NamedService> {
@@ -1769,42 +1920,29 @@ const INGRESS_HTTP_SERVERS = new Map<number, any>();
1769
1920
  // eslint-disable-next-line @typescript-eslint/no-explicit-any
1770
1921
  const INGRESS_HTTPS_SERVERS = new Map<number, any>();
1771
1922
  /**
1772
- * Servers replaced by a rebind and now draining. On Bun 1.3.14 a request
1773
- * arriving on a kept-alive connection of a `stop(false)`-drained server
1774
- * dispatches into freed per-server state and can SEGFAULT the process
1775
- * (use-after-free class fixed upstream by oven-sh/bun#36790, first in Bun
1776
- * 1.4.0; observed here as `panic: Segmentation fault at address 0xA` in
1777
- * `server.zig onRequestFor` on ~2-3 % of runtime-TLS rebinds). Until the
1778
- * Bun bump lands, shrink the number of requests a drained server can ever
1779
- * see: every response it still serves carries `Connection: close` (one
1780
- * more request per surviving connection, not unlimited), and a grace timer
1781
- * force-closes whatever is left ({@link REBIND_DRAIN_GRACE_MS}).
1923
+ * The :443 TLS terminator, once per harness process.
1924
+ *
1925
+ * There is deliberately no draining machinery beside it any more. :443
1926
+ * used to be a `Bun.serve` that had to be swapped for every new
1927
+ * certificate, and no swap can be made safe: `stop(false)` FINs an idle
1928
+ * kept-alive connection within 1-2 ms, so a client that wrote a request in
1929
+ * that instant failed with `SocketError: other side closed` — in whatever
1930
+ * unrelated test happened to be talking through the ingress.
1931
+ * {@link startTlsTerminator} chooses the leaf per handshake instead, so
1932
+ * the listener is bound once and lives as long as the project does.
1782
1933
  */
1783
- const DRAINING_INGRESS = new WeakSet<object>();
1784
- /** How long a drained listener may keep serving in-flight work before its
1785
- * remaining connections are force-closed. Long enough for a slow proxied
1786
- * response to finish, short enough to bound the 1.3.14 UAF window. */
1787
- const REBIND_DRAIN_GRACE_MS = 15_000;
1788
-
1789
- /** Stamp `Connection: close` on a response served by a draining listener so
1790
- * the kept-alive connection retires instead of lingering as a UAF trigger.
1791
- * Proxied responses can carry immutable headers; rewrap when needed. */
1792
- // eslint-disable-next-line @typescript-eslint/no-explicit-any
1793
- function withDrainClose(server: any, res: Response): Response {
1794
- if (!DRAINING_INGRESS.has(server)) return res;
1795
- try {
1796
- res.headers.set("connection", "close");
1797
- return res;
1798
- } catch {
1799
- const headers = new Headers(res.headers);
1800
- headers.set("connection", "close");
1801
- return new Response(res.body, {
1802
- status: res.status,
1803
- statusText: res.statusText,
1804
- headers,
1805
- });
1806
- }
1807
- }
1934
+ let HTTPS_TERMINATOR: TlsTerminator | undefined;
1935
+ /**
1936
+ * The in-flight first bind of :443, if one is running.
1937
+ *
1938
+ * {@link ensureHttpsIngress} is async where the swap it replaced was
1939
+ * synchronous, and that reintroduces an interleaving the old code could
1940
+ * not have: two `ctx.startService({ tls })` calls issued together by a
1941
+ * fake handler would both find :443 absent and both try to bind it, and
1942
+ * the loser gets EADDRINUSE — a provisioning failure in place of the
1943
+ * dropped connection this change exists to remove. One promise, shared.
1944
+ */
1945
+ let HTTPS_INGRESS_STARTING: Promise<void> | undefined;
1808
1946
  /**
1809
1947
  * The live ingress tables — per-port routes and the :443 SNI cert table.
1810
1948
  *
@@ -1852,6 +1990,12 @@ function stopIngressServers(): void {
1852
1990
  }
1853
1991
  }
1854
1992
  INGRESS_HTTPS_SERVERS.clear();
1993
+ // Force-closed for the same reason the listeners are: the containers
1994
+ // behind these routes are going away, so a connection still riding them
1995
+ // has nothing left to reach.
1996
+ HTTPS_TERMINATOR?.close();
1997
+ HTTPS_TERMINATOR = undefined;
1998
+ HTTPS_INGRESS_STARTING = undefined;
1855
1999
  clearTables(INGRESS);
1856
2000
  }
1857
2001
 
@@ -2089,7 +2233,7 @@ async function startIngress(): Promise<void> {
2089
2233
  console.log(`[ingress] http :${port} for ${[...byHost.keys()].join(", ")}`);
2090
2234
  }
2091
2235
  // ── HTTPS listener on INGRESS_HTTPS_PORT: SNI per certificated hostname.
2092
- if (INGRESS.certByHost.size > 0) rebindHttpsListener(Bun);
2236
+ if (INGRESS.certByHost.size > 0) await ensureHttpsIngress(Bun);
2093
2237
 
2094
2238
  // Seed the resolver's names registry: ingress hostnames (fakes, TLS
2095
2239
  // proxies, dnsName(→ingress)) → bridge gateway, plus ingress-targeted
@@ -2110,86 +2254,85 @@ function requireBun(): any {
2110
2254
  return Bun;
2111
2255
  }
2112
2256
 
2113
- /** Flatten the live SNI cert table into Bun's TLS-entry array. */
2114
- function tlsEntriesFromCerts(): Array<{ cert: string; key: string; serverName: string }> {
2115
- return certEntries(INGRESS);
2116
- }
2117
2257
 
2118
2258
  /**
2119
- * (Re)bind the :443 listener from the current cert table + route map.
2259
+ * Bring :443 up, once.
2120
2260
  *
2121
- * Bun fixes a server's TLS config at `Bun.serve` time — `reload()` accepts a
2122
- * new `tls` option and silently keeps serving the old certificates (measured
2123
- * on Bun 1.3.14: after reloading with a second SNI entry, the new hostname
2124
- * still gets the first one's leaf). So adding a certificate really does mean
2125
- * a second listener.
2261
+ * Two pieces: a plaintext `Bun.serve` on an ephemeral loopback port that
2262
+ * carries the HTTPS route table (and therefore all the routing, fake
2263
+ * dispatch, interception, reverse-proxying and `server.upgrade()`
2264
+ * WebSocket bridging that :80 already gets, unchanged), and the TLS
2265
+ * terminator in front of it on :443.
2126
2266
  *
2127
- * **Bind the new one before stopping the old one.** Doing it the other way
2128
- * round — which is what this used to do, with `stop(true)` — has two teeth:
2129
- * the force-close kills every established connection on :443, and the gap
2130
- * before the new listener binds refuses new ones. Neither is limited to the
2131
- * hostname being added; they hit all the unrelated traffic the listener is
2132
- * carrying. On a project that mints a certificate per provisioned database
2133
- * while deploys stream through the same port, that surfaced as the app under
2134
- * test dying with `SocketError: other side closed` or `ECONNREFUSED
2135
- * <gateway>:443` — a different test each run, and nothing pointing at
2136
- * ingress.
2267
+ * **This is idempotent, and that is the feature.** Adding a certificate
2268
+ * mutates {@link IngressTables.certByHost} and nothing else: the
2269
+ * terminator reads that table on every handshake, so the new hostname is
2270
+ * served by the next connection and no established connection is
2271
+ * disturbed. The listener this replaced had to be stopped and re-served
2272
+ * for each new certificate, and every such swap severed the idle
2273
+ * kept-alive connections it was carrying — see
2274
+ * `harness/tls-terminator.ts` for the measurements and the history.
2137
2275
  *
2138
- * Overlapping the two needs SO_REUSEPORT on both sockets ({@link
2139
- * bindIngressServer} sets it unconditionally for that reason), after which
2140
- * the old listener is drained with `stop(false)` so requests in flight
2141
- * finish. During the overlap the kernel may hand a new connection to either
2142
- * socket, which is safe: the only hostname the two disagree about is the one
2143
- * being added, and it does not resolve until the caller writes the names
2144
- * registry after this returns.
2276
+ * The route `Map` is the persistent module object, so the plaintext
2277
+ * server closes over the same table and later route additions need no
2278
+ * restart either.
2145
2279
  *
2146
- * The route Map is the persistent module object, so the new listener closes
2147
- * over the same table and later route additions need no rebind at all.
2280
+ * The loopback port is ephemeral rather than fixed so it can never
2281
+ * collide with a fake's declared port; it is bound on 127.0.0.1, which no
2282
+ * container can reach (containers arrive at the bridge gateway), so the
2283
+ * only way in is through the terminator.
2148
2284
  */
2149
2285
  // eslint-disable-next-line @typescript-eslint/no-explicit-any
2150
- function rebindHttpsListener(Bun: any): void {
2151
- // The *same* table object the previous listener closed over: the new
2152
- // listener must serve the routes bound since, and any bound later.
2286
+ async function ensureHttpsIngress(Bun: any): Promise<void> {
2287
+ if (INGRESS_HTTPS_SERVERS.has(INGRESS_HTTPS_PORT)) return;
2288
+ if (!HTTPS_INGRESS_STARTING) {
2289
+ // Cleared on both outcomes: once bound, the check above short-circuits
2290
+ // every later caller, and a failed bind must be retryable rather than
2291
+ // remembered as a rejection for the life of the project.
2292
+ HTTPS_INGRESS_STARTING = startHttpsIngress(Bun).finally(() => {
2293
+ HTTPS_INGRESS_STARTING = undefined;
2294
+ });
2295
+ }
2296
+ return HTTPS_INGRESS_STARTING;
2297
+ }
2298
+
2299
+ // eslint-disable-next-line @typescript-eslint/no-explicit-any
2300
+ async function startHttpsIngress(Bun: any): Promise<void> {
2153
2301
  const routes = routesFor(INGRESS, INGRESS_HTTPS_PORT);
2154
- const old = INGRESS_HTTPS_SERVERS.get(INGRESS_HTTPS_PORT);
2155
- const server = bindIngressServer(
2156
- Bun,
2157
- INGRESS_HTTPS_PORT,
2158
- routes,
2159
- `https :${INGRESS_HTTPS_PORT}`,
2160
- tlsEntriesFromCerts(),
2161
- );
2162
- if (old) {
2302
+ // `proto` is passed rather than inferred: this listener speaks plain
2303
+ // HTTP, but everything reaching it arrived over TLS, so X-Forwarded-Proto
2304
+ // must say https.
2305
+ const plain = bindIngressServer(Bun, 0, routes, `https :${INGRESS_HTTPS_PORT}`, {
2306
+ proto: "https",
2307
+ hostname: "127.0.0.1",
2308
+ });
2309
+ INGRESS_HTTPS_SERVERS.set(INGRESS_HTTPS_PORT, plain);
2310
+ try {
2311
+ HTTPS_TERMINATOR = await startTlsTerminator({
2312
+ port: INGRESS_HTTPS_PORT,
2313
+ hostname: "0.0.0.0",
2314
+ upstreamPort: plain.port,
2315
+ certFor: (serverName) => {
2316
+ if (serverName === undefined) return INGRESS.certByHost.values().next().value;
2317
+ const name = selectCertName(INGRESS.certByHost.keys(), serverName.toLowerCase());
2318
+ return name === undefined ? undefined : INGRESS.certByHost.get(name);
2319
+ },
2320
+ });
2321
+ } catch (err) {
2322
+ // Leave nothing half-built: a registered plaintext server with no
2323
+ // terminator would make `planBind` believe :443 is up.
2324
+ INGRESS_HTTPS_SERVERS.delete(INGRESS_HTTPS_PORT);
2163
2325
  try {
2164
- // Graceful: stop accepting, let in-flight requests finish. The old
2165
- // server stays alive until they do, which is the point — a long
2166
- // upload through ingress must not be collateral damage of another
2167
- // service being provisioned.
2168
- old.stop(false);
2169
- // Bun 1.3.14 landmine: a request arriving later on one of the old
2170
- // server's kept-alive connections dispatches into freed state and can
2171
- // segfault the daemon (see {@link DRAINING_INGRESS}). Mark it so any
2172
- // response it still serves closes its connection, and force-close the
2173
- // stragglers once in-flight work has had a fair window to finish.
2174
- DRAINING_INGRESS.add(old);
2175
- const graceTimer = setTimeout(() => {
2176
- try {
2177
- old.stop(true);
2178
- } catch {
2179
- /* already fully stopped */
2180
- }
2181
- }, REBIND_DRAIN_GRACE_MS);
2182
- // Don't let the grace timer keep the process alive on shutdown.
2183
- // eslint-disable-next-line @typescript-eslint/no-explicit-any
2184
- (graceTimer as any).unref?.();
2185
- } catch (err) {
2186
- // eslint-disable-next-line no-console
2187
- console.warn("[ingress] failed to drain the previous https listener:", err);
2326
+ plain.stop(true);
2327
+ } catch {
2328
+ /* already gone */
2188
2329
  }
2330
+ throw err;
2189
2331
  }
2190
- INGRESS_HTTPS_SERVERS.set(INGRESS_HTTPS_PORT, server);
2191
2332
  // eslint-disable-next-line no-console
2192
- console.log(`[ingress] https :${INGRESS_HTTPS_PORT} for ${[...routes.keys()].join(", ")}`);
2333
+ console.log(
2334
+ `[ingress] https :${INGRESS_HTTPS_PORT} (tls terminator -> :${plain.port}) for ${[...routes.keys()].join(", ")}`,
2335
+ );
2193
2336
  }
2194
2337
 
2195
2338
  /**
@@ -2258,9 +2401,12 @@ async function bindRuntimeTls(hostname: string, service: string, port: number):
2258
2401
  );
2259
2402
  }
2260
2403
  if (plan.needsCert) {
2404
+ // The terminator reads this table on every handshake, so the leaf is
2405
+ // live the moment it lands — no listener is touched, and no connection
2406
+ // already on :443 notices.
2261
2407
  INGRESS.certByHost.set(host, await generateHostCert(host, [host]));
2262
2408
  }
2263
- if (plan.needsHttpsRebind) rebindHttpsListener(Bun);
2409
+ if (plan.needsHttpsListener) await ensureHttpsIngress(Bun);
2264
2410
 
2265
2411
  // Resolve the hostname to the daemon gateway (where :443/:80 listen).
2266
2412
  // A wildcard can only live in the resolver's suffix table.
@@ -2312,22 +2458,27 @@ function bindIngressServer(
2312
2458
  port: number,
2313
2459
  byHost: Map<string, Route>,
2314
2460
  listenerLabel: string,
2315
- tlsEntries?: Array<{ cert: string; key: string; serverName: string }>,
2461
+ // Every ingress listener speaks plain HTTP now — the one behind :443
2462
+ // sits under the TLS terminator. So the scheme a client actually used
2463
+ // can no longer be inferred from the socket and is declared instead; it
2464
+ // stamps X-Forwarded-Proto, which upstreams that build absolute URLs or
2465
+ // redirect depend on.
2466
+ serve: { proto: "http" | "https"; hostname?: string } = { proto: "http" },
2316
2467
  // eslint-disable-next-line @typescript-eslint/no-explicit-any
2317
2468
  ): any {
2318
- // A TLS listener terminates https; everything else is plain http. Used
2319
- // to stamp X-Forwarded-Proto so upstreams that build absolute URLs or
2320
- // redirect see the scheme the client actually used, not our http hop.
2321
- const proto = tlsEntries ? "https" : "http";
2469
+ const proto = serve.proto;
2322
2470
  // eslint-disable-next-line @typescript-eslint/no-explicit-any
2323
2471
  const opts: Record<string, any> = {
2324
2472
  port,
2325
- hostname: "0.0.0.0",
2326
- // SO_REUSEPORT on every ingress listener, so a replacement can be bound
2327
- // while the old one is still serving. That overlap is the only way to
2328
- // add an SNI certificate without a gap — see {@link rebindHttpsListener}
2329
- // — and it only works if *both* sockets opt in: a second plain bind
2330
- // fails with "Is port 443 in use?".
2473
+ hostname: serve.hostname ?? "0.0.0.0",
2474
+ // SO_REUSEPORT on every ingress listener. It was introduced so a
2475
+ // replacement :443 could overlap the original during a certificate
2476
+ // swap; the terminator removed that swap, and this stays for the
2477
+ // remaining case — the `/load` teardown force-closes and rebinds, and
2478
+ // a socket the kernel has not finished releasing would otherwise fail
2479
+ // the next bind with "Is port 443 in use?". Note the flip side, which
2480
+ // {@link stopIngressServers} depends on: a half-alive listener shares
2481
+ // the port silently instead of colliding loudly.
2331
2482
  reusePort: true,
2332
2483
  // Bun.serve defaults to a 10s idleTimeout, which kills any proxied
2333
2484
  // request whose upstream takes >10s to produce bytes — under parallel
@@ -2340,9 +2491,7 @@ function bindIngressServer(
2340
2491
  idleTimeout: 0,
2341
2492
  // eslint-disable-next-line @typescript-eslint/no-explicit-any
2342
2493
  fetch: (req: Request, server: any): Response | Promise<Response> =>
2343
- dispatchIngress(req, server, byHost, listenerLabel, proto).then((res) =>
2344
- withDrainClose(server, res),
2345
- ),
2494
+ dispatchIngress(req, server, byHost, listenerLabel, proto),
2346
2495
  websocket: {
2347
2496
  // eslint-disable-next-line @typescript-eslint/no-explicit-any
2348
2497
  async open(ws: any) {
@@ -2412,7 +2561,6 @@ function bindIngressServer(
2412
2561
  },
2413
2562
  },
2414
2563
  };
2415
- if (tlsEntries) opts.tls = tlsEntries;
2416
2564
  return Bun.serve(opts);
2417
2565
  }
2418
2566
 
@@ -2489,6 +2637,27 @@ async function dispatchIngress(
2489
2637
  return augmentCorsResponse(req, res);
2490
2638
  }
2491
2639
 
2640
+ /**
2641
+ * The address to report as the client's in `X-Forwarded-For`.
2642
+ *
2643
+ * On :80 that is simply the socket's peer. On :443 the peer is the TLS
2644
+ * terminator on loopback, so the real address has to be recovered from it:
2645
+ * the terminator keys `source port -> client address` for the life of each
2646
+ * upstream connection, and the source port is what `requestIP` reports
2647
+ * here. Gated on `proto === "https"` so a genuinely loopback caller on :80
2648
+ * can never pick up an unrelated terminator connection's port.
2649
+ *
2650
+ * Falling back to the socket's own address means the worst case is the
2651
+ * pre-terminator answer for a plain HTTP hop, never a wrong tenant.
2652
+ */
2653
+ // eslint-disable-next-line @typescript-eslint/no-explicit-any
2654
+ function ingressClientIp(server: any, req: Request, proto: string): string | undefined {
2655
+ const peer = server.requestIP?.(req) as { address?: string; port?: number } | undefined;
2656
+ if (!peer?.address) return undefined;
2657
+ if (proto !== "https") return peer.address;
2658
+ return HTTPS_TERMINATOR?.clientIpFor(peer.port) ?? peer.address;
2659
+ }
2660
+
2492
2661
  /**
2493
2662
  * Reverse-proxy a request to `http://<service>:<port>` on
2494
2663
  * `spectest-net`. Handles plain HTTP/1.1 + 2 and WebSocket upgrades:
@@ -2548,7 +2717,7 @@ async function proxyToService(
2548
2717
  // Standard reverse-proxy provenance headers: the upstream sees the
2549
2718
  // public scheme/host it was reached through and the client's address,
2550
2719
  // even though we rewrite Host below to the service-net name.
2551
- const clientIp = server.requestIP?.(req)?.address as string | undefined;
2720
+ const clientIp = ingressClientIp(server, req, proto);
2552
2721
  const priorXff = req.headers.get("x-forwarded-for");
2553
2722
  const xff = clientIp ? (priorXff ? `${priorXff}, ${clientIp}` : clientIp) : priorXff;
2554
2723
  if (xff) fwdHeaders.set("x-forwarded-for", xff);
@@ -3429,18 +3598,14 @@ async function bootstrapInner(): Promise<BootstrapTimings> {
3429
3598
  // at "image ready" waiting for an unrelated slow build elsewhere.
3430
3599
  //
3431
3600
  // Prep concurrency: registry pulls always run in parallel (network-bound,
3432
- // low VM RAM). Dockerfile builds parallelize *only* when the host
3433
- // buildkitd is in play — there the build executes host-side under runc, so
3434
- // N concurrent builds don't touch the VM's memory ceiling. When we fall
3435
- // back to the in-VM builder, two or more concurrent builds routinely OOM a
3436
- // single VM on monorepos with parallel pnpm/npm installs (each install
3437
- // fans out to ~16 fetchers + lifecycle workers, ~70 MB/process), so we
3438
- // serialize that case behind a FIFO chain — but only the in-VM builds
3439
- // serialize; pulls and starts run freely alongside them. The remote-builder
3440
- // probe is memoized, so this up-front call is free; skip it with no builds.
3601
+ // low VM RAM). Dockerfile builds run inside the VM, and two or more
3602
+ // concurrent builds routinely OOM a single VM on monorepos with parallel
3603
+ // pnpm/npm installs (each install fans out to ~16 fetchers + lifecycle
3604
+ // workers, ~70 MB/process), so builds serialize behind a FIFO chain —
3605
+ // but only the builds; pulls and starts run freely alongside them.
3441
3606
  const tags = new Map<string, string>();
3442
3607
  const builds = services.filter((s) => s.image.type === "dockerfile");
3443
- const buildsRunHostSide = builds.length > 0 && (await ensureRemoteBuilder());
3608
+ const buildsRunHostSide = false;
3444
3609
  // A promise chain is a fair FIFO mutex: when builds run in-VM, each build
3445
3610
  // waits for the previous to settle. Pulls and host-side builds bypass it.
3446
3611
  let inVmBuildChain: Promise<unknown> = Promise.resolve();
@@ -3527,11 +3692,15 @@ async function bootstrapInner(): Promise<BootstrapTimings> {
3527
3692
  if (svc.setup) {
3528
3693
  progressService(svc.name, { status: "probing", detail: "running setup" });
3529
3694
  const helpers = await ensureHelpers(svc.name, svc);
3530
- await svc.setup({
3531
- name: svc.name,
3532
- helpers,
3533
- ...(await spectestContext({ service: svc.name, includeSelf: true })),
3534
- });
3695
+ try {
3696
+ await svc.setup({
3697
+ name: svc.name,
3698
+ helpers,
3699
+ ...(await spectestContext({ service: svc.name, includeSelf: true })),
3700
+ });
3701
+ } catch (err) {
3702
+ throw await withContainerPostMortem(svc.name, err);
3703
+ }
3535
3704
  }
3536
3705
  progressService(svc.name, { status: "ready", detail: undefined });
3537
3706
  const ti = timings.get(svc.name);