@specific.dev/spectest 0.69.0 → 0.71.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/daemon.js CHANGED
@@ -40,13 +40,15 @@ import { LOG_DELTA_MAX_BYTES, capMiddle, streamDelta } from "./harness/log-delta
40
40
  import { resolveHostPath as resolveVolumeHostPath, sanitizeSegment, } from "./harness/volume-paths.js";
41
41
  import { pollUntilReady } from "./harness/ready-poll.js";
42
42
  import { runWrapperRules } from "./harness/wrapper-rules.js";
43
+ import { cpus } from "node:os";
43
44
  import { APP_DIR, WORKSPACE, resolveProjectPath } from "./project-files.js";
44
45
  import { isTextualContentType, looksBinary, omittedBody, parseContentLength, } from "./harness/http-body.js";
45
46
  import { encodeRegistry } from "./harness/names-registry.js";
46
47
  import { InterceptRegistry, parseTarget, runChain, } from "./harness/intercept.js";
47
48
  import { HOP_BY_HOP_HEADERS, augmentCorsResponse, corsPreflightResponse, isCorsPreflight, } from "./harness/http-proxy.js";
48
- import { certCovers as hostmatchCertCovers, hostWithoutPort, matchRoute, wildcardSuffix, } from "./harness/hostmatch.js";
49
- import { INGRESS_HTTPS_PORT, INGRESS_HTTP_PORT, bindRoute, certEntries, clearTables, emptyTables, planBind, registryTarget, routesFor, unbindRoute, } from "./harness/ingress-table.js";
49
+ import { certCovers as hostmatchCertCovers, hostWithoutPort, matchRoute, selectCertName, wildcardSuffix, } from "./harness/hostmatch.js";
50
+ import { INGRESS_HTTPS_PORT, INGRESS_HTTP_PORT, bindRoute, clearTables, emptyTables, planBind, registryTarget, routesFor, unbindRoute, } from "./harness/ingress-table.js";
51
+ import { startTlsTerminator } from "./harness/tls-terminator.js";
50
52
  import { runContainerArgs } from "./harness/container-run.js";
51
53
  import { assertAbsolute, certificateHostnames, defaultKeyMode, expandServiceToken, isNoopChown, mountFlag, needsIdTables, numericId, resolveChownIds, } from "./harness/file-mounts.js";
52
54
  import { conflict, notFound, requireString, } from "./harness/methods.js";
@@ -66,29 +68,51 @@ function namedServices(cfg) {
66
68
  const DEFAULT_TEST_TIMEOUT_MS = 60_000;
67
69
  const NETWORK_NAME = process.env.SPECTEST_NETWORK ?? "spectest-net";
68
70
  // Stable hostname every service container resolves to the host (the
69
- // `spectest-br0` gateway) — so apps that build or pull images at runtime
70
- // can point a builder at `spectest-host:5000` (the zot Docker Hub mirror)
71
- // or `spectest-host:1234` (the shared buildkitd) without hard-coding the
72
- // gateway IP. Injected into each container's /etc/hosts in runContainer.
71
+ // `spectest-br0` gateway) — so an app that runs its OWN BuildKit inside a
72
+ // test can point its cache export at the host's build-cache registry,
73
+ // `spectest-host:5007`, without hard-coding the gateway IP. Nothing else
74
+ // lives behind it any more: images are pulled and built inside the VM
75
+ // against the container store (CONTAINER_STORE.md). Injected into each
76
+ // container's /etc/hosts in runContainer.
73
77
  const SPECTEST_HOST_NAME = "spectest-host";
74
- // The host image-cache gateway, discovered once from the same
75
- // `registry-mirrors` entry the in-VM dockerd already uses (baked into the
76
- // local provider's golden /etc/docker/daemon.json). `null` when there's
77
- // no host cache, so nothing is injected.
78
+ // The host gateway, read once from the guest's own default route
79
+ // (/proc/net/route: destination 0, gateway as a little-endian hex word).
80
+ // `null` when the guest has no default route, in which case nothing is
81
+ // injected.
78
82
  let _hostCacheGateway;
79
83
  function hostCacheGateway() {
80
84
  if (_hostCacheGateway !== undefined)
81
85
  return _hostCacheGateway;
86
+ _hostCacheGateway = null;
82
87
  try {
83
- const cfg = JSON.parse(readFileSync("/etc/docker/daemon.json", "utf8"));
84
- const first = cfg["registry-mirrors"]?.[0];
85
- _hostCacheGateway = first ? new URL(first).hostname || null : null;
88
+ for (const line of readFileSync("/proc/net/route", "utf8").split("\n").slice(1)) {
89
+ const f = line.trim().split(/\s+/);
90
+ if (f.length < 3 || f[1] !== "00000000")
91
+ continue;
92
+ const hex = f[2];
93
+ const octets = [6, 4, 2, 0].map((i) => parseInt(hex.slice(i, i + 2), 16));
94
+ if (octets.every((o) => Number.isFinite(o))) {
95
+ _hostCacheGateway = octets.join(".");
96
+ break;
97
+ }
98
+ }
86
99
  }
87
100
  catch {
88
101
  _hostCacheGateway = null;
89
102
  }
90
103
  return _hostCacheGateway;
91
104
  }
105
+ /** vCPUs this guest has, for BuildKit's `max-parallelism` — the number
106
+ * Blacksmith sets too, and the one that matters now that the build
107
+ * competes for the guest's own cores rather than the host's. */
108
+ function cpuCount() {
109
+ try {
110
+ return Math.max(1, cpus().length);
111
+ }
112
+ catch {
113
+ return 4;
114
+ }
115
+ }
92
116
  // WORKSPACE (/workspace) and APP_DIR (/opt/spectest/app) both live in
93
117
  // project-files.ts, next to the rule that decides which copy of a project
94
118
  // file is the current one.
@@ -418,42 +442,108 @@ async function hasBuildx() {
418
442
  }
419
443
  return _buildxAvailable;
420
444
  }
421
- // A single buildkitd runs on the host (see scripts/install-buildkitd.sh),
422
- // reachable from every VM at the bridge gateway. Building against it as a
423
- // `remote` buildx builder gives a persistent, shared layer/mount cache that
424
- // survives forks and warm-template misses — a fresh VM no longer rebuilds
425
- // from scratch. The build runs on the host (runc-isolated); `--load` pulls
426
- // the finished image back into the in-VM dockerd. Detected once; if the
427
- // builder can't be created or buildkitd is unreachable we fall back to the
428
- // in-VM builder, so a missing/dead buildkitd just means slower builds.
429
- const REMOTE_BUILDER_ADDR = process.env.SPECTEST_BUILDKIT_ADDR ?? "tcp://10.42.0.1:1234";
430
- const REMOTE_BUILDER_NAME = "spectest-remote";
431
- /** Parent of the per-build buildx config dirs (see isolatedBuildxConfig).
432
- * On tmpfs: each holds a builder stub and an 8-byte node id. */
433
- let _remoteBuilder;
434
- async function ensureRemoteBuilder() {
435
- if (_remoteBuilder !== undefined)
436
- return _remoteBuilder;
437
- if (!(await hasBuildx())) {
438
- _remoteBuilder = false;
445
+ // Every VM carries its project's container store (CONTAINER_STORE.md):
446
+ // the disk every pull and every build lands on, mounted by the control
447
+ // plane before this harness starts. The manifest below says where; the
448
+ // in-VM buildkitd keeps its exported cache there too, so a fresh VM finds
449
+ // every layer it built before. Detected once; if the daemon will not
450
+ // start, dockerd's own BuildKit builds instead.
451
+ const IMAGE_CACHE_MANIFEST = "/run/spectest-image-cache.json";
452
+ const LOCAL_BUILDER_NAME = "spectest-local";
453
+ const LOCAL_BUILDKIT_ADDR = "tcp://127.0.0.1:1234";
454
+ /** The bring-up script the cache base bakes (`base.rs::BUILDKITD_UP_SH`). */
455
+ const BUILDKITD_UP_PATH = "/usr/local/bin/spectest-buildkitd-up";
456
+ /** Where the control plane mounted this VM's image cache: containerd's
457
+ * root (read-write, this VM's own) and the layers disk (read-only, shared
458
+ * by every VM of a generation). `null` when the VM carries no cache. */
459
+ async function imageCachePaths() {
460
+ try {
461
+ const raw = await fs.readFile(IMAGE_CACHE_MANIFEST, "utf8");
462
+ const parsed = JSON.parse(raw);
463
+ const root = (parsed.disks ?? []).find((d) => d.role === "root" && d.path)?.path;
464
+ const layers = (parsed.disks ?? []).find((d) => d.role === "layers" && d.path)?.path;
465
+ return root && layers ? { root, layers } : null;
466
+ }
467
+ catch {
468
+ return null;
469
+ }
470
+ }
471
+ let _localBuilder;
472
+ /**
473
+ * Start buildkitd inside this VM with its state on the cache disk, and
474
+ * register it as a buildx `remote` builder.
475
+ *
476
+ * Lazy on purpose: it runs on the first build a project actually does,
477
+ * so a project with no dockerfile service never pays for a builder. And
478
+ * best-effort: a daemon that will not start falls back to the shared host
479
+ * one, which is a slower build and not a failed one.
480
+ */
481
+ async function ensureLocalBuildkitd() {
482
+ if (_localBuilder !== undefined)
483
+ return _localBuilder;
484
+ _localBuilder = false;
485
+ const paths = await imageCachePaths();
486
+ if (!paths)
487
+ return false;
488
+ const state = paths.root;
489
+ if (!(await hasBuildx()))
490
+ return false;
491
+ // The bring-up itself is a script baked into the image cache's base
492
+ // snapshot (`base.rs::BUILDKITD_UP_SH`), so production and the real-VM
493
+ // test drive exactly the same daemon with exactly the same config. All
494
+ // it takes from us is where the state lives — the registry routing is
495
+ // its own (Docker Hub via `mirror.gcr.io`, deliberately not the host
496
+ // zot instance).
497
+ const started = await shx("/bin/bash", [BUILDKITD_UP_PATH, state], 900_000);
498
+ if (started.code !== 0) {
499
+ // eslint-disable-next-line no-console
500
+ console.warn(`[build] in-VM buildkitd would not start; building with dockerd's own BuildKit:\n${(started.stderr || started.stdout).trim()}`);
439
501
  return false;
440
502
  }
441
- // Idempotent: a repeat create with the same name errors ("existing
442
- // instance"), which we treat as already-present.
443
- const create = await docker(["buildx", "create", "--name", REMOTE_BUILDER_NAME, "--driver", "remote", REMOTE_BUILDER_ADDR], 30_000);
503
+ const create = await docker(["buildx", "create", "--name", LOCAL_BUILDER_NAME, "--driver", "remote", LOCAL_BUILDKIT_ADDR], 30_000);
444
504
  if (create.code !== 0 && !/existing instance|already exists/i.test(create.stderr)) {
445
- _remoteBuilder = false;
505
+ // eslint-disable-next-line no-console
506
+ console.warn(`[build] could not register the in-VM builder:\n${create.stderr.trim()}`);
446
507
  return false;
447
508
  }
448
- // `inspect --bootstrap` actually dials buildkitd, so it's our reachability
449
- // probe. If buildkitd is down this fails and we fall back.
450
- const boot = await docker(["buildx", "inspect", "--bootstrap", REMOTE_BUILDER_NAME], 60_000);
451
- _remoteBuilder = boot.code === 0;
452
- if (!_remoteBuilder) {
509
+ // `inspect --bootstrap` dials the daemon, so it is both the readiness
510
+ // wait and the proof it is really answering.
511
+ const boot = await docker(["buildx", "inspect", "--bootstrap", LOCAL_BUILDER_NAME], 120_000);
512
+ if (boot.code !== 0) {
453
513
  // eslint-disable-next-line no-console
454
- console.warn(`[build] remote buildkitd at ${REMOTE_BUILDER_ADDR} unreachable; using in-VM builder:\n${boot.stderr.trim()}`);
514
+ console.warn(`[build] in-VM buildkitd never answered; building with dockerd's own BuildKit:\n${boot.stderr.trim()}`);
515
+ return false;
455
516
  }
456
- return _remoteBuilder;
517
+ // eslint-disable-next-line no-console
518
+ console.log(`[build] building in this VM against the image cache (root ${paths.root}, layers ${paths.layers})`);
519
+ // Imports come from the merged cache on the read-only layers disk and
520
+ // from this lineage's own exports on the root; exports go to the root,
521
+ // where the merge picks them up (image_cache/merge.rs).
522
+ _localCacheDir = `${state}/spectest-buildkit-cache`;
523
+ _localCacheImports = [`${paths.layers}/spectest-buildkit-cache`, `${state}/spectest-buildkit-cache`];
524
+ _localBuilder = true;
525
+ return true;
526
+ }
527
+ /** The exported-cache directory on the cache disk, once the in-VM builder is up. */
528
+ let _localCacheDir = null;
529
+ /** The cache directories a build imports from: the merged one on the
530
+ * layers disk, then this lineage's own exports. */
531
+ let _localCacheImports = [];
532
+ /**
533
+ * A reference as containerd names it. buildx's `-t` on a remote builder
534
+ * stores an unqualified name (`probe:bx`) that dockerd then cannot
535
+ * resolve (measured, CONTAINER_STORE.md), so the in-VM build names its
536
+ * output in full.
537
+ */
538
+ function qualifyImageRef(ref) {
539
+ const slash = ref.indexOf("/");
540
+ const first = slash < 0 ? "" : ref.slice(0, slash);
541
+ const isRegistry = first.includes(".") || first.includes(":") || first === "localhost";
542
+ if (slash < 0)
543
+ return `docker.io/library/${ref}`;
544
+ if (!isRegistry)
545
+ return `docker.io/${ref}`;
546
+ return ref;
457
547
  }
458
548
  async function ensureNetwork() {
459
549
  const inspect = await docker(["network", "inspect", NETWORK_NAME], 30_000);
@@ -508,7 +598,12 @@ async function ensureVolumes(svc) {
508
598
  // created here nor recorded for the delta-restore wipe.
509
599
  if (!(await isExistingNonDirectory(host))) {
510
600
  await fs.mkdir(host, { recursive: true });
511
- if (vol.source?.startsWith("/") && !host.startsWith("/var/cache/spectest/")) {
601
+ // …and neither is the pre-disks cache tree, kept for a server still
602
+ // serving older SDKs. Leaving it out of the manifest is what
603
+ // protects a project running this SDK against a server whose
604
+ // teardown guard predates it.
605
+ const durable = host.startsWith("/var/cache/spectest/");
606
+ if (vol.source?.startsWith("/") && !durable) {
512
607
  await recordAbsoluteVolumeDir(host);
513
608
  }
514
609
  }
@@ -1002,24 +1097,47 @@ async function runServiceBuild(name, image, tag, caSuffix) {
1002
1097
  // predates per-Dockerfile ignores — and is written ONLY when the
1003
1098
  // project ships none of its own (see readProjectDockerignore).
1004
1099
  await fs.writeFile(`${dfPath}.dockerignore`, serviceDockerignore(image.exclude));
1005
- const useRemote = await ensureRemoteBuilder();
1006
- // Both the remote builder and a local buildx are BuildKit, so both emit
1007
- // per-step timing on stderr under `--progress=plain` (parsed below). Only
1008
- // the legacy in-VM builder takes no progress flag.
1009
- const useBuildKit = useRemote || (await hasBuildx());
1100
+ // The in-VM buildkitd on the container store first: the build runs
1101
+ // inside the guest's own isolation boundary, against this project's
1102
+ // own layer cache, and the finished image is already in the store
1103
+ // dockerd reads. dockerd's built-in BuildKit is the fallback (a guest
1104
+ // with no store, or a daemon that would not start).
1105
+ const useLocal = await ensureLocalBuildkitd();
1106
+ // Both the in-VM daemon and dockerd's buildx are BuildKit, so both
1107
+ // emit per-step timing on stderr under `--progress=plain` (parsed
1108
+ // below). Only the legacy builder takes no progress flag.
1109
+ const useBuildKit = useLocal || (await hasBuildx());
1010
1110
  const buildEnv = {};
1011
1111
  let buildArgs;
1012
1112
  // The user's `buildArgs`, as `--build-arg` flags; a plain client flag,
1013
1113
  // so every builder — host buildkitd, in-VM BuildKit, legacy — takes it.
1014
1114
  const argFlags = buildArgFlags(image.buildArgs);
1015
- if (useRemote) {
1016
- // Build on the host-side shared buildkitd (persistent cross-VM cache);
1017
- // `--load` brings the finished image back into the in-VM dockerd so
1018
- // runContainer can `docker run` it. The build context (WORKSPACE, minus
1019
- // .dockerignore) streams to buildkitd over the bridge.
1115
+ if (useLocal && _localCacheDir) {
1116
+ // The in-VM builder is BuildKit's containerd worker on this VM's
1117
+ // own image store (CONTAINER_STORE.md): the output is an image
1118
+ // record in dockerd's namespace, unpacked, so there is no `--load`
1119
+ // and nothing crosses a socket. The cache directory on the image cache
1120
+ // disk is what outlives the VM; `mode=max` keeps every
1121
+ // intermediate layer, uncompressed so an import never inflates,
1122
+ // and one tag per service so exports do not replace each other.
1123
+ const cacheTag = name.replace(/[^a-z0-9-]/gi, "-").toLowerCase();
1020
1124
  buildArgs = [
1021
1125
  "buildx", "build",
1022
- "--builder", REMOTE_BUILDER_NAME,
1126
+ "--builder", LOCAL_BUILDER_NAME,
1127
+ "--progress=plain",
1128
+ "--output", `type=image,name=${qualifyImageRef(tag)},unpack=true`,
1129
+ ..._localCacheImports.flatMap((src) => ["--cache-from", `type=local,src=${src}`]),
1130
+ "--cache-to", `type=local,dest=${_localCacheDir},mode=max,compression=uncompressed,force-compression=true,tag=${cacheTag}`,
1131
+ ...argFlags,
1132
+ "-f", dfPath, WORKSPACE,
1133
+ ];
1134
+ }
1135
+ else if (useLocal) {
1136
+ // The daemon came up but reported no cache directory: build on it
1137
+ // and `--load` the result into dockerd.
1138
+ buildArgs = [
1139
+ "buildx", "build",
1140
+ "--builder", LOCAL_BUILDER_NAME,
1023
1141
  "--load",
1024
1142
  "--progress=plain",
1025
1143
  ...argFlags,
@@ -1286,6 +1404,43 @@ async function waitForReady(svc) {
1286
1404
  msg += output ? `\nRecent container logs:\n${output}` : `\n(the container logged nothing)`;
1287
1405
  throw new Error(msg);
1288
1406
  }
1407
+ /**
1408
+ * Add the container's own account of its death to an error raised by its
1409
+ * `setup` hook.
1410
+ *
1411
+ * `waitForReady` already does this, because a container that never becomes
1412
+ * ready is obviously the container's fault. A `setup` hook is the case that
1413
+ * was missing, and it is the one that reads most misleadingly: the hook
1414
+ * talks to the service over the network, so when the container dies
1415
+ * mid-hook what surfaces is a name that no longer resolves or a rollout
1416
+ * that never finished — a symptom from the far end of a connection to
1417
+ * something that is not there any more. The reason is in a log that goes
1418
+ * with the VM at teardown.
1419
+ *
1420
+ * Only for a container that is **gone**: one that is still up did not cause
1421
+ * this, and its log would bury the real error. Best-effort throughout — a
1422
+ * diagnostic must never replace the failure it explains.
1423
+ */
1424
+ async function withContainerPostMortem(name, err) {
1425
+ try {
1426
+ const state = await docker(["inspect", "-f", "{{.State.Running}} {{.State.ExitCode}} {{.State.OOMKilled}}", name], 15_000);
1427
+ const [running, code, oom] = state.stdout.trim().split(/\s+/);
1428
+ if (state.code !== 0 || running !== "false")
1429
+ return err;
1430
+ const logs = await docker(["logs", "--tail=120", name], 30_000);
1431
+ const output = `${logs.stdout}\n${logs.stderr}`.trim();
1432
+ const base = err instanceof Error ? err : new Error(String(err));
1433
+ base.message +=
1434
+ `\n\nThe "${name}" container exited (code ${code}` +
1435
+ `${oom === "true" ? ", OOM-killed" : ""}) while its setup hook was running, ` +
1436
+ `which is why the hook could not reach it.` +
1437
+ (output ? `\nIts last output:\n${output}` : `\nIt logged nothing.`);
1438
+ return base;
1439
+ }
1440
+ catch {
1441
+ return err;
1442
+ }
1443
+ }
1289
1444
  /** Validate the `dependsOn` graph and return the name→service map used to
1290
1445
  * walk it. Rules live in `harness/service-graph.ts`. */
1291
1446
  function validateServiceGraph(services) {
@@ -1356,43 +1511,29 @@ const INGRESS_HTTP_SERVERS = new Map();
1356
1511
  // eslint-disable-next-line @typescript-eslint/no-explicit-any
1357
1512
  const INGRESS_HTTPS_SERVERS = new Map();
1358
1513
  /**
1359
- * Servers replaced by a rebind and now draining. On Bun 1.3.14 a request
1360
- * arriving on a kept-alive connection of a `stop(false)`-drained server
1361
- * dispatches into freed per-server state and can SEGFAULT the process
1362
- * (use-after-free class fixed upstream by oven-sh/bun#36790, first in Bun
1363
- * 1.4.0; observed here as `panic: Segmentation fault at address 0xA` in
1364
- * `server.zig onRequestFor` on ~2-3 % of runtime-TLS rebinds). Until the
1365
- * Bun bump lands, shrink the number of requests a drained server can ever
1366
- * see: every response it still serves carries `Connection: close` (one
1367
- * more request per surviving connection, not unlimited), and a grace timer
1368
- * force-closes whatever is left ({@link REBIND_DRAIN_GRACE_MS}).
1514
+ * The :443 TLS terminator, once per harness process.
1515
+ *
1516
+ * There is deliberately no draining machinery beside it any more. :443
1517
+ * used to be a `Bun.serve` that had to be swapped for every new
1518
+ * certificate, and no swap can be made safe: `stop(false)` FINs an idle
1519
+ * kept-alive connection within 1-2 ms, so a client that wrote a request in
1520
+ * that instant failed with `SocketError: other side closed` — in whatever
1521
+ * unrelated test happened to be talking through the ingress.
1522
+ * {@link startTlsTerminator} chooses the leaf per handshake instead, so
1523
+ * the listener is bound once and lives as long as the project does.
1369
1524
  */
1370
- const DRAINING_INGRESS = new WeakSet();
1371
- /** How long a drained listener may keep serving in-flight work before its
1372
- * remaining connections are force-closed. Long enough for a slow proxied
1373
- * response to finish, short enough to bound the 1.3.14 UAF window. */
1374
- const REBIND_DRAIN_GRACE_MS = 15_000;
1375
- /** Stamp `Connection: close` on a response served by a draining listener so
1376
- * the kept-alive connection retires instead of lingering as a UAF trigger.
1377
- * Proxied responses can carry immutable headers; rewrap when needed. */
1378
- // eslint-disable-next-line @typescript-eslint/no-explicit-any
1379
- function withDrainClose(server, res) {
1380
- if (!DRAINING_INGRESS.has(server))
1381
- return res;
1382
- try {
1383
- res.headers.set("connection", "close");
1384
- return res;
1385
- }
1386
- catch {
1387
- const headers = new Headers(res.headers);
1388
- headers.set("connection", "close");
1389
- return new Response(res.body, {
1390
- status: res.status,
1391
- statusText: res.statusText,
1392
- headers,
1393
- });
1394
- }
1395
- }
1525
+ let HTTPS_TERMINATOR;
1526
+ /**
1527
+ * The in-flight first bind of :443, if one is running.
1528
+ *
1529
+ * {@link ensureHttpsIngress} is async where the swap it replaced was
1530
+ * synchronous, and that reintroduces an interleaving the old code could
1531
+ * not have: two `ctx.startService({ tls })` calls issued together by a
1532
+ * fake handler would both find :443 absent and both try to bind it, and
1533
+ * the loser gets EADDRINUSE — a provisioning failure in place of the
1534
+ * dropped connection this change exists to remove. One promise, shared.
1535
+ */
1536
+ let HTTPS_INGRESS_STARTING;
1396
1537
  /**
1397
1538
  * The live ingress tables — per-port routes and the :443 SNI cert table.
1398
1539
  *
@@ -1441,6 +1582,12 @@ function stopIngressServers() {
1441
1582
  }
1442
1583
  }
1443
1584
  INGRESS_HTTPS_SERVERS.clear();
1585
+ // Force-closed for the same reason the listeners are: the containers
1586
+ // behind these routes are going away, so a connection still riding them
1587
+ // has nothing left to reach.
1588
+ HTTPS_TERMINATOR?.close();
1589
+ HTTPS_TERMINATOR = undefined;
1590
+ HTTPS_INGRESS_STARTING = undefined;
1444
1591
  clearTables(INGRESS);
1445
1592
  }
1446
1593
  function buildIngress(project) {
@@ -1655,7 +1802,7 @@ async function startIngress() {
1655
1802
  }
1656
1803
  // ── HTTPS listener on INGRESS_HTTPS_PORT: SNI per certificated hostname.
1657
1804
  if (INGRESS.certByHost.size > 0)
1658
- rebindHttpsListener(Bun);
1805
+ await ensureHttpsIngress(Bun);
1659
1806
  // Seed the resolver's names registry: ingress hostnames (fakes, TLS
1660
1807
  // proxies, dnsName(→ingress)) → bridge gateway, plus ingress-targeted
1661
1808
  // wildcards. Service-targeted wildcards wait for the post-container pass
@@ -1671,81 +1818,85 @@ function requireBun() {
1671
1818
  }
1672
1819
  return Bun;
1673
1820
  }
1674
- /** Flatten the live SNI cert table into Bun's TLS-entry array. */
1675
- function tlsEntriesFromCerts() {
1676
- return certEntries(INGRESS);
1677
- }
1678
1821
  /**
1679
- * (Re)bind the :443 listener from the current cert table + route map.
1822
+ * Bring :443 up, once.
1680
1823
  *
1681
- * Bun fixes a server's TLS config at `Bun.serve` time — `reload()` accepts a
1682
- * new `tls` option and silently keeps serving the old certificates (measured
1683
- * on Bun 1.3.14: after reloading with a second SNI entry, the new hostname
1684
- * still gets the first one's leaf). So adding a certificate really does mean
1685
- * a second listener.
1824
+ * Two pieces: a plaintext `Bun.serve` on an ephemeral loopback port that
1825
+ * carries the HTTPS route table (and therefore all the routing, fake
1826
+ * dispatch, interception, reverse-proxying and `server.upgrade()`
1827
+ * WebSocket bridging that :80 already gets, unchanged), and the TLS
1828
+ * terminator in front of it on :443.
1686
1829
  *
1687
- * **Bind the new one before stopping the old one.** Doing it the other way
1688
- * round — which is what this used to do, with `stop(true)` — has two teeth:
1689
- * the force-close kills every established connection on :443, and the gap
1690
- * before the new listener binds refuses new ones. Neither is limited to the
1691
- * hostname being added; they hit all the unrelated traffic the listener is
1692
- * carrying. On a project that mints a certificate per provisioned database
1693
- * while deploys stream through the same port, that surfaced as the app under
1694
- * test dying with `SocketError: other side closed` or `ECONNREFUSED
1695
- * <gateway>:443` — a different test each run, and nothing pointing at
1696
- * ingress.
1830
+ * **This is idempotent, and that is the feature.** Adding a certificate
1831
+ * mutates {@link IngressTables.certByHost} and nothing else: the
1832
+ * terminator reads that table on every handshake, so the new hostname is
1833
+ * served by the next connection and no established connection is
1834
+ * disturbed. The listener this replaced had to be stopped and re-served
1835
+ * for each new certificate, and every such swap severed the idle
1836
+ * kept-alive connections it was carrying — see
1837
+ * `harness/tls-terminator.ts` for the measurements and the history.
1697
1838
  *
1698
- * Overlapping the two needs SO_REUSEPORT on both sockets ({@link
1699
- * bindIngressServer} sets it unconditionally for that reason), after which
1700
- * the old listener is drained with `stop(false)` so requests in flight
1701
- * finish. During the overlap the kernel may hand a new connection to either
1702
- * socket, which is safe: the only hostname the two disagree about is the one
1703
- * being added, and it does not resolve until the caller writes the names
1704
- * registry after this returns.
1839
+ * The route `Map` is the persistent module object, so the plaintext
1840
+ * server closes over the same table and later route additions need no
1841
+ * restart either.
1705
1842
  *
1706
- * The route Map is the persistent module object, so the new listener closes
1707
- * over the same table and later route additions need no rebind at all.
1843
+ * The loopback port is ephemeral rather than fixed so it can never
1844
+ * collide with a fake's declared port; it is bound on 127.0.0.1, which no
1845
+ * container can reach (containers arrive at the bridge gateway), so the
1846
+ * only way in is through the terminator.
1708
1847
  */
1709
1848
  // eslint-disable-next-line @typescript-eslint/no-explicit-any
1710
- function rebindHttpsListener(Bun) {
1711
- // The *same* table object the previous listener closed over: the new
1712
- // listener must serve the routes bound since, and any bound later.
1849
+ async function ensureHttpsIngress(Bun) {
1850
+ if (INGRESS_HTTPS_SERVERS.has(INGRESS_HTTPS_PORT))
1851
+ return;
1852
+ if (!HTTPS_INGRESS_STARTING) {
1853
+ // Cleared on both outcomes: once bound, the check above short-circuits
1854
+ // every later caller, and a failed bind must be retryable rather than
1855
+ // remembered as a rejection for the life of the project.
1856
+ HTTPS_INGRESS_STARTING = startHttpsIngress(Bun).finally(() => {
1857
+ HTTPS_INGRESS_STARTING = undefined;
1858
+ });
1859
+ }
1860
+ return HTTPS_INGRESS_STARTING;
1861
+ }
1862
+ // eslint-disable-next-line @typescript-eslint/no-explicit-any
1863
+ async function startHttpsIngress(Bun) {
1713
1864
  const routes = routesFor(INGRESS, INGRESS_HTTPS_PORT);
1714
- const old = INGRESS_HTTPS_SERVERS.get(INGRESS_HTTPS_PORT);
1715
- const server = bindIngressServer(Bun, INGRESS_HTTPS_PORT, routes, `https :${INGRESS_HTTPS_PORT}`, tlsEntriesFromCerts());
1716
- if (old) {
1865
+ // `proto` is passed rather than inferred: this listener speaks plain
1866
+ // HTTP, but everything reaching it arrived over TLS, so X-Forwarded-Proto
1867
+ // must say https.
1868
+ const plain = bindIngressServer(Bun, 0, routes, `https :${INGRESS_HTTPS_PORT}`, {
1869
+ proto: "https",
1870
+ hostname: "127.0.0.1",
1871
+ });
1872
+ INGRESS_HTTPS_SERVERS.set(INGRESS_HTTPS_PORT, plain);
1873
+ try {
1874
+ HTTPS_TERMINATOR = await startTlsTerminator({
1875
+ port: INGRESS_HTTPS_PORT,
1876
+ hostname: "0.0.0.0",
1877
+ upstreamPort: plain.port,
1878
+ certFor: (serverName) => {
1879
+ if (serverName === undefined)
1880
+ return INGRESS.certByHost.values().next().value;
1881
+ const name = selectCertName(INGRESS.certByHost.keys(), serverName.toLowerCase());
1882
+ return name === undefined ? undefined : INGRESS.certByHost.get(name);
1883
+ },
1884
+ });
1885
+ }
1886
+ catch (err) {
1887
+ // Leave nothing half-built: a registered plaintext server with no
1888
+ // terminator would make `planBind` believe :443 is up.
1889
+ INGRESS_HTTPS_SERVERS.delete(INGRESS_HTTPS_PORT);
1717
1890
  try {
1718
- // Graceful: stop accepting, let in-flight requests finish. The old
1719
- // server stays alive until they do, which is the point — a long
1720
- // upload through ingress must not be collateral damage of another
1721
- // service being provisioned.
1722
- old.stop(false);
1723
- // Bun 1.3.14 landmine: a request arriving later on one of the old
1724
- // server's kept-alive connections dispatches into freed state and can
1725
- // segfault the daemon (see {@link DRAINING_INGRESS}). Mark it so any
1726
- // response it still serves closes its connection, and force-close the
1727
- // stragglers once in-flight work has had a fair window to finish.
1728
- DRAINING_INGRESS.add(old);
1729
- const graceTimer = setTimeout(() => {
1730
- try {
1731
- old.stop(true);
1732
- }
1733
- catch {
1734
- /* already fully stopped */
1735
- }
1736
- }, REBIND_DRAIN_GRACE_MS);
1737
- // Don't let the grace timer keep the process alive on shutdown.
1738
- // eslint-disable-next-line @typescript-eslint/no-explicit-any
1739
- graceTimer.unref?.();
1891
+ plain.stop(true);
1740
1892
  }
1741
- catch (err) {
1742
- // eslint-disable-next-line no-console
1743
- console.warn("[ingress] failed to drain the previous https listener:", err);
1893
+ catch {
1894
+ /* already gone */
1744
1895
  }
1896
+ throw err;
1745
1897
  }
1746
- INGRESS_HTTPS_SERVERS.set(INGRESS_HTTPS_PORT, server);
1747
1898
  // eslint-disable-next-line no-console
1748
- console.log(`[ingress] https :${INGRESS_HTTPS_PORT} for ${[...routes.keys()].join(", ")}`);
1899
+ console.log(`[ingress] https :${INGRESS_HTTPS_PORT} (tls terminator -> :${plain.port}) for ${[...routes.keys()].join(", ")}`);
1749
1900
  }
1750
1901
  /**
1751
1902
  * RFC 6125 wildcard match: `*.example.com` covers `api.example.com` but NOT
@@ -1800,10 +1951,13 @@ async function bindRuntimeTls(hostname, service, port) {
1800
1951
  INGRESS_HTTP_SERVERS.set(INGRESS_HTTP_PORT, bindIngressServer(Bun, INGRESS_HTTP_PORT, routesFor(INGRESS, INGRESS_HTTP_PORT), `port ${INGRESS_HTTP_PORT}`));
1801
1952
  }
1802
1953
  if (plan.needsCert) {
1954
+ // The terminator reads this table on every handshake, so the leaf is
1955
+ // live the moment it lands — no listener is touched, and no connection
1956
+ // already on :443 notices.
1803
1957
  INGRESS.certByHost.set(host, await generateHostCert(host, [host]));
1804
1958
  }
1805
- if (plan.needsHttpsRebind)
1806
- rebindHttpsListener(Bun);
1959
+ if (plan.needsHttpsListener)
1960
+ await ensureHttpsIngress(Bun);
1807
1961
  // Resolve the hostname to the daemon gateway (where :443/:80 listen).
1808
1962
  // A wildcard can only live in the resolver's suffix table.
1809
1963
  const gw = await bridgeGatewayIp();
@@ -1851,20 +2005,26 @@ async function unbindRuntimeTls(hostname) {
1851
2005
  // eslint-disable-next-line @typescript-eslint/no-explicit-any
1852
2006
  function bindIngressServer(
1853
2007
  // eslint-disable-next-line @typescript-eslint/no-explicit-any
1854
- Bun, port, byHost, listenerLabel, tlsEntries) {
1855
- // A TLS listener terminates https; everything else is plain http. Used
1856
- // to stamp X-Forwarded-Proto so upstreams that build absolute URLs or
1857
- // redirect see the scheme the client actually used, not our http hop.
1858
- const proto = tlsEntries ? "https" : "http";
2008
+ Bun, port, byHost, listenerLabel,
2009
+ // Every ingress listener speaks plain HTTP now — the one behind :443
2010
+ // sits under the TLS terminator. So the scheme a client actually used
2011
+ // can no longer be inferred from the socket and is declared instead; it
2012
+ // stamps X-Forwarded-Proto, which upstreams that build absolute URLs or
2013
+ // redirect depend on.
2014
+ serve = { proto: "http" }) {
2015
+ const proto = serve.proto;
1859
2016
  // eslint-disable-next-line @typescript-eslint/no-explicit-any
1860
2017
  const opts = {
1861
2018
  port,
1862
- hostname: "0.0.0.0",
1863
- // SO_REUSEPORT on every ingress listener, so a replacement can be bound
1864
- // while the old one is still serving. That overlap is the only way to
1865
- // add an SNI certificate without a gap — see {@link rebindHttpsListener}
1866
- // — and it only works if *both* sockets opt in: a second plain bind
1867
- // fails with "Is port 443 in use?".
2019
+ hostname: serve.hostname ?? "0.0.0.0",
2020
+ // SO_REUSEPORT on every ingress listener. It was introduced so a
2021
+ // replacement :443 could overlap the original during a certificate
2022
+ // swap; the terminator removed that swap, and this stays for the
2023
+ // remaining case — the `/load` teardown force-closes and rebinds, and
2024
+ // a socket the kernel has not finished releasing would otherwise fail
2025
+ // the next bind with "Is port 443 in use?". Note the flip side, which
2026
+ // {@link stopIngressServers} depends on: a half-alive listener shares
2027
+ // the port silently instead of colliding loudly.
1868
2028
  reusePort: true,
1869
2029
  // Bun.serve defaults to a 10s idleTimeout, which kills any proxied
1870
2030
  // request whose upstream takes >10s to produce bytes — under parallel
@@ -1876,7 +2036,7 @@ Bun, port, byHost, listenerLabel, tlsEntries) {
1876
2036
  // short-lived, leaked-connection risk is bounded by the fork.
1877
2037
  idleTimeout: 0,
1878
2038
  // eslint-disable-next-line @typescript-eslint/no-explicit-any
1879
- fetch: (req, server) => dispatchIngress(req, server, byHost, listenerLabel, proto).then((res) => withDrainClose(server, res)),
2039
+ fetch: (req, server) => dispatchIngress(req, server, byHost, listenerLabel, proto),
1880
2040
  websocket: {
1881
2041
  // eslint-disable-next-line @typescript-eslint/no-explicit-any
1882
2042
  async open(ws) {
@@ -1953,8 +2113,6 @@ Bun, port, byHost, listenerLabel, tlsEntries) {
1953
2113
  },
1954
2114
  },
1955
2115
  };
1956
- if (tlsEntries)
1957
- opts.tls = tlsEntries;
1958
2116
  return Bun.serve(opts);
1959
2117
  }
1960
2118
  /**
@@ -2018,6 +2176,28 @@ server, byHost, listenerLabel, proto) {
2018
2176
  : await runChain(chain, req, upstream, recordInterceptedRequest);
2019
2177
  return augmentCorsResponse(req, res);
2020
2178
  }
2179
+ /**
2180
+ * The address to report as the client's in `X-Forwarded-For`.
2181
+ *
2182
+ * On :80 that is simply the socket's peer. On :443 the peer is the TLS
2183
+ * terminator on loopback, so the real address has to be recovered from it:
2184
+ * the terminator keys `source port -> client address` for the life of each
2185
+ * upstream connection, and the source port is what `requestIP` reports
2186
+ * here. Gated on `proto === "https"` so a genuinely loopback caller on :80
2187
+ * can never pick up an unrelated terminator connection's port.
2188
+ *
2189
+ * Falling back to the socket's own address means the worst case is the
2190
+ * pre-terminator answer for a plain HTTP hop, never a wrong tenant.
2191
+ */
2192
+ // eslint-disable-next-line @typescript-eslint/no-explicit-any
2193
+ function ingressClientIp(server, req, proto) {
2194
+ const peer = server.requestIP?.(req);
2195
+ if (!peer?.address)
2196
+ return undefined;
2197
+ if (proto !== "https")
2198
+ return peer.address;
2199
+ return HTTPS_TERMINATOR?.clientIpFor(peer.port) ?? peer.address;
2200
+ }
2021
2201
  /**
2022
2202
  * Reverse-proxy a request to `http://<service>:<port>` on
2023
2203
  * `spectest-net`. Handles plain HTTP/1.1 + 2 and WebSocket upgrades:
@@ -2067,7 +2247,7 @@ server, service, port, listenerLabel, proto) {
2067
2247
  // Standard reverse-proxy provenance headers: the upstream sees the
2068
2248
  // public scheme/host it was reached through and the client's address,
2069
2249
  // even though we rewrite Host below to the service-net name.
2070
- const clientIp = server.requestIP?.(req)?.address;
2250
+ const clientIp = ingressClientIp(server, req, proto);
2071
2251
  const priorXff = req.headers.get("x-forwarded-for");
2072
2252
  const xff = clientIp ? (priorXff ? `${priorXff}, ${clientIp}` : clientIp) : priorXff;
2073
2253
  if (xff)
@@ -2860,18 +3040,14 @@ async function bootstrapInner() {
2860
3040
  // at "image ready" waiting for an unrelated slow build elsewhere.
2861
3041
  //
2862
3042
  // Prep concurrency: registry pulls always run in parallel (network-bound,
2863
- // low VM RAM). Dockerfile builds parallelize *only* when the host
2864
- // buildkitd is in play — there the build executes host-side under runc, so
2865
- // N concurrent builds don't touch the VM's memory ceiling. When we fall
2866
- // back to the in-VM builder, two or more concurrent builds routinely OOM a
2867
- // single VM on monorepos with parallel pnpm/npm installs (each install
2868
- // fans out to ~16 fetchers + lifecycle workers, ~70 MB/process), so we
2869
- // serialize that case behind a FIFO chain — but only the in-VM builds
2870
- // serialize; pulls and starts run freely alongside them. The remote-builder
2871
- // probe is memoized, so this up-front call is free; skip it with no builds.
3043
+ // low VM RAM). Dockerfile builds run inside the VM, and two or more
3044
+ // concurrent builds routinely OOM a single VM on monorepos with parallel
3045
+ // pnpm/npm installs (each install fans out to ~16 fetchers + lifecycle
3046
+ // workers, ~70 MB/process), so builds serialize behind a FIFO chain —
3047
+ // but only the builds; pulls and starts run freely alongside them.
2872
3048
  const tags = new Map();
2873
3049
  const builds = services.filter((s) => s.image.type === "dockerfile");
2874
- const buildsRunHostSide = builds.length > 0 && (await ensureRemoteBuilder());
3050
+ const buildsRunHostSide = false;
2875
3051
  // A promise chain is a fair FIFO mutex: when builds run in-VM, each build
2876
3052
  // waits for the previous to settle. Pulls and host-side builds bypass it.
2877
3053
  let inVmBuildChain = Promise.resolve();
@@ -2952,11 +3128,16 @@ async function bootstrapInner() {
2952
3128
  if (svc.setup) {
2953
3129
  progressService(svc.name, { status: "probing", detail: "running setup" });
2954
3130
  const helpers = await ensureHelpers(svc.name, svc);
2955
- await svc.setup({
2956
- name: svc.name,
2957
- helpers,
2958
- ...(await spectestContext({ service: svc.name, includeSelf: true })),
2959
- });
3131
+ try {
3132
+ await svc.setup({
3133
+ name: svc.name,
3134
+ helpers,
3135
+ ...(await spectestContext({ service: svc.name, includeSelf: true })),
3136
+ });
3137
+ }
3138
+ catch (err) {
3139
+ throw await withContainerPostMortem(svc.name, err);
3140
+ }
2960
3141
  }
2961
3142
  progressService(svc.name, { status: "ready", detail: undefined });
2962
3143
  const ti = timings.get(svc.name);