@specific.dev/spectest 0.69.0 → 0.71.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -332,4 +332,7 @@ export declare function k3s(opts?: K3sOptions): {
332
332
  tmpfs: string[];
333
333
  cgroupns: string;
334
334
  ports: number[];
335
+ volumes: {
336
+ target: string;
337
+ }[];
335
338
  };
@@ -1,7 +1,7 @@
1
1
  import { AsyncLocalStorage } from "node:async_hooks";
2
2
  import { spawn as nodeSpawn } from "node:child_process";
3
3
  import { randomUUID } from "node:crypto";
4
- import { existsSync, readFileSync } from "node:fs";
4
+ import { existsSync } from "node:fs";
5
5
  import { readFile, unlink } from "node:fs/promises";
6
6
  import { AppsV1Api, BatchV1Api, CoreV1Api, KubeConfig, KubernetesObjectApi, PatchStrategy, ResponseContext, ServerConfiguration, createConfiguration, loadAllYaml, } from "@kubernetes/client-node";
7
7
  import { Observable } from "@kubernetes/client-node/dist/gen/rxjsStub.js";
@@ -10,6 +10,114 @@ import { deepUnwrap, readTag, wrap } from "../inspect.js";
10
10
  import { recorderAnnotate, recorderRemove } from "../recorder.js";
11
11
  /** Port the in-cluster registry listens on (plain HTTP). */
12
12
  const K3S_REGISTRY_PORT = 5000;
13
+ /** The cluster's own containerd root, inside the k3s container. */
14
+ const CLUSTER_STORE_TARGET = "/var/lib/rancher/k3s/agent/containerd";
15
+ /**
16
+ * Written before the server starts and removed by the ready probe.
17
+ *
18
+ * It survives in the store only when a boot never reached a live API
19
+ * server, which is the one thing the sweep below cannot work out for
20
+ * itself. Blacksmith's rule, from the other end: they refuse to commit a
21
+ * cache when the integrity check *did not run*, not only when it failed.
22
+ */
23
+ const BOOT_MARKER = ".spectest-boot-incomplete";
24
+ /**
25
+ * Bring the cluster's image store back into service, or throw it away.
26
+ *
27
+ * The store is the cluster's containerd root on a directory that outlives
28
+ * this container (`NESTED_STORE_ROOT`), so a build finds coredns, traefik,
29
+ * the local-path provisioner and every image the project deployed already
30
+ * pulled **and already extracted**. Extraction is 48x the I/O of the
31
+ * download (`STICKY_DISKS.md` §7.2), which is why the zot mirror never
32
+ * helped here and why an extracted store is the whole prize.
33
+ *
34
+ * What it inherits with them is one build's crash state: the store was
35
+ * captured from a running guest, and the teardown's `docker rm -f` killed
36
+ * that cluster where it stood. **Deleting the dead files is not enough,
37
+ * and the failure is not subtle.** containerd's index still lists the
38
+ * previous build's containers, so the CRI plugin offers them to a kubelet
39
+ * whose API server has never heard of them; the kubelet spends its startup
40
+ * reconciling containers whose tasks are gone (`Failed to create existing
41
+ * container … task not found`) and does not register its own Node in time,
42
+ * and the CSI plugin — which waits on that Node with a fixed budget —
43
+ * calls `Fatalf`. The kubelet dies, k3s exits 2, and what the author sees
44
+ * is a rollout that timed out against a hostname that no longer resolves.
45
+ * Measured: 2 of 2 delta restores, against 2 of 2 clean without the store.
46
+ *
47
+ * So the sweep removes those containers through **containerd itself**,
48
+ * which is the only thing that can edit its index. The k3s image ships a
49
+ * standalone `containerd` and `ctr`, so a throwaway daemon runs against
50
+ * the store with the CRI plugin disabled, deletes the previous build's
51
+ * containers and their tasks, and stops. Images, content blobs and
52
+ * extracted snapshots — the whole cache — stay. It is the same discipline
53
+ * `spectest-image-cache-up` applies to the guest's own store one level up,
54
+ * where it sweeps the `moby` namespace before dockerd starts.
55
+ *
56
+ * Two failures cost a cache and never a build:
57
+ *
58
+ * 1. **A sweep that cannot run wipes the store.** A daemon that will not
59
+ * start leaves behind exactly the containers this cluster fatals on,
60
+ * so carrying on is the one option that is known bad.
61
+ * 2. **A store that did not work is never inherited twice.** The marker
62
+ * is written before the server starts and removed by the ready probe,
63
+ * so it survives only when a boot never reached a live API server.
64
+ * That is the backstop for the sweep being incomplete in some way we
65
+ * have not seen: the cost is one cold build's pulls, and it repairs
66
+ * itself. Blacksmith's rule from the other end — they refuse to commit
67
+ * a cache when the integrity check *did not run*, not only when it
68
+ * failed.
69
+ */
70
+ /** Plugins the sweep daemon does not need. Every plugin is startup time,
71
+ * and the sweep sits on the critical path of the cluster's boot. CRI is
72
+ * off for a second reason: this daemon exists to edit an index, and a CRI
73
+ * plugin would set about being a container runtime on a store we are
74
+ * seconds from handing to the real one. */
75
+ const SWEEP_DISABLED_PLUGINS = [
76
+ "io.containerd.grpc.v1.cri",
77
+ "io.containerd.snapshotter.v1.btrfs",
78
+ "io.containerd.snapshotter.v1.native",
79
+ "io.containerd.snapshotter.v1.aufs",
80
+ "io.containerd.snapshotter.v1.zfs",
81
+ "io.containerd.snapshotter.v1.devmapper",
82
+ "io.containerd.snapshotter.v1.stargz",
83
+ "io.containerd.snapshotter.v1.fuse-overlayfs",
84
+ ];
85
+ const STORE_SWEEP_SH = `S=${CLUSTER_STORE_TARGET}; M="$S/${BOOT_MARKER}"; A=/run/spectest-sweep.sock; ` +
86
+ `sweep_store() { ` +
87
+ `printf '%s\\n' 'version = 2' ` +
88
+ `'disabled_plugins = [${SWEEP_DISABLED_PLUGINS.map((p) => `"${p}"`).join(", ")}]' ` +
89
+ `> /run/spectest-sweep.toml || return 1; ` +
90
+ `containerd -c /run/spectest-sweep.toml --root "$S" --state /run/spectest-sweep ` +
91
+ `--address "$A" > /run/spectest-sweep.log 2>&1 & ` +
92
+ `cd_pid=$!; i=0; ` +
93
+ `while [ ! -S "$A" ] && [ $i -lt 600 ]; do sleep 0.1; i=$((i+1)); done; ` +
94
+ `if [ ! -S "$A" ]; then kill $cd_pid 2>/dev/null; return 1; fi; ` +
95
+ // One invocation for the whole set, not two per container. Each `ctr` is
96
+ // a Go binary start plus a gRPC round trip, and a build leaves ~20
97
+ // containers behind — the per-container form spent seconds of the
98
+ // cluster's boot on process starts alone (measured: ready 3.6s -> 7.0s).
99
+ `ids=$(timeout 30 ctr -a "$A" -n k8s.io containers ls -q 2>/dev/null); ` +
100
+ `if [ -n "$ids" ]; then ` +
101
+ `timeout 60 ctr -a "$A" -n k8s.io tasks rm -f $ids >/dev/null 2>&1 || true; ` +
102
+ `timeout 60 ctr -a "$A" -n k8s.io containers rm $ids >/dev/null 2>&1 || true; fi; ` +
103
+ // containerd 2.x keeps sandboxes in a store of their own; on the 1.7 k3s
104
+ // ships they are ordinary containers and this is a no-op.
105
+ `sb=$(timeout 30 ctr -a "$A" -n k8s.io sandboxes ls -q 2>/dev/null); ` +
106
+ `[ -n "$sb" ] && timeout 60 ctr -a "$A" -n k8s.io sandboxes rm $sb >/dev/null 2>&1; ` +
107
+ `kill $cd_pid 2>/dev/null; wait $cd_pid 2>/dev/null; ` +
108
+ `rm -rf "$S/io.containerd.runtime.v2.task" "$S/tmpmounts" /run/spectest-sweep "$A" 2>/dev/null || true; ` +
109
+ `return 0; }; ` +
110
+ `if [ -e "$M" ]; then ` +
111
+ `echo 'spectest: the previous cluster on this image store never became ready; starting from an empty one' >&2; ` +
112
+ `rm -rf "$S"/* "$S"/.[!.]* 2>/dev/null || true; ` +
113
+ `elif [ -d "$S/io.containerd.metadata.v1.bolt" ]; then ` +
114
+ // Timed and printed always: the sweep is inside the cluster's ready time,
115
+ // so it is the first thing to suspect if that time moves.
116
+ `T0=$(date +%s); ` +
117
+ `if sweep_store; then echo "spectest: reused the cluster image store at $S (swept in $(($(date +%s)-T0))s)"; ` +
118
+ `else echo 'spectest: could not sweep the inherited image store; starting from an empty one' >&2; ` +
119
+ `rm -rf "$S"/* "$S"/.[!.]* 2>/dev/null || true; fi; fi; ` +
120
+ `mkdir -p "$S" && : > "$M" 2>/dev/null || true; `;
13
121
  /**
14
122
  * `extraArgs` entries the component overrides anyway. Passing one of
15
123
  * these looks like it works — the flag really is appended to the k3s
@@ -551,50 +659,21 @@ ${ports}${volumeMounts}${volumes}
551
659
  `;
552
660
  }
553
661
  /**
554
- * Host-side `zot` pull-through cache layout (local Firecracker provider
555
- * only). One zot instance per upstream registry, all bound to the
556
- * `spectest-br0` gateway `10.42.0.1` on the ports below — **kept in sync
557
- * with `scripts/install-zot.sh`**. We mirror the cluster's containerd
558
- * through these so every image pull reuses the shared host cache instead
559
- * of hitting the public registry, and we list the canonical upstream as
560
- * a fallback endpoint so a missing/cold mirror only ever slows a pull,
561
- * never breaks it.
662
+ * Docker Hub through Google's public mirror, with the canonical upstream
663
+ * as the fallback endpoint so a cold or unavailable mirror only ever
664
+ * slows a pull, never breaks it. The same route the guest's own dockerd
665
+ * takes (golden's daemon.json) and its in-VM BuildKit. Every other
666
+ * registry is pulled direct. Nothing here points at a host-side service:
667
+ * the cluster's own containerd store lives in the environment and is
668
+ * captured with it (CONTAINER_STORE.md).
562
669
  */
563
- const ZOT_MIRRORS = [
564
- { registry: "docker.io", port: 5000, upstream: "https://registry-1.docker.io" },
565
- { registry: "ghcr.io", port: 5001, upstream: "https://ghcr.io" },
566
- { registry: "quay.io", port: 5002, upstream: "https://quay.io" },
567
- { registry: "registry.k8s.io", port: 5003, upstream: "https://registry.k8s.io" },
568
- { registry: "public.ecr.aws", port: 5004, upstream: "https://public.ecr.aws" },
569
- { registry: "gcr.io", port: 5005, upstream: "https://gcr.io" },
570
- { registry: "mcr.microsoft.com", port: 5006, upstream: "https://mcr.microsoft.com" },
571
- ];
572
- /**
573
- * Discover the host-side image cache gateway by reading the same
574
- * `registry-mirrors` entry the in-VM dockerd already uses (baked into
575
- * the golden rootfs's `/etc/docker/daemon.json`). Returns the
576
- * gateway host (`"10.42.0.1"`) when present, or `null` when there's no
577
- * host cache, in which case the cluster pulls every image direct. Runs inside the daemon (VM) at `index.ts` load time, so
578
- * the result is stable per host and never poisons the warm-template
579
- * cache.
580
- */
581
- function detectHostMirrorGateway() {
582
- try {
583
- const cfg = JSON.parse(readFileSync("/etc/docker/daemon.json", "utf8"));
584
- const first = cfg["registry-mirrors"]?.[0];
585
- return first ? new URL(first).hostname || null : null;
586
- }
587
- catch {
588
- return null;
589
- }
590
- }
670
+ const HUB_MIRROR = { registry: "docker.io", mirror: "https://mirror.gcr.io", upstream: "https://registry-1.docker.io" };
591
671
  /**
592
672
  * Build `/etc/rancher/k3s/registries.yaml`. k3s reads this **once, at
593
673
  * startup**, to configure its embedded containerd — which is why it has
594
674
  * to be seeded via `files` (a pre-start bind mount) rather than a
595
675
  * `setup` hook. Two jobs:
596
- * 1. Mirror the cluster's image pulls through the host `zot` cache
597
- * (omitted when there's no host cache).
676
+ * 1. Mirror the cluster's Docker Hub pulls through `mirror.gcr.io`.
598
677
  * 2. Trust the in-cluster registry, addressed as `<key>.internal:5000`
599
678
  * (the `{{SPECTEST_SERVICE}}` token is expanded to the cluster's
600
679
  * service key when the file is written). Image *references* use
@@ -602,19 +681,15 @@ function detectHostMirrorGateway() {
602
681
  * endpoint `http://127.0.0.1:5000` — the hostNetwork registry pod
603
682
  * shares the node's netns, so this needs no in-container DNS and
604
683
  * can't be broken by a clobbered `hostnames`.
605
- * Returns `null` when there's nothing to configure (no host cache and
606
- * `registry` disabled), in which case no file is injected.
607
684
  */
608
685
  function buildRegistriesYaml(registryEnabled) {
609
- const gateway = detectHostMirrorGateway();
610
- if (!gateway && !registryEnabled)
611
- return null;
612
- const lines = ["mirrors:"];
613
- if (gateway) {
614
- for (const { registry, port, upstream } of ZOT_MIRRORS) {
615
- lines.push(` "${registry}":`, ` endpoint:`, ` - "http://${gateway}:${port}"`, ` - "${upstream}"`);
616
- }
617
- }
686
+ const lines = [
687
+ "mirrors:",
688
+ ` "${HUB_MIRROR.registry}":`,
689
+ ` endpoint:`,
690
+ ` - "${HUB_MIRROR.mirror}"`,
691
+ ` - "${HUB_MIRROR.upstream}"`,
692
+ ];
618
693
  if (registryEnabled) {
619
694
  const host = `{{SPECTEST_SERVICE}}.internal:${K3S_REGISTRY_PORT}`;
620
695
  lines.push(` "${host}":`, ` endpoint:`, ` - "http://127.0.0.1:${K3S_REGISTRY_PORT}"`, "configs:",
@@ -1312,12 +1387,18 @@ export function k3s(opts = {}) {
1312
1387
  // is affected.
1313
1388
  "mount --make-rshared / 2>/dev/null || " +
1314
1389
  "echo 'spectest: could not make / rshared; CSI node plugins may fail to publish volumes' >&2; " +
1390
+ STORE_SWEEP_SH +
1315
1391
  `exec ${serverArgs}`;
1316
1392
  // Plain /readyz probe. On a warm zot cache the cluster's images are
1317
1393
  // already local, so the first boot completes in seconds; the
1318
1394
  // first-ever boot on a cold-cache host pulls through the mirror and
1319
1395
  // can take a couple of minutes (covered by readyTimeoutSecs).
1320
- const readyCmd = "kubectl get --raw=/readyz >/dev/null 2>&1";
1396
+ //
1397
+ // Clearing the boot marker is part of the probe on purpose: "the API
1398
+ // server answers" is exactly the condition that proves this cluster came
1399
+ // up on the store it was given, and it is the only signal the sweep can
1400
+ // read on the next boot. See STORE_SWEEP_SH.
1401
+ const readyCmd = `kubectl get --raw=/readyz >/dev/null 2>&1 && rm -f ${CLUSTER_STORE_TARGET}/${BOOT_MARKER}`;
1321
1402
  const def = {
1322
1403
  image: { type: "registry", reference: `rancher/k3s:${version}` },
1323
1404
  command: cmd,
@@ -1333,15 +1414,30 @@ export function k3s(opts = {}) {
1333
1414
  // the same netns and are reachable the same way, without appearing
1334
1415
  // in this list (it's documentation, not a firewall).
1335
1416
  ports: registryEnabled ? [80, 443, 6443, K3S_REGISTRY_PORT] : [80, 443, 6443],
1336
- // NOTE: do NOT mount /var/lib/rancher/k3s/agent/containerd as a cache
1337
- // volume. It was tried (to spare a recreated cluster re-pulling its
1338
- // system images on delta restores) and a fresh k3s server against the
1339
- // previous container's containerd store — killed un-cleanly by the
1340
- // teardown's `docker rm -f` — wedged the apiserver minutes in
1341
- // (rollouts never settled, pod listing started failing). The zot
1342
- // mirror already makes those re-pulls cheap; the residual win wasn't
1343
- // worth the recovery semantics of a crash-state store under a fresh
1344
- // cluster db.
1417
+ // The cluster's image store, on a volume that outlives this
1418
+ // container. Without it the store is the container's writable layer,
1419
+ // which `docker rm -f` deletes at teardown — so every build re-pulls
1420
+ // and, far more expensively, re-extracts coredns, traefik, the
1421
+ // local-path provisioner and everything the project deploys.
1422
+ // Measured on the one project on the image cache: 20.6 s per
1423
+ // build, identical on the cold and delta tiers, i.e. the one term
1424
+ // neither the delta restore nor the cache image store reached.
1425
+ //
1426
+ // `cache` names a cache disk, which is the ordinary way any project
1427
+ // asks for one — there is nothing Kubernetes-shaped in the platform
1428
+ // for this. Two clusters in one environment may name the same disk
1429
+ // and still get separate containerd roots, because a volume is rooted
1430
+ // per service on whatever disk it names; sharing a root would corrupt
1431
+ // it, since two containerd daemons cannot share a store (bolt takes
1432
+ // an exclusive lock).
1433
+ //
1434
+ // This reverses a NOTE that stood here for a year: a fresh k3s server
1435
+ // over a store killed un-cleanly by `docker rm -f` wedged the
1436
+ // apiserver minutes in, and nothing recovered. What was missing was
1437
+ // the discipline `base.rs::IMAGE_CACHE_UP_SH` applies to the guest's own
1438
+ // store — sweep the crash state, and never inherit a store that did
1439
+ // not work. Both are in STORE_SWEEP_SH.
1440
+ volumes: [{ target: CLUSTER_STORE_TARGET }],
1345
1441
  ...(registriesYaml
1346
1442
  ? {
1347
1443
  files: [