@specific.dev/spectest 0.71.0 → 0.72.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -3,7 +3,13 @@ import type { ServiceHelpersContext } from "../index.js";
3
3
  import type { Wrapped } from "../inspect.js";
4
4
  export type { KubernetesObject, KubernetesListObject, } from "@kubernetes/client-node";
5
5
  export interface K3sOptions {
6
- /** Image tag for the official `rancher/k3s` image. Default `"v1.30.6-k3s1"`. */
6
+ /**
7
+ * Image tag for the official `rancher/k3s` image. Default
8
+ * `"v1.33.5-k3s1"`. Any tag works; a k3s older than 1.33 embeds a
9
+ * containerd (1.7) without the EROFS snapshotter, so its cluster keeps
10
+ * its images on overlayfs and shares the environment's image cache only
11
+ * as a directory that survives — see `k3s()`'s store section.
12
+ */
7
13
  version?: string;
8
14
  /**
9
15
  * Extra arguments appended to `k3s server`. Useful for `--tls-san=...`,
@@ -309,20 +315,6 @@ export interface K3sHelpers {
309
315
  }): Promise<Wrapped<KubernetesListObject<T>>>;
310
316
  }
311
317
  export declare function k3s(opts?: K3sOptions): {
312
- readyCheck: {
313
- type: "exec";
314
- command: string;
315
- timeoutSecs: number;
316
- };
317
- setup: ({ name, helpers }: {
318
- name: string;
319
- helpers: K3sHelpers;
320
- }) => Promise<void>;
321
- helpers: ({ name, exec, poll }: ServiceHelpersContext) => Promise<K3sHelpers>;
322
- files?: {
323
- path: string;
324
- content: string;
325
- }[] | undefined;
326
318
  image: {
327
319
  type: "registry";
328
320
  reference: string;
@@ -332,7 +324,29 @@ export declare function k3s(opts?: K3sOptions): {
332
324
  tmpfs: string[];
333
325
  cgroupns: string;
334
326
  ports: number[];
335
- volumes: {
327
+ volumes: ({
328
+ source: string;
329
+ target: string;
330
+ readOnly?: undefined;
331
+ } | {
332
+ source: string;
336
333
  target: string;
334
+ readOnly: true;
335
+ })[] | {
336
+ target: string;
337
+ }[];
338
+ files: {
339
+ path: string;
340
+ content: string;
337
341
  }[];
342
+ readyCheck: {
343
+ type: "exec";
344
+ command: string;
345
+ timeoutSecs: number;
346
+ };
347
+ setup: ({ name, helpers }: {
348
+ name: string;
349
+ helpers: K3sHelpers;
350
+ }) => Promise<void>;
351
+ helpers: ({ name, exec, poll }: ServiceHelpersContext) => Promise<K3sHelpers>;
338
352
  };
@@ -6,12 +6,22 @@ import { readFile, unlink } from "node:fs/promises";
6
6
  import { AppsV1Api, BatchV1Api, CoreV1Api, KubeConfig, KubernetesObjectApi, PatchStrategy, ResponseContext, ServerConfiguration, createConfiguration, loadAllYaml, } from "@kubernetes/client-node";
7
7
  import { Observable } from "@kubernetes/client-node/dist/gen/rxjsStub.js";
8
8
  import { dnsName, provides, SELF_SERVICE_TOKEN } from "../index.js";
9
+ import { MKFS_EROFS_PATH, STORE_ADOPT_PATH, imageCachePathsSync, nestedStoreDir, } from "../harness/image-cache.js";
9
10
  import { deepUnwrap, readTag, wrap } from "../inspect.js";
10
11
  import { recorderAnnotate, recorderRemove } from "../recorder.js";
11
12
  /** Port the in-cluster registry listens on (plain HTTP). */
12
13
  const K3S_REGISTRY_PORT = 5000;
13
14
  /** The cluster's own containerd root, inside the k3s container. */
14
15
  const CLUSTER_STORE_TARGET = "/var/lib/rancher/k3s/agent/containerd";
16
+ /** Where k3s (containerd 2.x layout) reads a containerd config template. */
17
+ const CONTAINERD_TEMPLATE_PATH = "/var/lib/rancher/k3s/agent/etc/containerd/config-v3.toml.tmpl";
18
+ /** The EROFS settings that template imports; the store's prepare daemon
19
+ * imports the same file, so both daemons agree on the snapshotter. */
20
+ const CONTAINERD_EROFS_CONFIG_PATH = "/etc/spectest/containerd-erofs.toml";
21
+ /** The socket of the throwaway daemon that prepares the store. */
22
+ const PREPARE_SOCKET = "/run/spectest-prepare.sock";
23
+ /** containerd's CRI plugin lists only images carrying this label. */
24
+ const CRI_MANAGED_LABEL = "io.cri-containerd.image=managed";
15
25
  /**
16
26
  * Written before the server starts and removed by the ready probe.
17
27
  *
@@ -22,36 +32,62 @@ const CLUSTER_STORE_TARGET = "/var/lib/rancher/k3s/agent/containerd";
22
32
  */
23
33
  const BOOT_MARKER = ".spectest-boot-incomplete";
24
34
  /**
25
- * Bring the cluster's image store back into service, or throw it away.
35
+ * The cluster's image store, and how it shares the environment's image
36
+ * cache (`CONTAINER_STORE.md` §12).
26
37
  *
27
- * The store is the cluster's containerd root on a directory that outlives
28
- * this container (`NESTED_STORE_ROOT`), so a build finds coredns, traefik,
29
- * the local-path provisioner and every image the project deployed already
30
- * pulled **and already extracted**. Extraction is 48x the I/O of the
31
- * download (`STICKY_DISKS.md` §7.2), which is why the zot mirror never
32
- * helped here and why an extracted store is the whole prize.
38
+ * The cluster's embedded containerd keeps its root on the **root disk**
39
+ * of the environment's image cache, in a directory of its own
40
+ * (`<root>/spectest-nested/<service>`, bind-mounted at
41
+ * `CLUSTER_STORE_TARGET`), and runs the **EROFS snapshotter** over the
42
+ * same read-only **layers disk** the guest's own containerd uses,
43
+ * mounted into the container at the same path. So a layer is one file
44
+ * on the host whichever runtime holds it, the cluster's pulls land in
45
+ * the lineage's root clone like a `docker pull` does, and the merge
46
+ * reads the cluster's indexes exactly as it reads dockerd's — at every
47
+ * VM teardown, whether the pull happened at env boot or in the middle
48
+ * of a test. Nothing distinguishes the two: a fork's root clone grew,
49
+ * so it is harvested. The next cold build's cluster **adopts** what the
50
+ * merge put on the layers disk under its own namespace (`k8s.io`) with
51
+ * the same helper dockerd's store uses, before the cluster starts.
33
52
  *
34
- * What it inherits with them is one build's crash state: the store was
35
- * captured from a running guest, and the teardown's `docker rm -f` killed
36
- * that cluster where it stood. **Deleting the dead files is not enough,
37
- * and the failure is not subtle.** containerd's index still lists the
38
- * previous build's containers, so the CRI plugin offers them to a kubelet
39
- * whose API server has never heard of them; the kubelet spends its startup
40
- * reconciling containers whose tasks are gone (`Failed to create existing
41
- * container … task not found`) and does not register its own Node in time,
42
- * and the CSI plugin — which waits on that Node with a fixed budget —
43
- * calls `Fatalf`. The kubelet dies, k3s exits 2, and what the author sees
44
- * is a rollout that timed out against a hostname that no longer resolves.
45
- * Measured: 2 of 2 delta restores, against 2 of 2 clean without the store.
53
+ * Extraction is 48x the I/O of the download (`STICKY_DISKS.md` §7.2),
54
+ * which is why a mirror never helped here and an extracted, shared
55
+ * store is the whole prize. Measured on the one project on the old
56
+ * per-environment store: 20.6 s of cluster bring-up per build.
46
57
  *
47
- * So the sweep removes those containers through **containerd itself**,
48
- * which is the only thing that can edit its index. The k3s image ships a
49
- * standalone `containerd` and `ctr`, so a throwaway daemon runs against
50
- * the store with the CRI plugin disabled, deletes the previous build's
51
- * containers and their tasks, and stops. Images, content blobs and
52
- * extracted snapshots — the whole cache — stay. It is the same discipline
53
- * `spectest-image-cache-up` applies to the guest's own store one level up,
54
- * where it sweeps the `moby` namespace before dockerd starts.
58
+ * Three things the container needs for it, all mounted by `k3s()`:
59
+ * golden's static `mkfs.erofs` (the EROFS differ needs it and no k3s
60
+ * image ships one); the adopt helper (POSIX sh, run by busybox); and a
61
+ * containerd config template that imports the EROFS settings — k3s
62
+ * generates its config from `config-v3.toml.tmpl` when present, and an
63
+ * `imports` line on top of the stock template is the one way to add
64
+ * plugin sections without colliding with the ones it writes. The
65
+ * EROFS snapshotter mounts every layer through a loop device, and a
66
+ * privileged container's `/dev` is a static copy that never sees the
67
+ * nodes the kernel allocates later, so the wrapper pre-creates 1024 of
68
+ * them (measured: 0 s).
69
+ *
70
+ * What the store inherits with the cache is one build's crash state:
71
+ * the root was captured from a running guest, and the teardown's
72
+ * `docker rm -f` killed that cluster where it stood. **Deleting the dead
73
+ * files is not enough, and the failure is not subtle.** containerd's
74
+ * index still lists the previous build's containers, so the CRI plugin
75
+ * offers them to a kubelet whose API server has never heard of them; the
76
+ * kubelet spends its startup reconciling containers whose tasks are gone
77
+ * (`Failed to create existing container … task not found`) and does not
78
+ * register its own Node in time, and the CSI plugin — which waits on that
79
+ * Node with a fixed budget — calls `Fatalf`. The kubelet dies, k3s exits
80
+ * 2, and what the author sees is a rollout that timed out against a
81
+ * hostname that no longer resolves. Measured: 2 of 2 delta restores,
82
+ * against 2 of 2 clean without the store.
83
+ *
84
+ * So a **throwaway daemon** on the store, with the CRI plugin disabled,
85
+ * does both jobs before k3s starts: it deletes the previous build's
86
+ * containers and their tasks through containerd itself (the only thing
87
+ * that can edit its index), and it adopts the inbox. Images, content
88
+ * blobs and extracted snapshots — the whole cache — stay. It is the same
89
+ * discipline `spectest-image-cache-up` applies to the guest's own store
90
+ * one level up.
55
91
  *
56
92
  * Two failures cost a cache and never a build:
57
93
  *
@@ -63,15 +99,17 @@ const BOOT_MARKER = ".spectest-boot-incomplete";
63
99
  * so it survives only when a boot never reached a live API server.
64
100
  * That is the backstop for the sweep being incomplete in some way we
65
101
  * have not seen: the cost is one cold build's pulls, and it repairs
66
- * itself. Blacksmith's rule from the other end — they refuse to commit
67
- * a cache when the integrity check *did not run*, not only when it
68
- * failed.
102
+ * itself.
103
+ *
104
+ * Without an image cache (the fake backend, a server that predates it)
105
+ * the store is a per-environment volume as before, on overlayfs, with
106
+ * the same sweep.
69
107
  */
70
- /** Plugins the sweep daemon does not need. Every plugin is startup time,
71
- * and the sweep sits on the critical path of the cluster's boot. CRI is
72
- * off for a second reason: this daemon exists to edit an index, and a CRI
73
- * plugin would set about being a container runtime on a store we are
74
- * seconds from handing to the real one. */
108
+ /** Plugins the prepare daemon does not need. Every plugin is startup
109
+ * time, and the daemon sits on the critical path of the cluster's boot.
110
+ * CRI is off for a second reason: this daemon exists to edit an index,
111
+ * and a CRI plugin would set about being a container runtime on a store
112
+ * we are seconds from handing to the real one. */
75
113
  const SWEEP_DISABLED_PLUGINS = [
76
114
  "io.containerd.grpc.v1.cri",
77
115
  "io.containerd.snapshotter.v1.btrfs",
@@ -82,42 +120,101 @@ const SWEEP_DISABLED_PLUGINS = [
82
120
  "io.containerd.snapshotter.v1.stargz",
83
121
  "io.containerd.snapshotter.v1.fuse-overlayfs",
84
122
  ];
85
- const STORE_SWEEP_SH = `S=${CLUSTER_STORE_TARGET}; M="$S/${BOOT_MARKER}"; A=/run/spectest-sweep.sock; ` +
86
- `sweep_store() { ` +
87
- `printf '%s\\n' 'version = 2' ` +
88
- `'disabled_plugins = [${SWEEP_DISABLED_PLUGINS.map((p) => `"${p}"`).join(", ")}]' ` +
89
- `> /run/spectest-sweep.toml || return 1; ` +
90
- `containerd -c /run/spectest-sweep.toml --root "$S" --state /run/spectest-sweep ` +
91
- `--address "$A" > /run/spectest-sweep.log 2>&1 & ` +
92
- `cd_pid=$!; i=0; ` +
93
- `while [ ! -S "$A" ] && [ $i -lt 600 ]; do sleep 0.1; i=$((i+1)); done; ` +
94
- `if [ ! -S "$A" ]; then kill $cd_pid 2>/dev/null; return 1; fi; ` +
95
- // One invocation for the whole set, not two per container. Each `ctr` is
96
- // a Go binary start plus a gRPC round trip, and a build leaves ~20
97
- // containers behind — the per-container form spent seconds of the
98
- // cluster's boot on process starts alone (measured: ready 3.6s -> 7.0s).
99
- `ids=$(timeout 30 ctr -a "$A" -n k8s.io containers ls -q 2>/dev/null); ` +
100
- `if [ -n "$ids" ]; then ` +
101
- `timeout 60 ctr -a "$A" -n k8s.io tasks rm -f $ids >/dev/null 2>&1 || true; ` +
102
- `timeout 60 ctr -a "$A" -n k8s.io containers rm $ids >/dev/null 2>&1 || true; fi; ` +
103
- // containerd 2.x keeps sandboxes in a store of their own; on the 1.7 k3s
104
- // ships they are ordinary containers and this is a no-op.
105
- `sb=$(timeout 30 ctr -a "$A" -n k8s.io sandboxes ls -q 2>/dev/null); ` +
106
- `[ -n "$sb" ] && timeout 60 ctr -a "$A" -n k8s.io sandboxes rm $sb >/dev/null 2>&1; ` +
107
- `kill $cd_pid 2>/dev/null; wait $cd_pid 2>/dev/null; ` +
108
- `rm -rf "$S/io.containerd.runtime.v2.task" "$S/tmpmounts" /run/spectest-sweep "$A" 2>/dev/null || true; ` +
109
- `return 0; }; ` +
110
- `if [ -e "$M" ]; then ` +
111
- `echo 'spectest: the previous cluster on this image store never became ready; starting from an empty one' >&2; ` +
112
- `rm -rf "$S"/* "$S"/.[!.]* 2>/dev/null || true; ` +
113
- `elif [ -d "$S/io.containerd.metadata.v1.bolt" ]; then ` +
114
- // Timed and printed always: the sweep is inside the cluster's ready time,
115
- // so it is the first thing to suspect if that time moves.
116
- `T0=$(date +%s); ` +
117
- `if sweep_store; then echo "spectest: reused the cluster image store at $S (swept in $(($(date +%s)-T0))s)"; ` +
118
- `else echo 'spectest: could not sweep the inherited image store; starting from an empty one' >&2; ` +
119
- `rm -rf "$S"/* "$S"/.[!.]* 2>/dev/null || true; fi; fi; ` +
120
- `mkdir -p "$S" && : > "$M" 2>/dev/null || true; `;
123
+ const DISABLED_PLUGINS_TOML = `disabled_plugins = [${SWEEP_DISABLED_PLUGINS.map((p) => `"${p}"`).join(", ")}]`;
124
+ /** The containerd config template: the stock one, importing our EROFS
125
+ * settings. `imports` is a top-level key and has to come first. */
126
+ const CONTAINERD_TEMPLATE = `imports = ["${CONTAINERD_EROFS_CONFIG_PATH}"]\n{{ template "base" . }}\n`;
127
+ /**
128
+ * containerd's EROFS settings, the same golden bakes for the guest's
129
+ * own daemon (`local-vms-rootfs.Dockerfile`): the differ first, the
130
+ * transfer service unpacking onto EROFS, and `set_immutable` off because
131
+ * a committed layer's file is a symlink into a read-only disk, on which
132
+ * the immutable ioctl would fail the commit.
133
+ */
134
+ function erofsConfigToml() {
135
+ const arch = process.arch === "arm64" ? "arm64" : process.arch === "x64" ? "amd64" : process.arch;
136
+ return [
137
+ "version = 3",
138
+ '[plugins."io.containerd.snapshotter.v1.erofs"]',
139
+ " set_immutable = false",
140
+ '[plugins."io.containerd.differ.v1.erofs"]',
141
+ ' mkfs_options = ["-T0", "--mkfs-time", "--sort=none"]',
142
+ '[plugins."io.containerd.service.v1.diff-service"]',
143
+ ' default = ["erofs","walking"]',
144
+ '[plugins."io.containerd.transfer.v1.local"]',
145
+ ' [[plugins."io.containerd.transfer.v1.local".unpack_config]]',
146
+ ` platform = "linux/${arch}"`,
147
+ ' snapshotter = "erofs"',
148
+ ' differ = "erofs"',
149
+ "",
150
+ ].join("\n");
151
+ }
152
+ /**
153
+ * The store-preparation half of the command wrapper: sweep the crash
154
+ * state, adopt the cache's inbox (image-cache mode), leave `SNAP` set to
155
+ * the snapshotter the cluster starts on.
156
+ *
157
+ * Runs under busybox `sh`. Every `ctr` call is one invocation for the
158
+ * whole set, not one per container: each is a Go binary start plus a
159
+ * gRPC round trip, and a build leaves ~20 containers behind — the
160
+ * per-container form spent seconds of the cluster's boot on process
161
+ * starts alone (measured: ready 3.6 s -> 7.0 s).
162
+ */
163
+ function storePrepareSh(cache) {
164
+ const daemonConfig = cache
165
+ ? // containerd 2.x (v3 config, EROFS) when the image's containerd has
166
+ // the snapshotter; a 1.7 image (k3s < 1.33) gets the v2 shape it
167
+ // understands and stays on overlayfs.
168
+ `if [ "$EROFS" = 1 ]; then printf '%s\\n' 'version = 3' 'imports = ["${CONTAINERD_EROFS_CONFIG_PATH}"]' '${DISABLED_PLUGINS_TOML}' > /run/spectest-prepare.toml; ` +
169
+ `else printf '%s\\n' 'version = 2' '${DISABLED_PLUGINS_TOML}' > /run/spectest-prepare.toml; fi; `
170
+ : `printf '%s\\n' 'version = 2' '${DISABLED_PLUGINS_TOML}' > /run/spectest-prepare.toml; `;
171
+ const detect = cache
172
+ ? `EROFS=0; if containerd config default 2>/dev/null | grep -q 'snapshotter.v1.erofs'; then EROFS=1; fi; ` +
173
+ `if [ "$EROFS" = 1 ]; then SNAP=erofs; i=0; while [ $i -lt 1024 ]; do [ -e /dev/loop$i ] || mknod /dev/loop$i b 7 $i 2>/dev/null; i=$((i+1)); done; ` +
174
+ `else SNAP=overlayfs; echo 'spectest: this k3s image has no EROFS snapshotter; the cluster keeps its images on overlayfs and shares the image cache as a directory only' >&2; fi; `
175
+ : `EROFS=0; SNAP=overlayfs; `;
176
+ const adopt = cache
177
+ ? `adopt_store() { [ "$EROFS" = 1 ] && [ -f "$L/spectest-inbox/layers.txt" ] || return 0; ` +
178
+ `${STORE_ADOPT_PATH} "$S" "$L" -n k8s.io -a "$A" -l ${CRI_MANAGED_LABEL} 2>&1 | tail -1; }; `
179
+ : `adopt_store() { :; }; `;
180
+ return (`S=${CLUSTER_STORE_TARGET}; M="$S/${BOOT_MARKER}"; A=${PREPARE_SOCKET}; L=${cache ? cache.layers : "/nonexistent"}; ` +
181
+ detect +
182
+ `start_daemon() { ` +
183
+ daemonConfig +
184
+ `containerd -c /run/spectest-prepare.toml --root "$S" --state /run/spectest-prepare ` +
185
+ `--address "$A" > /run/spectest-prepare.log 2>&1 & ` +
186
+ `cd_pid=$!; i=0; ` +
187
+ `while [ ! -S "$A" ] && [ $i -lt 600 ]; do sleep 0.1; i=$((i+1)); done; ` +
188
+ `if [ ! -S "$A" ]; then kill $cd_pid 2>/dev/null; return 1; fi; return 0; }; ` +
189
+ `stop_daemon() { kill $cd_pid 2>/dev/null; wait $cd_pid 2>/dev/null; ` +
190
+ `rm -rf "$S/io.containerd.runtime.v2.task" "$S/tmpmounts" /run/spectest-prepare "$A" 2>/dev/null || true; }; ` +
191
+ `sweep_store() { ` +
192
+ `ids=$(timeout 30 ctr -a "$A" -n k8s.io containers ls -q 2>/dev/null); ` +
193
+ `if [ -n "$ids" ]; then ` +
194
+ `timeout 60 ctr -a "$A" -n k8s.io tasks rm -f $ids >/dev/null 2>&1 || true; ` +
195
+ `timeout 60 ctr -a "$A" -n k8s.io containers rm $ids >/dev/null 2>&1 || true; fi; ` +
196
+ // containerd 2.x keeps sandboxes in a store of their own; on the 1.7
197
+ // k3s ships they are ordinary containers and this is a no-op.
198
+ `sb=$(timeout 30 ctr -a "$A" -n k8s.io sandboxes ls -q 2>/dev/null); ` +
199
+ `[ -n "$sb" ] && timeout 60 ctr -a "$A" -n k8s.io sandboxes rm $sb >/dev/null 2>&1; return 0; }; ` +
200
+ adopt +
201
+ `if [ -e "$M" ]; then ` +
202
+ `echo 'spectest: the previous cluster on this image store never became ready; starting from an empty one' >&2; ` +
203
+ `rm -rf "$S"/* "$S"/.[!.]* 2>/dev/null || true; fi; ` +
204
+ `mkdir -p "$S"; ` +
205
+ `had_state=0; [ -d "$S/io.containerd.metadata.v1.bolt" ] && had_state=1; ` +
206
+ `T0=$(date +%s); ` +
207
+ `if [ "$had_state" = 1 ] || [ "$EROFS" = 1 ]; then ` +
208
+ `if start_daemon; then ` +
209
+ `[ "$had_state" = 1 ] && sweep_store; adopt_store; stop_daemon; ` +
210
+ // Timed and printed always: this is inside the cluster's ready time,
211
+ // so it is the first thing to suspect if that time moves.
212
+ `echo "spectest: cluster image store at $S ready on $SNAP (prepared in $(($(date +%s)-T0))s)"; ` +
213
+ `elif [ "$had_state" = 1 ]; then ` +
214
+ `echo 'spectest: could not sweep the inherited image store; starting from an empty one' >&2; ` +
215
+ `rm -rf "$S"/* "$S"/.[!.]* 2>/dev/null || true; fi; fi; ` +
216
+ `: > "$M" 2>/dev/null || true; `);
217
+ }
121
218
  /**
122
219
  * `extraArgs` entries the component overrides anyway. Passing one of
123
220
  * these looks like it works — the flag really is appended to the k3s
@@ -1286,11 +1383,14 @@ async function collectDiagnostics(client, namespace) {
1286
1383
  * so neither cold-cache nor warm starts re-pull. Any `opts.version`
1287
1384
  * works — there's no base-snapshot release to keep in sync with.
1288
1385
  *
1289
- * **Why v1.32.x:** kube-proxy's `nftables` proxy mode is GA in k8s 1.32
1290
- * (beta in 1.31, alpha-gated in 1.30). The component runs kube-proxy in
1291
- * that mode; dropping below 1.31 falls back to the iptables path.
1386
+ * **Why v1.33.x:** k3s 1.33.5 is the first line embedding containerd
1387
+ * 2.1 (2.1.4-k3s1), whose in-tree EROFS snapshotter is what lets the
1388
+ * cluster share the environment's image cache (see the store section
1389
+ * above); 1.32 ships containerd 1.7. kube-proxy's `nftables` proxy mode
1390
+ * is GA since k8s 1.32 (beta in 1.31, alpha-gated in 1.30); dropping
1391
+ * below 1.31 falls back to the iptables path.
1292
1392
  */
1293
- const DEFAULT_K3S_VERSION = "v1.32.1-k3s1";
1393
+ const DEFAULT_K3S_VERSION = "v1.33.5-k3s1";
1294
1394
  export function k3s(opts = {}) {
1295
1395
  const version = opts.version ?? DEFAULT_K3S_VERSION;
1296
1396
  const extra = opts.extraArgs ?? [];
@@ -1306,6 +1406,10 @@ export function k3s(opts = {}) {
1306
1406
  // the in-cluster registry). Seeded via `files` because k3s reads it
1307
1407
  // only at startup, before any setup hook could run.
1308
1408
  const registriesYaml = buildRegistriesYaml(registryEnabled);
1409
+ // The environment's image cache, when this VM carries one: the
1410
+ // cluster's store goes on its root disk and its layers come off the
1411
+ // shared layers disk (see the store section above).
1412
+ const cache = imageCachePathsSync();
1309
1413
  const serverArgs = [
1310
1414
  "k3s",
1311
1415
  "server",
@@ -1387,8 +1491,10 @@ export function k3s(opts = {}) {
1387
1491
  // is affected.
1388
1492
  "mount --make-rshared / 2>/dev/null || " +
1389
1493
  "echo 'spectest: could not make / rshared; CSI node plugins may fail to publish volumes' >&2; " +
1390
- STORE_SWEEP_SH +
1391
- `exec ${serverArgs}`;
1494
+ storePrepareSh(cache) +
1495
+ // The snapshotter is decided by the wrapper: EROFS on the image cache
1496
+ // when the image's containerd has it, overlayfs otherwise.
1497
+ `exec ${serverArgs} --snapshotter=$SNAP`;
1392
1498
  // Plain /readyz probe. On a warm zot cache the cluster's images are
1393
1499
  // already local, so the first boot completes in seconds; the
1394
1500
  // first-ever boot on a cold-cache host pulls through the mirror and
@@ -1414,37 +1520,30 @@ export function k3s(opts = {}) {
1414
1520
  // the same netns and are reachable the same way, without appearing
1415
1521
  // in this list (it's documentation, not a firewall).
1416
1522
  ports: registryEnabled ? [80, 443, 6443, K3S_REGISTRY_PORT] : [80, 443, 6443],
1417
- // The cluster's image store, on a volume that outlives this
1418
- // container. Without it the store is the container's writable layer,
1419
- // which `docker rm -f` deletes at teardown — so every build re-pulls
1420
- // and, far more expensively, re-extracts coredns, traefik, the
1421
- // local-path provisioner and everything the project deploys.
1422
- // Measured on the one project on the image cache: 20.6 s per
1423
- // build, identical on the cold and delta tiers, i.e. the one term
1424
- // neither the delta restore nor the cache image store reached.
1425
- //
1426
- // `cache` names a cache disk, which is the ordinary way any project
1427
- // asks for one — there is nothing Kubernetes-shaped in the platform
1428
- // for this. Two clusters in one environment may name the same disk
1429
- // and still get separate containerd roots, because a volume is rooted
1430
- // per service on whatever disk it names; sharing a root would corrupt
1431
- // it, since two containerd daemons cannot share a store (bolt takes
1432
- // an exclusive lock).
1433
- //
1434
- // This reverses a NOTE that stood here for a year: a fresh k3s server
1435
- // over a store killed un-cleanly by `docker rm -f` wedged the
1436
- // apiserver minutes in, and nothing recovered. What was missing was
1437
- // the discipline `base.rs::IMAGE_CACHE_UP_SH` applies to the guest's own
1438
- // store — sweep the crash state, and never inherit a store that did
1439
- // not work. Both are in STORE_SWEEP_SH.
1440
- volumes: [{ target: CLUSTER_STORE_TARGET }],
1441
- ...(registriesYaml
1442
- ? {
1443
- files: [
1444
- { path: "/etc/rancher/k3s/registries.yaml", content: registriesYaml },
1445
- ],
1446
- }
1447
- : {}),
1523
+ // The cluster's image store (see the store section above): its
1524
+ // containerd root on the image cache's root disk, the shared layers
1525
+ // disk read-only at its guest path, golden's mkfs.erofs and the adopt
1526
+ // helper — or, with no cache, a per-environment volume. Two clusters
1527
+ // in one environment get separate roots either way (the service
1528
+ // token names the directory), because two containerd daemons cannot
1529
+ // share a store: bolt takes an exclusive lock.
1530
+ volumes: cache
1531
+ ? [
1532
+ { source: nestedStoreDir(cache, SELF_SERVICE_TOKEN), target: CLUSTER_STORE_TARGET },
1533
+ { source: cache.layers, target: cache.layers, readOnly: true },
1534
+ { source: MKFS_EROFS_PATH, target: MKFS_EROFS_PATH, readOnly: true },
1535
+ { source: STORE_ADOPT_PATH, target: STORE_ADOPT_PATH, readOnly: true },
1536
+ ]
1537
+ : [{ target: CLUSTER_STORE_TARGET }],
1538
+ files: [
1539
+ ...(registriesYaml ? [{ path: "/etc/rancher/k3s/registries.yaml", content: registriesYaml }] : []),
1540
+ ...(cache
1541
+ ? [
1542
+ { path: CONTAINERD_TEMPLATE_PATH, content: CONTAINERD_TEMPLATE },
1543
+ { path: CONTAINERD_EROFS_CONFIG_PATH, content: erofsConfigToml() },
1544
+ ]
1545
+ : []),
1546
+ ],
1448
1547
  readyCheck: {
1449
1548
  type: "exec",
1450
1549
  command: readyCmd,
package/dist/daemon.js CHANGED
@@ -50,6 +50,7 @@ import { certCovers as hostmatchCertCovers, hostWithoutPort, matchRoute, selectC
50
50
  import { INGRESS_HTTPS_PORT, INGRESS_HTTP_PORT, bindRoute, clearTables, emptyTables, planBind, registryTarget, routesFor, unbindRoute, } from "./harness/ingress-table.js";
51
51
  import { startTlsTerminator } from "./harness/tls-terminator.js";
52
52
  import { runContainerArgs } from "./harness/container-run.js";
53
+ import { IMAGE_CACHE_MANIFEST, imageCachePathsSync, isOnImageCache } from "./harness/image-cache.js";
53
54
  import { assertAbsolute, certificateHostnames, defaultKeyMode, expandServiceToken, isNoopChown, mountFlag, needsIdTables, numericId, resolveChownIds, } from "./harness/file-mounts.js";
54
55
  import { conflict, notFound, requireString, } from "./harness/methods.js";
55
56
  import { openTerminal } from "./terminal.js";
@@ -448,7 +449,6 @@ async function hasBuildx() {
448
449
  // in-VM buildkitd keeps its exported cache there too, so a fresh VM finds
449
450
  // every layer it built before. Detected once; if the daemon will not
450
451
  // start, dockerd's own BuildKit builds instead.
451
- const IMAGE_CACHE_MANIFEST = "/run/spectest-image-cache.json";
452
452
  const LOCAL_BUILDER_NAME = "spectest-local";
453
453
  const LOCAL_BUILDKIT_ADDR = "tcp://127.0.0.1:1234";
454
454
  /** The bring-up script the cache base bakes (`base.rs::BUILDKITD_UP_SH`). */
@@ -457,16 +457,7 @@ const BUILDKITD_UP_PATH = "/usr/local/bin/spectest-buildkitd-up";
457
457
  * root (read-write, this VM's own) and the layers disk (read-only, shared
458
458
  * by every VM of a generation). `null` when the VM carries no cache. */
459
459
  async function imageCachePaths() {
460
- try {
461
- const raw = await fs.readFile(IMAGE_CACHE_MANIFEST, "utf8");
462
- const parsed = JSON.parse(raw);
463
- const root = (parsed.disks ?? []).find((d) => d.role === "root" && d.path)?.path;
464
- const layers = (parsed.disks ?? []).find((d) => d.role === "layers" && d.path)?.path;
465
- return root && layers ? { root, layers } : null;
466
- }
467
- catch {
468
- return null;
469
- }
460
+ return imageCachePathsSync(IMAGE_CACHE_MANIFEST);
470
461
  }
471
462
  let _localBuilder;
472
463
  /**
@@ -602,7 +593,10 @@ async function ensureVolumes(svc) {
602
593
  // serving older SDKs. Leaving it out of the manifest is what
603
594
  // protects a project running this SDK against a server whose
604
595
  // teardown guard predates it.
605
- const durable = host.startsWith("/var/cache/spectest/");
596
+ // A directory on a cache disk is the same kind of thing: a nested
597
+ // runtime's containerd root (`k3s()`), kept as a cache by the
598
+ // lineage exactly as the container store one level up is.
599
+ const durable = host.startsWith("/var/cache/spectest/") || isOnImageCache(host, imageCachePathsSync(IMAGE_CACHE_MANIFEST));
606
600
  if (vol.source?.startsWith("/") && !durable) {
607
601
  await recordAbsoluteVolumeDir(host);
608
602
  }
@@ -1121,12 +1115,18 @@ async function runServiceBuild(name, image, tag, caSuffix) {
1121
1115
  // intermediate layer, uncompressed so an import never inflates,
1122
1116
  // and one tag per service so exports do not replace each other.
1123
1117
  const cacheTag = name.replace(/[^a-z0-9-]/gi, "-").toLowerCase();
1118
+ // LANDMINE: the local cache importer reads the `latest` entry of
1119
+ // the directory's index unless told otherwise, and the exporter
1120
+ // below writes this service's entry under `tag=<service>`. An
1121
+ // import without the same tag misses every time and every
1122
+ // `RUN` re-executes on a seeded cache (seen on the first deploy,
1123
+ // 2026-09-05: the disk carried the blobs, the build used none).
1124
1124
  buildArgs = [
1125
1125
  "buildx", "build",
1126
1126
  "--builder", LOCAL_BUILDER_NAME,
1127
1127
  "--progress=plain",
1128
1128
  "--output", `type=image,name=${qualifyImageRef(tag)},unpack=true`,
1129
- ..._localCacheImports.flatMap((src) => ["--cache-from", `type=local,src=${src}`]),
1129
+ ..._localCacheImports.flatMap((src) => ["--cache-from", `type=local,src=${src},tag=${cacheTag}`]),
1130
1130
  "--cache-to", `type=local,dest=${_localCacheDir},mode=max,compression=uncompressed,force-compression=true,tag=${cacheTag}`,
1131
1131
  ...argFlags,
1132
1132
  "-f", dfPath, WORKSPACE,
@@ -0,0 +1,48 @@
1
+ /**
2
+ * The image cache as the guest sees it (`CONTAINER_STORE.md`).
3
+ *
4
+ * The control plane writes one manifest per VM at start naming the two
5
+ * cache disks it attached: the **root** (containerd's own root,
6
+ * read-write, this VM's clone) and the **layers** disk (read-only, one
7
+ * EROFS file per layer, shared by every VM of a generation). Both paths
8
+ * are fixed by the control plane; the manifest is how a harness learns
9
+ * whether this VM carries a cache at all (the fake backend does not,
10
+ * and neither does a server older than the cache).
11
+ *
12
+ * Read synchronously as well as asynchronously: a component's service
13
+ * definition is built inside `defineEnvironment`, which is synchronous,
14
+ * and `k3s()` decides its mounts there.
15
+ */
16
+ /** Written by `env.rs` before the harness starts. */
17
+ export declare const IMAGE_CACHE_MANIFEST = "/run/spectest-image-cache.json";
18
+ /** Where the cache's paths are, when this VM carries one. */
19
+ export interface ImageCachePaths {
20
+ /** containerd's root: read-write, this VM's own clone. */
21
+ root: string;
22
+ /** The layers disk: read-only, shared by every VM of a generation. */
23
+ layers: string;
24
+ }
25
+ /**
26
+ * Directory under the root disk holding a nested runtime's containerd
27
+ * root, one per service: `<root>/spectest-nested/<service>`. The merge
28
+ * (`image_cache/merge.rs::NESTED_DIR`) reads every store it finds there
29
+ * exactly as it reads the disk's own.
30
+ */
31
+ export declare const NESTED_STORES_DIR = "spectest-nested";
32
+ /** The guest's static `mkfs.erofs`, which a nested runtime's EROFS
33
+ * differ needs and no runtime image ships. */
34
+ export declare const MKFS_EROFS_PATH = "/usr/local/bin/mkfs.erofs";
35
+ /** The guest's adopt helper (`base.rs::STORE_ADOPT_SH`), POSIX sh so a
36
+ * nested runtime's busybox can run the same file. */
37
+ export declare const STORE_ADOPT_PATH = "/usr/local/bin/spectest-store-adopt";
38
+ /** The cache's paths, or `null` when this VM carries none.
39
+ * `SPECTEST_IMAGE_CACHE_MANIFEST` points a test at another file. */
40
+ export declare function imageCachePathsSync(manifest?: string): ImageCachePaths | null;
41
+ /** The host directory a nested runtime keeps its containerd root in. */
42
+ export declare function nestedStoreDir(paths: ImageCachePaths, service: string): string;
43
+ /**
44
+ * Is a volume's host path on a cache disk? Such a directory is a cache
45
+ * the lineage keeps — like the container store one level up — and the
46
+ * delta-restore teardown must not wipe it.
47
+ */
48
+ export declare function isOnImageCache(hostPath: string, paths: ImageCachePaths | null): boolean;