@specific.dev/spectest 0.71.1 → 0.73.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -34,6 +34,13 @@ import type {
34
34
  SpectestContext,
35
35
  } from "../index.js";
36
36
  import { dnsName, provides, SELF_SERVICE_TOKEN } from "../index.js";
37
+ import {
38
+ MKFS_EROFS_PATH,
39
+ STORE_ADOPT_PATH,
40
+ imageCachePathsSync,
41
+ nestedStoreDir,
42
+ type ImageCachePaths,
43
+ } from "../harness/image-cache.js";
37
44
  import { deepUnwrap, readTag, wrap } from "../inspect.js";
38
45
  import type { Wrapped } from "../inspect.js";
39
46
  import { recorderAnnotate, recorderRemove } from "../recorder.js";
@@ -44,7 +51,13 @@ export type {
44
51
  } from "@kubernetes/client-node";
45
52
 
46
53
  export interface K3sOptions {
47
- /** Image tag for the official `rancher/k3s` image. Default `"v1.30.6-k3s1"`. */
54
+ /**
55
+ * Image tag for the official `rancher/k3s` image. Default
56
+ * `"v1.33.5-k3s1"`. Any tag works; a k3s older than 1.33 embeds a
57
+ * containerd (1.7) without the EROFS snapshotter, so its cluster keeps
58
+ * its images on overlayfs and shares the environment's image cache only
59
+ * as a directory that survives — see `k3s()`'s store section.
60
+ */
48
61
  version?: string;
49
62
  /**
50
63
  * Extra arguments appended to `k3s server`. Useful for `--tls-san=...`,
@@ -144,7 +157,15 @@ const K3S_REGISTRY_PORT = 5000;
144
157
 
145
158
  /** The cluster's own containerd root, inside the k3s container. */
146
159
  const CLUSTER_STORE_TARGET = "/var/lib/rancher/k3s/agent/containerd";
147
-
160
+ /** Where k3s (containerd 2.x layout) reads a containerd config template. */
161
+ const CONTAINERD_TEMPLATE_PATH = "/var/lib/rancher/k3s/agent/etc/containerd/config-v3.toml.tmpl";
162
+ /** The EROFS settings that template imports; the store's prepare daemon
163
+ * imports the same file, so both daemons agree on the snapshotter. */
164
+ const CONTAINERD_EROFS_CONFIG_PATH = "/etc/spectest/containerd-erofs.toml";
165
+ /** The socket of the throwaway daemon that prepares the store. */
166
+ const PREPARE_SOCKET = "/run/spectest-prepare.sock";
167
+ /** containerd's CRI plugin lists only images carrying this label. */
168
+ const CRI_MANAGED_LABEL = "io.cri-containerd.image=managed";
148
169
 
149
170
  /**
150
171
  * Written before the server starts and removed by the ready probe.
@@ -155,37 +176,64 @@ const CLUSTER_STORE_TARGET = "/var/lib/rancher/k3s/agent/containerd";
155
176
  * cache when the integrity check *did not run*, not only when it failed.
156
177
  */
157
178
  const BOOT_MARKER = ".spectest-boot-incomplete";
179
+
158
180
  /**
159
- * Bring the cluster's image store back into service, or throw it away.
181
+ * The cluster's image store, and how it shares the environment's image
182
+ * cache (`CONTAINER_STORE.md` §12).
183
+ *
184
+ * The cluster's embedded containerd keeps its root on the **root disk**
185
+ * of the environment's image cache, in a directory of its own
186
+ * (`<root>/spectest-nested/<service>`, bind-mounted at
187
+ * `CLUSTER_STORE_TARGET`), and runs the **EROFS snapshotter** over the
188
+ * same read-only **layers disk** the guest's own containerd uses,
189
+ * mounted into the container at the same path. So a layer is one file
190
+ * on the host whichever runtime holds it, the cluster's pulls land in
191
+ * the lineage's root clone like a `docker pull` does, and the merge
192
+ * reads the cluster's indexes exactly as it reads dockerd's — at every
193
+ * VM teardown, whether the pull happened at env boot or in the middle
194
+ * of a test. Nothing distinguishes the two: a fork's root clone grew,
195
+ * so it is harvested. The next cold build's cluster **adopts** what the
196
+ * merge put on the layers disk under its own namespace (`k8s.io`) with
197
+ * the same helper dockerd's store uses, before the cluster starts.
160
198
  *
161
- * The store is the cluster's containerd root on a directory that outlives
162
- * this container (`NESTED_STORE_ROOT`), so a build finds coredns, traefik,
163
- * the local-path provisioner and every image the project deployed already
164
- * pulled **and already extracted**. Extraction is 48x the I/O of the
165
- * download (`STICKY_DISKS.md` §7.2), which is why the zot mirror never
166
- * helped here and why an extracted store is the whole prize.
199
+ * Extraction is 48x the I/O of the download (`STICKY_DISKS.md` §7.2),
200
+ * which is why a mirror never helped here and an extracted, shared
201
+ * store is the whole prize. Measured on the one project on the old
202
+ * per-environment store: 20.6 s of cluster bring-up per build.
167
203
  *
168
- * What it inherits with them is one build's crash state: the store was
169
- * captured from a running guest, and the teardown's `docker rm -f` killed
170
- * that cluster where it stood. **Deleting the dead files is not enough,
171
- * and the failure is not subtle.** containerd's index still lists the
172
- * previous build's containers, so the CRI plugin offers them to a kubelet
173
- * whose API server has never heard of them; the kubelet spends its startup
174
- * reconciling containers whose tasks are gone (`Failed to create existing
175
- * container … task not found`) and does not register its own Node in time,
176
- * and the CSI plugin — which waits on that Node with a fixed budget —
177
- * calls `Fatalf`. The kubelet dies, k3s exits 2, and what the author sees
178
- * is a rollout that timed out against a hostname that no longer resolves.
179
- * Measured: 2 of 2 delta restores, against 2 of 2 clean without the store.
204
+ * Three things the container needs for it, all mounted by `k3s()`:
205
+ * golden's static `mkfs.erofs` (the EROFS differ needs it and no k3s
206
+ * image ships one); the adopt helper (POSIX sh, run by busybox); and a
207
+ * containerd config template that imports the EROFS settings — k3s
208
+ * generates its config from `config-v3.toml.tmpl` when present, and an
209
+ * `imports` line on top of the stock template is the one way to add
210
+ * plugin sections without colliding with the ones it writes. The
211
+ * EROFS snapshotter mounts every layer through a loop device, and a
212
+ * privileged container's `/dev` is a static copy that never sees the
213
+ * nodes the kernel allocates later, so the wrapper pre-creates 1024 of
214
+ * them (measured: 0 s).
180
215
  *
181
- * So the sweep removes those containers through **containerd itself**,
182
- * which is the only thing that can edit its index. The k3s image ships a
183
- * standalone `containerd` and `ctr`, so a throwaway daemon runs against
184
- * the store with the CRI plugin disabled, deletes the previous build's
185
- * containers and their tasks, and stops. Images, content blobs and
186
- * extracted snapshots — the whole cache — stay. It is the same discipline
187
- * `spectest-image-cache-up` applies to the guest's own store one level up,
188
- * where it sweeps the `moby` namespace before dockerd starts.
216
+ * What the store inherits with the cache is one build's crash state:
217
+ * the root was captured from a running guest, and the teardown's
218
+ * `docker rm -f` killed that cluster where it stood. **Deleting the dead
219
+ * files is not enough, and the failure is not subtle.** containerd's
220
+ * index still lists the previous build's containers, so the CRI plugin
221
+ * offers them to a kubelet whose API server has never heard of them; the
222
+ * kubelet spends its startup reconciling containers whose tasks are gone
223
+ * (`Failed to create existing container … task not found`) and does not
224
+ * register its own Node in time, and the CSI plugin — which waits on that
225
+ * Node with a fixed budget — calls `Fatalf`. The kubelet dies, k3s exits
226
+ * 2, and what the author sees is a rollout that timed out against a
227
+ * hostname that no longer resolves. Measured: 2 of 2 delta restores,
228
+ * against 2 of 2 clean without the store.
229
+ *
230
+ * So a **throwaway daemon** on the store, with the CRI plugin disabled,
231
+ * does both jobs before k3s starts: it deletes the previous build's
232
+ * containers and their tasks through containerd itself (the only thing
233
+ * that can edit its index), and it adopts the inbox. Images, content
234
+ * blobs and extracted snapshots — the whole cache — stay. It is the same
235
+ * discipline `spectest-image-cache-up` applies to the guest's own store
236
+ * one level up.
189
237
  *
190
238
  * Two failures cost a cache and never a build:
191
239
  *
@@ -197,15 +245,17 @@ const BOOT_MARKER = ".spectest-boot-incomplete";
197
245
  * so it survives only when a boot never reached a live API server.
198
246
  * That is the backstop for the sweep being incomplete in some way we
199
247
  * have not seen: the cost is one cold build's pulls, and it repairs
200
- * itself. Blacksmith's rule from the other end — they refuse to commit
201
- * a cache when the integrity check *did not run*, not only when it
202
- * failed.
248
+ * itself.
249
+ *
250
+ * Without an image cache (the fake backend, a server that predates it)
251
+ * the store is a per-environment volume as before, on overlayfs, with
252
+ * the same sweep.
203
253
  */
204
- /** Plugins the sweep daemon does not need. Every plugin is startup time,
205
- * and the sweep sits on the critical path of the cluster's boot. CRI is
206
- * off for a second reason: this daemon exists to edit an index, and a CRI
207
- * plugin would set about being a container runtime on a store we are
208
- * seconds from handing to the real one. */
254
+ /** Plugins the prepare daemon does not need. Every plugin is startup
255
+ * time, and the daemon sits on the critical path of the cluster's boot.
256
+ * CRI is off for a second reason: this daemon exists to edit an index,
257
+ * and a CRI plugin would set about being a container runtime on a store
258
+ * we are seconds from handing to the real one. */
209
259
  const SWEEP_DISABLED_PLUGINS = [
210
260
  "io.containerd.grpc.v1.cri",
211
261
  "io.containerd.snapshotter.v1.btrfs",
@@ -217,43 +267,106 @@ const SWEEP_DISABLED_PLUGINS = [
217
267
  "io.containerd.snapshotter.v1.fuse-overlayfs",
218
268
  ];
219
269
 
220
- const STORE_SWEEP_SH =
221
- `S=${CLUSTER_STORE_TARGET}; M="$S/${BOOT_MARKER}"; A=/run/spectest-sweep.sock; ` +
222
- `sweep_store() { ` +
223
- `printf '%s\\n' 'version = 2' ` +
224
- `'disabled_plugins = [${SWEEP_DISABLED_PLUGINS.map((p) => `"${p}"`).join(", ")}]' ` +
225
- `> /run/spectest-sweep.toml || return 1; ` +
226
- `containerd -c /run/spectest-sweep.toml --root "$S" --state /run/spectest-sweep ` +
227
- `--address "$A" > /run/spectest-sweep.log 2>&1 & ` +
228
- `cd_pid=$!; i=0; ` +
229
- `while [ ! -S "$A" ] && [ $i -lt 600 ]; do sleep 0.1; i=$((i+1)); done; ` +
230
- `if [ ! -S "$A" ]; then kill $cd_pid 2>/dev/null; return 1; fi; ` +
231
- // One invocation for the whole set, not two per container. Each `ctr` is
232
- // a Go binary start plus a gRPC round trip, and a build leaves ~20
233
- // containers behind — the per-container form spent seconds of the
234
- // cluster's boot on process starts alone (measured: ready 3.6s -> 7.0s).
235
- `ids=$(timeout 30 ctr -a "$A" -n k8s.io containers ls -q 2>/dev/null); ` +
236
- `if [ -n "$ids" ]; then ` +
237
- `timeout 60 ctr -a "$A" -n k8s.io tasks rm -f $ids >/dev/null 2>&1 || true; ` +
238
- `timeout 60 ctr -a "$A" -n k8s.io containers rm $ids >/dev/null 2>&1 || true; fi; ` +
239
- // containerd 2.x keeps sandboxes in a store of their own; on the 1.7 k3s
240
- // ships they are ordinary containers and this is a no-op.
241
- `sb=$(timeout 30 ctr -a "$A" -n k8s.io sandboxes ls -q 2>/dev/null); ` +
242
- `[ -n "$sb" ] && timeout 60 ctr -a "$A" -n k8s.io sandboxes rm $sb >/dev/null 2>&1; ` +
243
- `kill $cd_pid 2>/dev/null; wait $cd_pid 2>/dev/null; ` +
244
- `rm -rf "$S/io.containerd.runtime.v2.task" "$S/tmpmounts" /run/spectest-sweep "$A" 2>/dev/null || true; ` +
245
- `return 0; }; ` +
246
- `if [ -e "$M" ]; then ` +
247
- `echo 'spectest: the previous cluster on this image store never became ready; starting from an empty one' >&2; ` +
248
- `rm -rf "$S"/* "$S"/.[!.]* 2>/dev/null || true; ` +
249
- `elif [ -d "$S/io.containerd.metadata.v1.bolt" ]; then ` +
250
- // Timed and printed always: the sweep is inside the cluster's ready time,
251
- // so it is the first thing to suspect if that time moves.
252
- `T0=$(date +%s); ` +
253
- `if sweep_store; then echo "spectest: reused the cluster image store at $S (swept in $(($(date +%s)-T0))s)"; ` +
254
- `else echo 'spectest: could not sweep the inherited image store; starting from an empty one' >&2; ` +
255
- `rm -rf "$S"/* "$S"/.[!.]* 2>/dev/null || true; fi; fi; ` +
256
- `mkdir -p "$S" && : > "$M" 2>/dev/null || true; `;
270
+ const DISABLED_PLUGINS_TOML = `disabled_plugins = [${SWEEP_DISABLED_PLUGINS.map((p) => `"${p}"`).join(", ")}]`;
271
+
272
+ /** The containerd config template: the stock one, importing our EROFS
273
+ * settings. `imports` is a top-level key and has to come first. */
274
+ const CONTAINERD_TEMPLATE = `imports = ["${CONTAINERD_EROFS_CONFIG_PATH}"]\n{{ template "base" . }}\n`;
275
+
276
+ /**
277
+ * containerd's EROFS settings, the same golden bakes for the guest's
278
+ * own daemon (`local-vms-rootfs.Dockerfile`): the differ first, the
279
+ * transfer service unpacking onto EROFS, and `set_immutable` off because
280
+ * a committed layer's file is a symlink into a read-only disk, on which
281
+ * the immutable ioctl would fail the commit.
282
+ */
283
+ function erofsConfigToml(): string {
284
+ const arch = process.arch === "arm64" ? "arm64" : process.arch === "x64" ? "amd64" : process.arch;
285
+ return [
286
+ "version = 3",
287
+ '[plugins."io.containerd.snapshotter.v1.erofs"]',
288
+ " set_immutable = false",
289
+ '[plugins."io.containerd.differ.v1.erofs"]',
290
+ ' mkfs_options = ["-T0", "--mkfs-time", "--sort=none"]',
291
+ '[plugins."io.containerd.service.v1.diff-service"]',
292
+ ' default = ["erofs","walking"]',
293
+ '[plugins."io.containerd.transfer.v1.local"]',
294
+ ' [[plugins."io.containerd.transfer.v1.local".unpack_config]]',
295
+ ` platform = "linux/${arch}"`,
296
+ ' snapshotter = "erofs"',
297
+ ' differ = "erofs"',
298
+ "",
299
+ ].join("\n");
300
+ }
301
+
302
+ /**
303
+ * The store-preparation half of the command wrapper: sweep the crash
304
+ * state, adopt the cache's inbox (image-cache mode), leave `SNAP` set to
305
+ * the snapshotter the cluster starts on.
306
+ *
307
+ * Runs under busybox `sh`. Every `ctr` call is one invocation for the
308
+ * whole set, not one per container: each is a Go binary start plus a
309
+ * gRPC round trip, and a build leaves ~20 containers behind — the
310
+ * per-container form spent seconds of the cluster's boot on process
311
+ * starts alone (measured: ready 3.6 s -> 7.0 s).
312
+ */
313
+ function storePrepareSh(cache: ImageCachePaths | null): string {
314
+ const daemonConfig = cache
315
+ ? // containerd 2.x (v3 config, EROFS) when the image's containerd has
316
+ // the snapshotter; a 1.7 image (k3s < 1.33) gets the v2 shape it
317
+ // understands and stays on overlayfs.
318
+ `if [ "$EROFS" = 1 ]; then printf '%s\\n' 'version = 3' 'imports = ["${CONTAINERD_EROFS_CONFIG_PATH}"]' '${DISABLED_PLUGINS_TOML}' > /run/spectest-prepare.toml; ` +
319
+ `else printf '%s\\n' 'version = 2' '${DISABLED_PLUGINS_TOML}' > /run/spectest-prepare.toml; fi; `
320
+ : `printf '%s\\n' 'version = 2' '${DISABLED_PLUGINS_TOML}' > /run/spectest-prepare.toml; `;
321
+ const detect = cache
322
+ ? `EROFS=0; if containerd config default 2>/dev/null | grep -q 'snapshotter.v1.erofs'; then EROFS=1; fi; ` +
323
+ `if [ "$EROFS" = 1 ]; then SNAP=erofs; i=0; while [ $i -lt 1024 ]; do [ -e /dev/loop$i ] || mknod /dev/loop$i b 7 $i 2>/dev/null; i=$((i+1)); done; ` +
324
+ `else SNAP=overlayfs; echo 'spectest: this k3s image has no EROFS snapshotter; the cluster keeps its images on overlayfs and shares the image cache as a directory only' >&2; fi; `
325
+ : `EROFS=0; SNAP=overlayfs; `;
326
+ const adopt = cache
327
+ ? `adopt_store() { [ "$EROFS" = 1 ] && [ -f "$L/spectest-inbox/layers.txt" ] || return 0; ` +
328
+ `${STORE_ADOPT_PATH} "$S" "$L" -n k8s.io -a "$A" -l ${CRI_MANAGED_LABEL} 2>&1 | tail -1; }; `
329
+ : `adopt_store() { :; }; `;
330
+ return (
331
+ `S=${CLUSTER_STORE_TARGET}; M="$S/${BOOT_MARKER}"; A=${PREPARE_SOCKET}; L=${cache ? cache.layers : "/nonexistent"}; ` +
332
+ detect +
333
+ `start_daemon() { ` +
334
+ daemonConfig +
335
+ `containerd -c /run/spectest-prepare.toml --root "$S" --state /run/spectest-prepare ` +
336
+ `--address "$A" > /run/spectest-prepare.log 2>&1 & ` +
337
+ `cd_pid=$!; i=0; ` +
338
+ `while [ ! -S "$A" ] && [ $i -lt 600 ]; do sleep 0.1; i=$((i+1)); done; ` +
339
+ `if [ ! -S "$A" ]; then kill $cd_pid 2>/dev/null; return 1; fi; return 0; }; ` +
340
+ `stop_daemon() { kill $cd_pid 2>/dev/null; wait $cd_pid 2>/dev/null; ` +
341
+ `rm -rf "$S/io.containerd.runtime.v2.task" "$S/tmpmounts" /run/spectest-prepare "$A" 2>/dev/null || true; }; ` +
342
+ `sweep_store() { ` +
343
+ `ids=$(timeout 30 ctr -a "$A" -n k8s.io containers ls -q 2>/dev/null); ` +
344
+ `if [ -n "$ids" ]; then ` +
345
+ `timeout 60 ctr -a "$A" -n k8s.io tasks rm -f $ids >/dev/null 2>&1 || true; ` +
346
+ `timeout 60 ctr -a "$A" -n k8s.io containers rm $ids >/dev/null 2>&1 || true; fi; ` +
347
+ // containerd 2.x keeps sandboxes in a store of their own; on the 1.7
348
+ // k3s ships they are ordinary containers and this is a no-op.
349
+ `sb=$(timeout 30 ctr -a "$A" -n k8s.io sandboxes ls -q 2>/dev/null); ` +
350
+ `[ -n "$sb" ] && timeout 60 ctr -a "$A" -n k8s.io sandboxes rm $sb >/dev/null 2>&1; return 0; }; ` +
351
+ adopt +
352
+ `if [ -e "$M" ]; then ` +
353
+ `echo 'spectest: the previous cluster on this image store never became ready; starting from an empty one' >&2; ` +
354
+ `rm -rf "$S"/* "$S"/.[!.]* 2>/dev/null || true; fi; ` +
355
+ `mkdir -p "$S"; ` +
356
+ `had_state=0; [ -d "$S/io.containerd.metadata.v1.bolt" ] && had_state=1; ` +
357
+ `T0=$(date +%s); ` +
358
+ `if [ "$had_state" = 1 ] || [ "$EROFS" = 1 ]; then ` +
359
+ `if start_daemon; then ` +
360
+ `[ "$had_state" = 1 ] && sweep_store; adopt_store; stop_daemon; ` +
361
+ // Timed and printed always: this is inside the cluster's ready time,
362
+ // so it is the first thing to suspect if that time moves.
363
+ `echo "spectest: cluster image store at $S ready on $SNAP (prepared in $(($(date +%s)-T0))s)"; ` +
364
+ `elif [ "$had_state" = 1 ]; then ` +
365
+ `echo 'spectest: could not sweep the inherited image store; starting from an empty one' >&2; ` +
366
+ `rm -rf "$S"/* "$S"/.[!.]* 2>/dev/null || true; fi; fi; ` +
367
+ `: > "$M" 2>/dev/null || true; `
368
+ );
369
+ }
257
370
 
258
371
  /**
259
372
  * `extraArgs` entries the component overrides anyway. Passing one of
@@ -1883,11 +1996,14 @@ async function collectDiagnostics(
1883
1996
  * so neither cold-cache nor warm starts re-pull. Any `opts.version`
1884
1997
  * works — there's no base-snapshot release to keep in sync with.
1885
1998
  *
1886
- * **Why v1.32.x:** kube-proxy's `nftables` proxy mode is GA in k8s 1.32
1887
- * (beta in 1.31, alpha-gated in 1.30). The component runs kube-proxy in
1888
- * that mode; dropping below 1.31 falls back to the iptables path.
1999
+ * **Why v1.33.x:** k3s 1.33.5 is the first line embedding containerd
2000
+ * 2.1 (2.1.4-k3s1), whose in-tree EROFS snapshotter is what lets the
2001
+ * cluster share the environment's image cache (see the store section
2002
+ * above); 1.32 ships containerd 1.7. kube-proxy's `nftables` proxy mode
2003
+ * is GA since k8s 1.32 (beta in 1.31, alpha-gated in 1.30); dropping
2004
+ * below 1.31 falls back to the iptables path.
1889
2005
  */
1890
- const DEFAULT_K3S_VERSION = "v1.32.1-k3s1";
2006
+ const DEFAULT_K3S_VERSION = "v1.33.5-k3s1";
1891
2007
 
1892
2008
  export function k3s(opts: K3sOptions = {}) {
1893
2009
  const version = opts.version ?? DEFAULT_K3S_VERSION;
@@ -1904,6 +2020,10 @@ export function k3s(opts: K3sOptions = {}) {
1904
2020
  // the in-cluster registry). Seeded via `files` because k3s reads it
1905
2021
  // only at startup, before any setup hook could run.
1906
2022
  const registriesYaml = buildRegistriesYaml(registryEnabled);
2023
+ // The environment's image cache, when this VM carries one: the
2024
+ // cluster's store goes on its root disk and its layers come off the
2025
+ // shared layers disk (see the store section above).
2026
+ const cache = imageCachePathsSync();
1907
2027
  const serverArgs = [
1908
2028
  "k3s",
1909
2029
  "server",
@@ -1986,8 +2106,10 @@ export function k3s(opts: K3sOptions = {}) {
1986
2106
  // is affected.
1987
2107
  "mount --make-rshared / 2>/dev/null || " +
1988
2108
  "echo 'spectest: could not make / rshared; CSI node plugins may fail to publish volumes' >&2; " +
1989
- STORE_SWEEP_SH +
1990
- `exec ${serverArgs}`;
2109
+ storePrepareSh(cache) +
2110
+ // The snapshotter is decided by the wrapper: EROFS on the image cache
2111
+ // when the image's containerd has it, overlayfs otherwise.
2112
+ `exec ${serverArgs} --snapshotter=$SNAP`;
1991
2113
  // Plain /readyz probe. On a warm zot cache the cluster's images are
1992
2114
  // already local, so the first boot completes in seconds; the
1993
2115
  // first-ever boot on a cold-cache host pulls through the mirror and
@@ -2013,37 +2135,30 @@ export function k3s(opts: K3sOptions = {}) {
2013
2135
  // the same netns and are reachable the same way, without appearing
2014
2136
  // in this list (it's documentation, not a firewall).
2015
2137
  ports: registryEnabled ? [80, 443, 6443, K3S_REGISTRY_PORT] : [80, 443, 6443],
2016
- // The cluster's image store, on a volume that outlives this
2017
- // container. Without it the store is the container's writable layer,
2018
- // which `docker rm -f` deletes at teardown — so every build re-pulls
2019
- // and, far more expensively, re-extracts coredns, traefik, the
2020
- // local-path provisioner and everything the project deploys.
2021
- // Measured on the one project on the image cache: 20.6 s per
2022
- // build, identical on the cold and delta tiers, i.e. the one term
2023
- // neither the delta restore nor the cache image store reached.
2024
- //
2025
- // `cache` names a cache disk, which is the ordinary way any project
2026
- // asks for one — there is nothing Kubernetes-shaped in the platform
2027
- // for this. Two clusters in one environment may name the same disk
2028
- // and still get separate containerd roots, because a volume is rooted
2029
- // per service on whatever disk it names; sharing a root would corrupt
2030
- // it, since two containerd daemons cannot share a store (bolt takes
2031
- // an exclusive lock).
2032
- //
2033
- // This reverses a NOTE that stood here for a year: a fresh k3s server
2034
- // over a store killed un-cleanly by `docker rm -f` wedged the
2035
- // apiserver minutes in, and nothing recovered. What was missing was
2036
- // the discipline `base.rs::IMAGE_CACHE_UP_SH` applies to the guest's own
2037
- // store — sweep the crash state, and never inherit a store that did
2038
- // not work. Both are in STORE_SWEEP_SH.
2039
- volumes: [{ target: CLUSTER_STORE_TARGET }],
2040
- ...(registriesYaml
2041
- ? {
2042
- files: [
2043
- { path: "/etc/rancher/k3s/registries.yaml", content: registriesYaml },
2044
- ],
2045
- }
2046
- : {}),
2138
+ // The cluster's image store (see the store section above): its
2139
+ // containerd root on the image cache's root disk, the shared layers
2140
+ // disk read-only at its guest path, golden's mkfs.erofs and the adopt
2141
+ // helper — or, with no cache, a per-environment volume. Two clusters
2142
+ // in one environment get separate roots either way (the service
2143
+ // token names the directory), because two containerd daemons cannot
2144
+ // share a store: bolt takes an exclusive lock.
2145
+ volumes: cache
2146
+ ? [
2147
+ { source: nestedStoreDir(cache, SELF_SERVICE_TOKEN), target: CLUSTER_STORE_TARGET },
2148
+ { source: cache.layers, target: cache.layers, readOnly: true },
2149
+ { source: MKFS_EROFS_PATH, target: MKFS_EROFS_PATH, readOnly: true },
2150
+ { source: STORE_ADOPT_PATH, target: STORE_ADOPT_PATH, readOnly: true },
2151
+ ]
2152
+ : [{ target: CLUSTER_STORE_TARGET }],
2153
+ files: [
2154
+ ...(registriesYaml ? [{ path: "/etc/rancher/k3s/registries.yaml", content: registriesYaml }] : []),
2155
+ ...(cache
2156
+ ? [
2157
+ { path: CONTAINERD_TEMPLATE_PATH, content: CONTAINERD_TEMPLATE },
2158
+ { path: CONTAINERD_EROFS_CONFIG_PATH, content: erofsConfigToml() },
2159
+ ]
2160
+ : []),
2161
+ ],
2047
2162
  readyCheck: {
2048
2163
  type: "exec" as const,
2049
2164
  command: readyCmd,