@specific.dev/spectest 0.26.0 → 0.27.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (74) hide show
  1. package/dist/aws-sigv4.d.ts +42 -0
  2. package/dist/aws-sigv4.js +166 -0
  3. package/dist/browser.d.ts +314 -0
  4. package/dist/browser.js +1320 -0
  5. package/dist/components/email.d.ts +135 -0
  6. package/dist/components/email.js +271 -0
  7. package/dist/components/expo.d.ts +69 -0
  8. package/dist/components/expo.js +125 -0
  9. package/dist/components/index.d.ts +8 -0
  10. package/dist/components/index.js +18 -0
  11. package/dist/components/k3s.d.ts +143 -0
  12. package/dist/components/k3s.js +1067 -0
  13. package/dist/components/postgres.d.ts +93 -0
  14. package/dist/components/postgres.js +58 -0
  15. package/dist/components/replayFake.d.ts +169 -0
  16. package/dist/components/replayFake.js +738 -0
  17. package/dist/components/s3.d.ts +99 -0
  18. package/dist/components/s3.js +81 -0
  19. package/dist/components/supabase.d.ts +197 -0
  20. package/dist/components/supabase.js +1003 -0
  21. package/dist/daemon.d.ts +1 -0
  22. package/dist/daemon.js +4223 -0
  23. package/dist/ids.d.ts +2 -0
  24. package/{src/ids.ts → dist/ids.js} +46 -50
  25. package/dist/index.d.ts +1183 -0
  26. package/dist/index.js +769 -0
  27. package/dist/ingress.d.ts +114 -0
  28. package/dist/ingress.js +210 -0
  29. package/dist/inspect.d.ts +228 -0
  30. package/dist/inspect.js +429 -0
  31. package/dist/locator.d.ts +260 -0
  32. package/dist/locator.js +293 -0
  33. package/dist/mobile.d.ts +71 -0
  34. package/dist/mobile.js +65 -0
  35. package/dist/record-secrets.d.ts +9 -0
  36. package/{src/record-secrets.ts → dist/record-secrets.js} +13 -15
  37. package/dist/recorder.d.ts +516 -0
  38. package/dist/recorder.js +219 -0
  39. package/dist/redis.d.ts +54 -0
  40. package/dist/redis.js +126 -0
  41. package/dist/replay-bundle.d.ts +38 -0
  42. package/{src/replay-bundle.ts → dist/replay-bundle.js} +29 -47
  43. package/dist/resolver.d.ts +1 -0
  44. package/dist/resolver.js +309 -0
  45. package/dist/s3.d.ts +89 -0
  46. package/dist/s3.js +198 -0
  47. package/dist/sql.d.ts +74 -0
  48. package/dist/sql.js +151 -0
  49. package/dist/terminal.d.ts +161 -0
  50. package/dist/terminal.js +538 -0
  51. package/package.json +24 -9
  52. package/src/browser.ts +0 -1819
  53. package/src/components/email.ts +0 -398
  54. package/src/components/expo.ts +0 -167
  55. package/src/components/index.ts +0 -63
  56. package/src/components/k3s.ts +0 -1312
  57. package/src/components/postgres.ts +0 -105
  58. package/src/components/replayFake.ts +0 -848
  59. package/src/components/s3.ts +0 -132
  60. package/src/components/supabase.ts +0 -1299
  61. package/src/daemon.ts +0 -4969
  62. package/src/index.ts +0 -2350
  63. package/src/ingress.ts +0 -288
  64. package/src/inspect.ts +0 -673
  65. package/src/locator.ts +0 -594
  66. package/src/mobile.ts +0 -133
  67. package/src/recorder.ts +0 -817
  68. package/src/redis.ts +0 -202
  69. package/src/resolver.ts +0 -351
  70. package/src/s3.ts +0 -333
  71. package/src/sql.ts +0 -243
  72. package/src/terminal.ts +0 -740
  73. package/src/vendor/rrweb-plugin-console-record.umd.js +0 -521
  74. package/src/vendor/rrweb-record.min.js +0 -5061
@@ -0,0 +1,1067 @@
1
+ import { AsyncLocalStorage } from "node:async_hooks";
2
+ import { spawn as nodeSpawn } from "node:child_process";
3
+ import { randomUUID } from "node:crypto";
4
+ import { existsSync, readFileSync } from "node:fs";
5
+ import { readFile, unlink } from "node:fs/promises";
6
+ import { AppsV1Api, CoreV1Api, KubeConfig, KubernetesObjectApi, ResponseContext, ServerConfiguration, createConfiguration, loadAllYaml, } from "@kubernetes/client-node";
7
+ import { Observable } from "@kubernetes/client-node/dist/gen/rxjsStub.js";
8
+ import { dnsName, provides, SELF_SERVICE_TOKEN } from "../index.js";
9
+ import { readRaw, readTag, wrap } from "../inspect.js";
10
+ import { recorderAnnotate, recorderRemove } from "../recorder.js";
11
+ /** Port the in-cluster registry listens on (plain HTTP). */
12
+ const K3S_REGISTRY_PORT = 5000;
13
+ function runProcess(cmd, args, timeoutMs = 30_000) {
14
+ return new Promise((resolve, reject) => {
15
+ const cp = nodeSpawn(cmd, args, { stdio: ["ignore", "pipe", "pipe"] });
16
+ const out = [];
17
+ const err = [];
18
+ cp.stdout.on("data", (c) => out.push(c));
19
+ cp.stderr.on("data", (c) => err.push(c));
20
+ const t = setTimeout(() => cp.kill("SIGKILL"), timeoutMs);
21
+ cp.on("error", (e) => {
22
+ clearTimeout(t);
23
+ reject(e);
24
+ });
25
+ cp.on("close", (code) => {
26
+ clearTimeout(t);
27
+ resolve({
28
+ stdout: Buffer.concat(out).toString("utf8"),
29
+ stderr: Buffer.concat(err).toString("utf8"),
30
+ code: code ?? -1,
31
+ });
32
+ });
33
+ });
34
+ }
35
+ // In-VM root CA, generated once into the base snapshot (see
36
+ // control-plane `base.rs`). Trusted everywhere the test framework runs —
37
+ // Node (`NODE_EXTRA_CA_CERTS`), Chromium (NSS DB), Python, the system
38
+ // store — so a leaf signed by it gives `ctx.fetch`/`ctx.browser()` a
39
+ // clean HTTPS handshake. The k3s `setup` hook runs inside the daemon's
40
+ // Bun process (root in the VM), so it can read the CA key and mint
41
+ // directly. These constants are intentionally redeclared here rather than
42
+ // imported from the daemon: the SDK ships to end users and must not
43
+ // depend on daemon internals.
44
+ const CA_PATH = process.env.SPECTEST_CA_PATH ?? "/etc/spectest/ca.crt";
45
+ const CA_KEY_PATH = process.env.SPECTEST_CA_KEY_PATH ?? "/etc/spectest/ca.key";
46
+ function caPresent() {
47
+ return existsSync(CA_PATH) && existsSync(CA_KEY_PATH);
48
+ }
49
+ /**
50
+ * Mint a leaf certificate from the in-VM root CA covering `hostnames`
51
+ * (used here as the SANs of the cluster's wildcard ingress domains).
52
+ * Returns the cert + key as PEM strings. Self-contained openssl shell-out
53
+ * — deliberately not shared with the daemon's own cert minting to keep
54
+ * the distributed SDK decoupled from daemon code.
55
+ */
56
+ async function issueIngressCert(hostnames) {
57
+ const id = `spectest-k3s-ingress-${randomUUID().slice(0, 8)}`;
58
+ const keyPath = `/tmp/${id}.key`;
59
+ const crtPath = `/tmp/${id}.crt`;
60
+ const sans = hostnames.map((h) => `DNS:${h}`).join(",");
61
+ const r = await runProcess("openssl", [
62
+ "req",
63
+ "-newkey",
64
+ "rsa:2048",
65
+ "-nodes",
66
+ "-keyout",
67
+ keyPath,
68
+ "-out",
69
+ crtPath,
70
+ "-x509",
71
+ "-CA",
72
+ CA_PATH,
73
+ "-CAkey",
74
+ CA_KEY_PATH,
75
+ "-days",
76
+ "3650",
77
+ "-subj",
78
+ "/CN=spectest-k3s-ingress",
79
+ "-addext",
80
+ `subjectAltName=${sans}`,
81
+ "-addext",
82
+ "basicConstraints=CA:FALSE",
83
+ "-addext",
84
+ "extendedKeyUsage=serverAuth",
85
+ "-addext",
86
+ "keyUsage=digitalSignature,keyEncipherment",
87
+ ], 30_000);
88
+ if (r.code !== 0) {
89
+ throw new Error(`k3s ingress cert minting failed (openssl rc=${r.code}): ${r.stderr.trim() || r.stdout.trim()}`);
90
+ }
91
+ try {
92
+ const [cert, key] = await Promise.all([
93
+ readFile(crtPath, "utf8"),
94
+ readFile(keyPath, "utf8"),
95
+ ]);
96
+ return { cert, key };
97
+ }
98
+ finally {
99
+ await Promise.all([
100
+ unlink(keyPath).catch(() => { }),
101
+ unlink(crtPath).catch(() => { }),
102
+ ]);
103
+ }
104
+ }
105
+ const callContext = new AsyncLocalStorage();
106
+ // HTTP transport for `@kubernetes/client-node` that routes through
107
+ // `globalThis.fetch`. Two reasons:
108
+ // 1. The daemon's per-test `installFetchWrapper` already records every
109
+ // `globalThis.fetch` call as an HTTP event — using fetch here gets
110
+ // k8s API calls recorded for free, no library-specific wiring.
111
+ // 2. Bun's fetch needs Bun-shaped TLS options (`tls: { ... }`) for
112
+ // mTLS; node-fetch's `agent` parameter — which the library's
113
+ // default transport relies on — is silently ignored under Bun.
114
+ // We honor the lib's auth flow (`KubeConfig.applySecurityAuthentication`
115
+ // sets an Agent on the request) by extracting cert/key off that
116
+ // agent and passing them via the Bun-shaped option.
117
+ class FetchHttpLibrary {
118
+ send(request) {
119
+ const promise = doFetch(request);
120
+ return new Observable(promise);
121
+ }
122
+ }
123
+ /**
124
+ * Map a Kubernetes API request to its `(verb, group/version, resource,
125
+ * namespace, name, subresource)` from the URL + HTTP method alone.
126
+ *
127
+ * Path grammar (the two API roots):
128
+ * - core group: `/api/<version>/...`
129
+ * - named group: `/apis/<group>/<version>/...`
130
+ * after which the remainder is either a cluster-scoped resource
131
+ * (`nodes`, `namespaces`, …) or `namespaces/<ns>/<resource>...`. The
132
+ * trailing `<resource>[/<name>[/<subresource>]]` shape plus the method
133
+ * (and `?watch=`) yields the verb.
134
+ *
135
+ * Returns `null` for non-resource paths — discovery (`/api`, `/apis`,
136
+ * `/apis/<group>/<version>`), `/version`, `/healthz`, `/openapi/...` —
137
+ * so those stay rendered as plain `http`.
138
+ */
139
+ function describeKubeRequest(method, rawUrl) {
140
+ let path;
141
+ let query;
142
+ try {
143
+ const u = new URL(rawUrl);
144
+ path = u.pathname;
145
+ query = u.searchParams;
146
+ }
147
+ catch {
148
+ const q = rawUrl.indexOf("?");
149
+ path = q === -1 ? rawUrl : rawUrl.slice(0, q);
150
+ query = new URLSearchParams(q === -1 ? "" : rawUrl.slice(q + 1));
151
+ }
152
+ const segs = path.split("/").filter((s) => s.length > 0);
153
+ if (segs.length === 0)
154
+ return null;
155
+ let group;
156
+ let apiVersion;
157
+ let rest;
158
+ if (segs[0] === "api") {
159
+ group = "";
160
+ apiVersion = segs[1];
161
+ rest = segs.slice(2);
162
+ }
163
+ else if (segs[0] === "apis") {
164
+ group = segs[1];
165
+ apiVersion = segs[2];
166
+ rest = segs.slice(3);
167
+ }
168
+ else {
169
+ return null; // /version, /healthz, /openapi, …
170
+ }
171
+ if (!apiVersion)
172
+ return null; // discovery root (/api, /apis/<group>)
173
+ // `namespaces/<ns>/<resource>...` is namespaced; everything else
174
+ // (including `namespaces` and `namespaces/<name>` themselves, and
175
+ // cluster-scoped resources like `nodes`) is taken as-is.
176
+ let namespace;
177
+ let resourcePath = rest;
178
+ if (rest[0] === "namespaces" && rest.length >= 3) {
179
+ namespace = rest[1];
180
+ resourcePath = rest.slice(2);
181
+ }
182
+ if (resourcePath.length === 0)
183
+ return null; // APIResourceList discovery
184
+ const resource = resourcePath[0];
185
+ const name = resourcePath.length >= 2 ? resourcePath[1] : undefined;
186
+ const subresource = resourcePath.length >= 3 ? resourcePath[2] : undefined;
187
+ const watchParam = query.get("watch");
188
+ const watch = watchParam === "true" || watchParam === "1";
189
+ const hasName = name !== undefined;
190
+ let verb;
191
+ switch (method.toUpperCase()) {
192
+ case "GET":
193
+ case "HEAD":
194
+ verb = hasName ? "get" : watch ? "watch" : "list";
195
+ break;
196
+ case "POST":
197
+ verb = "create";
198
+ break;
199
+ case "PUT":
200
+ verb = "update";
201
+ break;
202
+ case "PATCH":
203
+ verb = "patch";
204
+ break;
205
+ case "DELETE":
206
+ verb = hasName ? "delete" : "deletecollection";
207
+ break;
208
+ default:
209
+ verb = method.toLowerCase();
210
+ }
211
+ return { verb, group, apiVersion, resource, subresource, name, namespace };
212
+ }
213
+ /**
214
+ * True for Kubernetes API *discovery* paths — the version/group/resource
215
+ * enumeration endpoints (`/api`, `/api/<version>`, `/apis`, `/apis/<group>`,
216
+ * `/apis/<group>/<version>`) the dynamic client hits to resolve a kind to
217
+ * its resource path. They carry no resource operation (so
218
+ * `describeKubeRequest` returns null), and `doFetch` retracts their events
219
+ * from the timeline. Non-resource paths that are NOT discovery (`/healthz`,
220
+ * `/version`, `/openapi`, …) are deliberately not matched — they stay as
221
+ * `http`.
222
+ */
223
+ function isKubeDiscoveryPath(rawUrl) {
224
+ let path;
225
+ try {
226
+ path = new URL(rawUrl).pathname;
227
+ }
228
+ catch {
229
+ const q = rawUrl.indexOf("?");
230
+ path = q === -1 ? rawUrl : rawUrl.slice(0, q);
231
+ }
232
+ const segs = path.split("/").filter((s) => s.length > 0);
233
+ if (segs.length === 0)
234
+ return false;
235
+ // `/api` + `/api/<version>`; `/apis` + `/apis/<group>` + `/apis/<group>/<version>`.
236
+ // Anything longer carries a resource segment and is handled as `kube`.
237
+ if (segs[0] === "api")
238
+ return segs.length <= 2;
239
+ if (segs[0] === "apis")
240
+ return segs.length <= 3;
241
+ return false;
242
+ }
243
+ async function doFetch(request) {
244
+ const url = request.getUrl();
245
+ const method = String(request.getHttpMethod());
246
+ const body = request.getBody();
247
+ const reqHeaders = {};
248
+ for (const [k, v] of Object.entries(request.getHeaders())) {
249
+ reqHeaders[k] = String(v);
250
+ }
251
+ // The library's auth flow puts client cert/key on an https.Agent
252
+ // attached to the request. Pull them out so we can hand them to Bun's
253
+ // fetch via its `tls` option.
254
+ // eslint-disable-next-line @typescript-eslint/no-explicit-any
255
+ const agent = request.getAgent();
256
+ const agentOpts = agent?.options ?? {};
257
+ const tlsOpts = { rejectUnauthorized: false };
258
+ if (agentOpts.cert)
259
+ tlsOpts.cert = agentOpts.cert;
260
+ if (agentOpts.key)
261
+ tlsOpts.key = agentOpts.key;
262
+ const wrapped = await fetch(url, {
263
+ method,
264
+ headers: reqHeaders,
265
+ body: body,
266
+ signal: request.getSignal(),
267
+ // Bun-specific TLS shape; under Node this option is ignored.
268
+ // eslint-disable-next-line @typescript-eslint/no-explicit-any
269
+ tls: tlsOpts,
270
+ });
271
+ // Before unwrapping, capture the inspector tag the daemon's fetch
272
+ // wrapper installed on the Response. We feed the seq back through
273
+ // AsyncLocalStorage so the API method's eventual return value can
274
+ // re-acquire it — otherwise the chain `core.listNode() →
275
+ // expect(result)` would record assertions with no back-reference.
276
+ const tag = readTag(wrapped);
277
+ const slot = callContext.getStore();
278
+ if (slot && tag)
279
+ slot.seq = tag.sourceSeq;
280
+ // Reclassify the `http` event the fetch wrapper just recorded into a
281
+ // Kubernetes-specific `kube` event (verb/resource/namespace/name), so
282
+ // the timeline reads `list pods · default` rather than the raw API URL.
283
+ // A `tag` is only present when recording was active for this call, so
284
+ // this is a no-op outside instrumented test runs.
285
+ if (tag && tag.sourceSeq !== undefined) {
286
+ const meta = describeKubeRequest(method, url);
287
+ if (meta) {
288
+ recorderAnnotate(tag.sourceSeq, { kind: "kube", ...meta });
289
+ }
290
+ else if (isKubeDiscoveryPath(url)) {
291
+ // The dynamic client (`objects`, KubernetesObjectApi) can't know a
292
+ // kind's resource path ahead of time, so before the real request it
293
+ // GETs the group's resource list (`/apis/<group>/<version>` →
294
+ // APIResourceList) to map e.g. Ingress → `ingresses`/namespaced, then
295
+ // caches it (apiVersionResourceCache). That discovery GET is library
296
+ // plumbing the test author never wrote, and because the cache is
297
+ // per-daemon-process it surfaces non-deterministically across forks
298
+ // (first access pays it; `dependsOn` children inheriting the warm
299
+ // cache don't). Retract it rather than leave a bare `http` row — the
300
+ // real list/read that follows is recorded and reclassified as usual.
301
+ recorderRemove(tag.sourceSeq);
302
+ }
303
+ // Other non-resource paths (/healthz, /version, /openapi, …) fall
304
+ // through and stay rendered as plain `http`.
305
+ }
306
+ // The daemon's fetch wrapper proxies `Response.status` and similar
307
+ // primitives as carrier objects (so test assertions can fold under
308
+ // the originating HTTP event). The kubernetes/client-node lib calls
309
+ // `httpStatusCode.toString()` which would then return
310
+ // "[object Object]" and the status-code dispatch falls through to
311
+ // "Unknown API Status Code!". Pull out the raw Response.
312
+ const response = wrapped.unwrap?.() ?? wrapped;
313
+ const resHeaders = {};
314
+ response.headers.forEach((v, k) => {
315
+ resHeaders[k] = v;
316
+ });
317
+ const buf = Buffer.from(await response.arrayBuffer());
318
+ return new ResponseContext(response.status, resHeaders, {
319
+ text: async () => buf.toString("utf8"),
320
+ binary: async () => buf,
321
+ });
322
+ }
323
+ /**
324
+ * Recursively strip the inspector's carrier/proxy wrappers from a
325
+ * value. Needed for arguments flowing into the kubernetes/client-node
326
+ * API methods — if a wrapped pod's `metadata.name` (a primitive-carrier
327
+ * object) reaches a URL template, the lib stringifies it to
328
+ * `"[object Object]"` and the request 404s.
329
+ */
330
+ function deepUnwrap(value) {
331
+ if (value === null || value === undefined)
332
+ return value;
333
+ const raw = readRaw(value);
334
+ if (raw !== value)
335
+ return deepUnwrap(raw);
336
+ if (typeof value !== "object")
337
+ return value;
338
+ if (Array.isArray(value))
339
+ return value.map(deepUnwrap);
340
+ const out = {};
341
+ for (const [k, v] of Object.entries(value)) {
342
+ out[k] = deepUnwrap(v);
343
+ }
344
+ return out;
345
+ }
346
+ /**
347
+ * Wrap a `@kubernetes/client-node` Api instance so each method call
348
+ * runs in its own AsyncLocalStorage slot — `doFetch` writes the HTTP
349
+ * event's `sourceSeq` into the slot, and after the lib parses the
350
+ * response we re-attach the seq to the returned object. Downstream
351
+ * `expect(result.items[0].status…)` assertions then fold under that
352
+ * HTTP event in the test event log, the same way `expect(res.status)`
353
+ * does for plain `fetch` calls.
354
+ *
355
+ * Method arguments are deep-unwrapped on the way in so values pulled
356
+ * from a previous API response (still carrying the inspector wrappers)
357
+ * can be passed straight back into another call.
358
+ *
359
+ * Non-function properties pass through untagged.
360
+ */
361
+ function withTagging(api) {
362
+ return new Proxy(api, {
363
+ get(target, prop, receiver) {
364
+ const value = Reflect.get(target, prop, receiver);
365
+ if (typeof value !== "function")
366
+ return value;
367
+ // Bind the original method to `target` so the lib's internal
368
+ // `this.configuration` accesses keep working.
369
+ // eslint-disable-next-line @typescript-eslint/no-explicit-any
370
+ const method = value.bind(target);
371
+ return (...args) => {
372
+ const unwrappedArgs = args.map(deepUnwrap);
373
+ const slot = {};
374
+ const result = callContext.run(slot, () => method(...unwrappedArgs));
375
+ // Wrap unconditionally — `slot.seq` is undefined when no event was
376
+ // recorded (setup/eval, no active recorder), but the result's type is
377
+ // wrapped, so the value must be wrapped at runtime too (just without a
378
+ // provenance link). Keeps `.unwrap()` available in every context.
379
+ if (result && typeof result.then === "function") {
380
+ return result.then((v) => wrap(v, slot.seq));
381
+ }
382
+ return wrap(result, slot.seq);
383
+ };
384
+ },
385
+ });
386
+ }
387
+ /**
388
+ * Image tag for the Traefik we install. Pulled on first cluster boot
389
+ * through the host `zot` mirror (`docker.io` → cache) configured in
390
+ * `registries.yaml`, then captured into the warm-template snapshot so
391
+ * warm starts never pull.
392
+ */
393
+ const TRAEFIK_IMAGE = "rancher/mirrored-library-traefik:3.3.2";
394
+ /**
395
+ * Names of the in-cluster resources that carry the CA-signed default
396
+ * ingress certificate (created in `setupK3sCluster` when TLS is enabled).
397
+ * The Secret holds the leaf cert+key; the ConfigMap holds the Traefik
398
+ * file-provider snippet that points the `default` TLS store at it.
399
+ */
400
+ const TRAEFIK_TLS_SECRET = "traefik-default-tls";
401
+ const TRAEFIK_DYNAMIC_CONFIGMAP = "traefik-dynamic";
402
+ /**
403
+ * Traefik file-provider dynamic config: make the in-VM-CA leaf the
404
+ * `default` store certificate, so every router on the `websecure`
405
+ * entrypoint (which we force to TLS) serves it with no per-Ingress
406
+ * `spec.tls` needed.
407
+ */
408
+ const TRAEFIK_DYNAMIC_TLS = `tls:
409
+ stores:
410
+ default:
411
+ defaultCertificate:
412
+ certFile: /certs/tls.crt
413
+ keyFile: /certs/tls.key
414
+ `;
415
+ /**
416
+ * Traefik manifest applied during setup(). `hostNetwork: true` puts
417
+ * Traefik in the k3s container's netns, so it binds the container's :80
418
+ * (and, with `tls`, :443) directly — no CNI portmap involved (that path
419
+ * still trips on the kernel's missing xt_comment match).
420
+ *
421
+ * When `tls` is set we add a `websecure` :443 entrypoint with TLS forced
422
+ * on (served from the `default` store, i.e. the in-VM-CA leaf mounted
423
+ * from the `traefik-default-tls` Secret via the file provider). HTTPS
424
+ * then works for any routed host under the cluster's `ingressDomains`
425
+ * with zero per-Ingress config; the :80 `web` entrypoint is unchanged.
426
+ */
427
+ function buildTraefikManifest(tls) {
428
+ const args = [
429
+ " - --entrypoints.web.address=:80",
430
+ ...(tls
431
+ ? [
432
+ " - --entrypoints.websecure.address=:443",
433
+ " - --entrypoints.websecure.http.tls=true",
434
+ ]
435
+ : []),
436
+ " - --providers.kubernetesingress=true",
437
+ " - --providers.kubernetesingress.ingressclass=traefik",
438
+ ...(tls
439
+ ? [
440
+ " - --providers.file.directory=/dynamic",
441
+ " - --providers.file.watch=true",
442
+ ]
443
+ : []),
444
+ " - --log.level=INFO",
445
+ ].join("\n");
446
+ const ports = [
447
+ " - name: web",
448
+ " containerPort: 80",
449
+ ...(tls
450
+ ? [" - name: websecure", " containerPort: 443"]
451
+ : []),
452
+ ].join("\n");
453
+ const volumeMounts = tls
454
+ ? `
455
+ volumeMounts:
456
+ - name: default-cert
457
+ mountPath: /certs
458
+ readOnly: true
459
+ - name: dynamic
460
+ mountPath: /dynamic
461
+ readOnly: true`
462
+ : "";
463
+ const volumes = tls
464
+ ? `
465
+ volumes:
466
+ - name: default-cert
467
+ secret:
468
+ secretName: ${TRAEFIK_TLS_SECRET}
469
+ - name: dynamic
470
+ configMap:
471
+ name: ${TRAEFIK_DYNAMIC_CONFIGMAP}`
472
+ : "";
473
+ return `apiVersion: v1
474
+ kind: ServiceAccount
475
+ metadata:
476
+ name: traefik
477
+ namespace: kube-system
478
+ ---
479
+ apiVersion: rbac.authorization.k8s.io/v1
480
+ kind: ClusterRole
481
+ metadata:
482
+ name: traefik
483
+ rules:
484
+ - apiGroups: [""]
485
+ resources: ["services", "endpoints", "secrets", "nodes"]
486
+ verbs: ["get", "list", "watch"]
487
+ - apiGroups: ["discovery.k8s.io"]
488
+ resources: ["endpointslices"]
489
+ verbs: ["get", "list", "watch"]
490
+ - apiGroups: ["networking.k8s.io"]
491
+ resources: ["ingresses", "ingressclasses"]
492
+ verbs: ["get", "list", "watch"]
493
+ - apiGroups: ["networking.k8s.io"]
494
+ resources: ["ingresses/status"]
495
+ verbs: ["update"]
496
+ ---
497
+ apiVersion: rbac.authorization.k8s.io/v1
498
+ kind: ClusterRoleBinding
499
+ metadata:
500
+ name: traefik
501
+ roleRef:
502
+ apiGroup: rbac.authorization.k8s.io
503
+ kind: ClusterRole
504
+ name: traefik
505
+ subjects:
506
+ - kind: ServiceAccount
507
+ name: traefik
508
+ namespace: kube-system
509
+ ---
510
+ apiVersion: networking.k8s.io/v1
511
+ kind: IngressClass
512
+ metadata:
513
+ name: traefik
514
+ annotations:
515
+ ingressclass.kubernetes.io/is-default-class: "true"
516
+ spec:
517
+ controller: traefik.io/ingress-controller
518
+ ---
519
+ apiVersion: apps/v1
520
+ kind: Deployment
521
+ metadata:
522
+ name: traefik
523
+ namespace: kube-system
524
+ labels:
525
+ app: traefik
526
+ spec:
527
+ replicas: 1
528
+ selector:
529
+ matchLabels:
530
+ app: traefik
531
+ template:
532
+ metadata:
533
+ labels:
534
+ app: traefik
535
+ spec:
536
+ serviceAccountName: traefik
537
+ hostNetwork: true
538
+ dnsPolicy: Default
539
+ tolerations:
540
+ - operator: Exists
541
+ containers:
542
+ - name: traefik
543
+ image: ${TRAEFIK_IMAGE}
544
+ imagePullPolicy: IfNotPresent
545
+ args:
546
+ ${args}
547
+ ports:
548
+ ${ports}${volumeMounts}${volumes}
549
+ `;
550
+ }
551
+ /**
552
+ * Host-side `zot` pull-through cache layout (local Firecracker provider
553
+ * only). One zot instance per upstream registry, all bound to the
554
+ * `spectest-br0` gateway `10.42.0.1` on the ports below — **kept in sync
555
+ * with `scripts/install-zot.sh`**. We mirror the cluster's containerd
556
+ * through these so every image pull reuses the shared host cache instead
557
+ * of hitting the public registry, and we list the canonical upstream as
558
+ * a fallback endpoint so a missing/cold mirror only ever slows a pull,
559
+ * never breaks it.
560
+ */
561
+ const ZOT_MIRRORS = [
562
+ { registry: "docker.io", port: 5000, upstream: "https://registry-1.docker.io" },
563
+ { registry: "ghcr.io", port: 5001, upstream: "https://ghcr.io" },
564
+ { registry: "quay.io", port: 5002, upstream: "https://quay.io" },
565
+ { registry: "registry.k8s.io", port: 5003, upstream: "https://registry.k8s.io" },
566
+ { registry: "public.ecr.aws", port: 5004, upstream: "https://public.ecr.aws" },
567
+ { registry: "gcr.io", port: 5005, upstream: "https://gcr.io" },
568
+ { registry: "mcr.microsoft.com", port: 5006, upstream: "https://mcr.microsoft.com" },
569
+ ];
570
+ /**
571
+ * Discover the host-side image cache gateway by reading the same
572
+ * `registry-mirrors` entry the in-VM dockerd already uses (baked into
573
+ * the local provider's golden `/etc/docker/daemon.json`). Returns the
574
+ * gateway host (`"10.42.0.1"`) when present, or `null` when there's no
575
+ * host cache — e.g. on Freestyle, where the cluster then pulls every
576
+ * image direct. Runs inside the daemon (VM) at `index.ts` load time, so
577
+ * the result is stable per host and never poisons the warm-template
578
+ * cache.
579
+ */
580
+ function detectHostMirrorGateway() {
581
+ try {
582
+ const cfg = JSON.parse(readFileSync("/etc/docker/daemon.json", "utf8"));
583
+ const first = cfg["registry-mirrors"]?.[0];
584
+ return first ? new URL(first).hostname || null : null;
585
+ }
586
+ catch {
587
+ return null;
588
+ }
589
+ }
590
+ /**
591
+ * Build `/etc/rancher/k3s/registries.yaml`. k3s reads this **once, at
592
+ * startup**, to configure its embedded containerd — which is why it has
593
+ * to be seeded via `files` (a pre-start bind mount) rather than a
594
+ * `setup` hook. Two jobs:
595
+ * 1. Mirror the cluster's image pulls through the host `zot` cache
596
+ * (local provider only; omitted when there's no host cache).
597
+ * 2. Trust the in-cluster registry, addressed as `<key>.internal:5000`
598
+ * (the `{{SPECTEST_SERVICE}}` token is expanded to the cluster's
599
+ * service key when the file is written). Image *references* use
600
+ * that peer-reachable name, but containerd pulls via the loopback
601
+ * endpoint `http://127.0.0.1:5000` — the hostNetwork registry pod
602
+ * shares the node's netns, so this needs no in-container DNS and
603
+ * can't be broken by a clobbered `hostnames`.
604
+ * Returns `null` when there's nothing to configure (no host cache and
605
+ * `registry` disabled), in which case no file is injected.
606
+ */
607
+ function buildRegistriesYaml(registryEnabled) {
608
+ const gateway = detectHostMirrorGateway();
609
+ if (!gateway && !registryEnabled)
610
+ return null;
611
+ const lines = ["mirrors:"];
612
+ if (gateway) {
613
+ for (const { registry, port, upstream } of ZOT_MIRRORS) {
614
+ lines.push(` "${registry}":`, ` endpoint:`, ` - "http://${gateway}:${port}"`, ` - "${upstream}"`);
615
+ }
616
+ }
617
+ if (registryEnabled) {
618
+ const host = `{{SPECTEST_SERVICE}}.internal:${K3S_REGISTRY_PORT}`;
619
+ lines.push(` "${host}":`, ` endpoint:`, ` - "http://127.0.0.1:${K3S_REGISTRY_PORT}"`, "configs:",
620
+ // The endpoint is plain HTTP; the config (keyed by endpoint host)
621
+ // makes that explicit and disables any TLS attempt against it.
622
+ ` "127.0.0.1:${K3S_REGISTRY_PORT}":`, ` tls:`, ` insecure_skip_verify: true`);
623
+ }
624
+ return lines.join("\n") + "\n";
625
+ }
626
+ /**
627
+ * In-cluster OCI registry (CNCF `distribution`). `hostNetwork: true`
628
+ * binds the cluster container's `:5000` directly — the same trick
629
+ * Traefik uses — so peer services reach it at `<cluster-key>.internal:5000`
630
+ * (the cluster service's own alias) and the node's own containerd reaches
631
+ * it at `127.0.0.1:5000`. Storage is an `emptyDir`, so pushed images live
632
+ * in the cluster and are captured by snapshot / isolated per test fork
633
+ * like all other in-VM state.
634
+ */
635
+ const REGISTRY_MANIFEST = `apiVersion: apps/v1
636
+ kind: Deployment
637
+ metadata:
638
+ name: spectest-registry
639
+ namespace: kube-system
640
+ labels:
641
+ app: spectest-registry
642
+ spec:
643
+ replicas: 1
644
+ selector:
645
+ matchLabels:
646
+ app: spectest-registry
647
+ template:
648
+ metadata:
649
+ labels:
650
+ app: spectest-registry
651
+ spec:
652
+ hostNetwork: true
653
+ dnsPolicy: Default
654
+ tolerations:
655
+ - operator: Exists
656
+ containers:
657
+ - name: registry
658
+ image: registry:2
659
+ imagePullPolicy: IfNotPresent
660
+ env:
661
+ - name: REGISTRY_HTTP_ADDR
662
+ value: ":${K3S_REGISTRY_PORT}"
663
+ - name: REGISTRY_STORAGE_DELETE_ENABLED
664
+ value: "true"
665
+ ports:
666
+ - name: registry
667
+ containerPort: ${K3S_REGISTRY_PORT}
668
+ volumeMounts:
669
+ - name: data
670
+ mountPath: /var/lib/registry
671
+ volumes:
672
+ - name: data
673
+ emptyDir: {}
674
+ `;
675
+ /**
676
+ * Wait for a Deployment to reach its desired ready-replica count,
677
+ * polling once a second up to `timeoutMs`. Throws with the last-seen
678
+ * status (plus kube-system pod diagnostics) on timeout.
679
+ */
680
+ async function waitForDeployment(clusterName, helpers, deployment, timeoutMs) {
681
+ const deadline = Date.now() + timeoutMs;
682
+ let lastErr;
683
+ while (Date.now() < deadline) {
684
+ try {
685
+ // `.unwrap()` recovers the plain object — the client wraps its result in
686
+ // every context now (provenance-free here, since this internal poll runs
687
+ // during setup with no active recorder). We're reading for control flow,
688
+ // not asserting, so go straight to raw.
689
+ const dep = (await helpers.client.apps.readNamespacedDeployment({
690
+ name: deployment,
691
+ namespace: "kube-system",
692
+ })).unwrap();
693
+ const ready = dep.status?.readyReplicas ?? 0;
694
+ const want = dep.spec?.replicas ?? 1;
695
+ if (ready >= want && want > 0)
696
+ return;
697
+ lastErr = `${deployment} Deployment exists but only ${ready}/${want} replicas Ready`;
698
+ }
699
+ catch (err) {
700
+ const msg = err?.message ?? String(err);
701
+ lastErr = /not found|404/i.test(msg)
702
+ ? `${deployment} Deployment does not exist yet`
703
+ : msg;
704
+ }
705
+ // 250ms: the two sequential rollout waits in setup sit on the cold
706
+ // start's critical path, and a 1s poll wasted up to ~2s of it.
707
+ await new Promise((r) => setTimeout(r, 250));
708
+ }
709
+ const diag = await collectTraefikDiagnostics(helpers);
710
+ throw new Error(`k3s(${clusterName}): ${deployment} did not reach Ready within ${timeoutMs / 1000}s. ${lastErr ?? ""}\n${diag}`);
711
+ }
712
+ /**
713
+ * Post-Ready setup. Apply the Traefik manifest (hostNetwork) and, when
714
+ * enabled, the in-cluster registry; wait for each Deployment to come
715
+ * Ready. Captured by the warm-template snapshot, so warm starts pay none
716
+ * of this cost.
717
+ *
718
+ * When the cluster declares `ingressDomains` (and the in-VM CA is
719
+ * present), TLS is enabled: we mint a CA-signed leaf covering `*.<domain>`
720
+ * for each domain, stash it in the `traefik-default-tls` Secret + a
721
+ * file-provider ConfigMap, and bring Traefik up with a `websecure` :443
722
+ * entrypoint serving it as the default cert. Those domains are then
723
+ * reachable over HTTPS with a cert the test framework already trusts.
724
+ */
725
+ async function setupK3sCluster(name, helpers, opts) {
726
+ const tlsEnabled = opts.ingressDomains.length > 0 && caPresent();
727
+ if (tlsEnabled) {
728
+ const { cert, key } = await issueIngressCert(opts.ingressDomains.map((d) => `*.${d}`));
729
+ // Apply the cert Secret + dynamic-config ConfigMap before the
730
+ // Deployment that mounts them. `stringData` lets us hand over plain
731
+ // PEM; the API server base64-encodes it.
732
+ await helpers.client.core.createNamespacedSecret({
733
+ namespace: "kube-system",
734
+ body: {
735
+ metadata: { name: TRAEFIK_TLS_SECRET, namespace: "kube-system" },
736
+ type: "kubernetes.io/tls",
737
+ stringData: { "tls.crt": cert, "tls.key": key },
738
+ },
739
+ });
740
+ await helpers.client.core.createNamespacedConfigMap({
741
+ namespace: "kube-system",
742
+ body: {
743
+ metadata: {
744
+ name: TRAEFIK_DYNAMIC_CONFIGMAP,
745
+ namespace: "kube-system",
746
+ },
747
+ data: { "tls.yaml": TRAEFIK_DYNAMIC_TLS },
748
+ },
749
+ });
750
+ }
751
+ await helpers.apply(buildTraefikManifest(tlsEnabled));
752
+ if (opts.registry)
753
+ await helpers.apply(REGISTRY_MANIFEST);
754
+ // Both rollouts proceed independently inside the cluster — wait on them
755
+ // concurrently (they used to serialize, wasting up to a rollout's tail).
756
+ const waits = [waitForDeployment(name, helpers, "traefik", 120_000)];
757
+ if (opts.registry) {
758
+ waits.push(waitForDeployment(name, helpers, "spectest-registry", 120_000));
759
+ }
760
+ await Promise.all(waits);
761
+ }
762
+ /**
763
+ * Snapshot of kube-system state, dumped on traefik-wait timeout. With
764
+ * the static install, the failure surface is just "did our Deployment
765
+ * schedule and become Ready?" — pod listing covers that.
766
+ */
767
+ async function collectTraefikDiagnostics(helpers) {
768
+ const lines = [];
769
+ try {
770
+ const pods = await helpers.client.core.listNamespacedPod({
771
+ namespace: "kube-system",
772
+ });
773
+ lines.push(`kube-system pods (${pods.items.length}):`);
774
+ for (const p of pods.items) {
775
+ const phase = p.status?.phase ?? "?";
776
+ const cs = p.status?.containerStatuses ?? [];
777
+ const reasons = cs
778
+ .map((c) => c.state?.waiting?.reason ?? c.state?.terminated?.reason ?? "")
779
+ .filter((s) => s)
780
+ .join(",");
781
+ lines.push(` ${p.metadata?.name ?? "?"}: phase=${phase}${reasons ? ` reasons=${reasons}` : ""}`);
782
+ // The waiting `message` carries containerd's actual error — e.g. the
783
+ // failing endpoint, an upstream `429 Too Many Requests`, or a
784
+ // `connection refused`. The `reason` alone (`ErrImagePull`) hides all
785
+ // of that, which is exactly what we need when a pull won't settle.
786
+ for (const c of cs) {
787
+ const msg = c.state?.waiting?.message ?? c.state?.terminated?.message ?? "";
788
+ if (msg)
789
+ lines.push(` ${c.name}: ${msg.replace(/\s+/g, " ").trim()}`);
790
+ }
791
+ }
792
+ }
793
+ catch (err) {
794
+ lines.push(`(listing pods failed: ${err?.message ?? String(err)})`);
795
+ }
796
+ // Recent Warning events surface pull failures the kubelet emits before a
797
+ // container status even settles (FailedPull / Failed / BackOff), with the
798
+ // raw containerd message attached. Best-effort: never let diagnostics throw.
799
+ try {
800
+ const events = await helpers.client.core.listNamespacedEvent({
801
+ namespace: "kube-system",
802
+ });
803
+ const warnings = (events.items ?? [])
804
+ .filter((e) => e.type === "Warning")
805
+ .map((e) => ({
806
+ obj: e.involvedObject?.name ?? "?",
807
+ reason: e.reason ?? "?",
808
+ message: (e.message ?? "").replace(/\s+/g, " ").trim(),
809
+ }))
810
+ .filter((e) => e.message);
811
+ if (warnings.length) {
812
+ lines.push(`kube-system Warning events (${warnings.length}):`);
813
+ // Keep the tail — newest events are appended last by the API.
814
+ for (const w of warnings.slice(-12)) {
815
+ lines.push(` ${w.obj} [${w.reason}] ${w.message}`);
816
+ }
817
+ }
818
+ }
819
+ catch (err) {
820
+ lines.push(`(listing events failed: ${err?.message ?? String(err)})`);
821
+ }
822
+ return lines.join("\n");
823
+ }
824
+ /**
825
+ * A ready-to-use single-node Kubernetes cluster (k3s). Drop into
826
+ * `environment.services`:
827
+ *
828
+ * ```ts
829
+ * services: { k8s: k3s() }
830
+ * ```
831
+ *
832
+ * Tests get `@kubernetes/client-node` API objects pre-wired to this
833
+ * cluster at `ctx.svc.<key>.client` — `core`, `apps`, and a generic
834
+ * `objects` (`KubernetesObjectApi`). Every API call is recorded on the
835
+ * test event log alongside `fetch` calls. There's also `apply(yaml)`
836
+ * sugar for piping a multi-document manifest in.
837
+ *
838
+ * **Ingress.** We deploy Traefik ourselves in `hostNetwork` mode
839
+ * during `setup()`. Traefik binds the cluster container's :80 directly
840
+ * (no ServiceLB / klipper-lb needed), watches Ingress objects via the
841
+ * API, and routes incoming requests to pod Endpoints. Any `hostnames`
842
+ * declared on this service in env.ts therefore route through Traefik:
843
+ * a peer doing `fetch("http://app.example.com")` resolves the host to
844
+ * the k3s container's IP (via systest-resolver), lands on Traefik,
845
+ * and gets dispatched to the matching Ingress rule's backend pods.
846
+ *
847
+ * **Workarounds for Freestyle's kernel** (Linux 6.1.0-x-freestyle).
848
+ * The stock kernel is missing the `xt_comment` netfilter match
849
+ * extension. Two consequences, each handled below:
850
+ *
851
+ * 1. *kube-proxy* in default iptables mode generates rules with
852
+ * `-m comment --comment "..."`, which the kernel rejects —
853
+ * breaking pod→ClusterIP routing and every pod that talks to the
854
+ * in-cluster API (helm-install Jobs, CoreDNS, …). Fixed by
855
+ * `--kube-proxy-arg=proxy-mode=nftables`: kube-proxy emits
856
+ * native nftables rules where comments are a first-class
857
+ * construct, no xt_comment dependency. nftables proxy mode is
858
+ * GA in k8s 1.32, which is why we pin that.
859
+ *
860
+ * 2. *CNI portmap plugin* (used by klipper-lb's hostPort to expose
861
+ * LoadBalancer ports on the host) still uses iptables-nft and
862
+ * hits the same xt_comment failure — there's no equivalent
863
+ * flag to switch it to native nftables. Workaround: disable the
864
+ * bundled traefik + ServiceLB and run Traefik with
865
+ * `hostNetwork: true` ourselves. hostNetwork pods don't go
866
+ * through portmap at all (they share the node's netns directly),
867
+ * so the broken plugin is never invoked.
868
+ *
869
+ * Flannel uses the `host-gw` backend because Freestyle's stock kernel
870
+ * lacks the VXLAN module — fine for single-node clusters.
871
+ */
872
+ /**
873
+ * Default k3s docker image tag.
874
+ *
875
+ * The cluster's system images (the `rancher/k3s` image itself, plus
876
+ * coredns / local-path-provisioner / pause and the Traefik we deploy)
877
+ * are pulled on first boot through the host `zot` pull-through cache —
878
+ * `registries.yaml` (seeded via `files`) mirrors `docker.io` and
879
+ * `registry.k8s.io` at it. The first-ever cluster boot on a cold-cache
880
+ * host pays the upstream pull once; thereafter zot serves the blobs
881
+ * host-wide and the warm-template snapshot captures the booted cluster,
882
+ * so neither cold-cache nor warm starts re-pull. Any `opts.version`
883
+ * works — there's no base-snapshot release to keep in sync with.
884
+ *
885
+ * **Why v1.32.x:** kube-proxy's `nftables` proxy mode is GA in k8s 1.32
886
+ * (beta in 1.31, alpha-gated in 1.30). The component runs kube-proxy in
887
+ * this mode to sidestep Freestyle's missing `xt_comment` netfilter
888
+ * extension; dropping below 1.31 reintroduces the broken iptables path.
889
+ */
890
+ const DEFAULT_K3S_VERSION = "v1.32.1-k3s1";
891
+ export function k3s(opts = {}) {
892
+ const version = opts.version ?? DEFAULT_K3S_VERSION;
893
+ const extra = opts.extraArgs ?? [];
894
+ const registryEnabled = opts.registry !== false;
895
+ // Wildcard ingress domains. Drives both the `provides(... dnsName)`
896
+ // wiring below and (when non-empty) the CA-signed TLS default cert that
897
+ // setupK3sCluster mints so these domains are reachable over HTTPS.
898
+ const ingressDomains = opts.ingressDomains ?? [];
899
+ // `/etc/rancher/k3s/registries.yaml` (host-cache mirrors + trust for
900
+ // the in-cluster registry). Seeded via `files` because k3s reads it
901
+ // only at startup, before any setup hook could run.
902
+ const registriesYaml = buildRegistriesYaml(registryEnabled);
903
+ const serverArgs = [
904
+ "k3s",
905
+ "server",
906
+ // CoreDNS / pod DNS upstream. Without this, k3s sees only the
907
+ // loopback 127.0.0.11 (Docker's embedded DNS) in the container's
908
+ // /etc/resolv.conf, decides no usable nameserver exists, and writes a
909
+ // fallback `nameserver 8.8.8.8` that CoreDNS then forwards to — so
910
+ // pods reach the public internet but NOT peer services on
911
+ // spectest-net (`<svc>.internal`, fakes, service-TLS hosts all
912
+ // NXDOMAIN). We instead point k3s at the container's default gateway
913
+ // — the spectest-net bridge gateway, where spectest-resolver binds a
914
+ // second listener for exactly this. The file is written by the
915
+ // command wrapper below because the gateway IP is only known at
916
+ // container start.
917
+ "--resolv-conf=/run/spectest-resolv.conf",
918
+ // metrics-server isn't useful in a test cluster.
919
+ "--disable=metrics-server",
920
+ // traefik + servicelb disabled: their klipper-lb DaemonSet uses
921
+ // CNI portmap to bind host port 80, which still needs xt_comment
922
+ // (the iptables compat path that the kernel can't satisfy).
923
+ // We install Traefik with hostNetwork in setup() — same effect,
924
+ // no portmap involved. local-storage stays enabled: it's a
925
+ // controller pod that doesn't bind host ports.
926
+ "--disable=traefik",
927
+ "--disable=servicelb",
928
+ // Pod CIDR MUST avoid 10.42.0.0/16: that's the spectest-br0 host
929
+ // bridge subnet, whose gateway 10.42.0.1 fronts the host image caches
930
+ // (zot :5000-5007, buildkitd :1234). k3s's *default* pod CIDR is also
931
+ // 10.42.0.0/16 — with --flannel-backend=host-gw, flannel programs that
932
+ // route into the node's own routing table and gives cni0 the subnet's
933
+ // .1 (10.42.0.1). That shadows the route to the host gateway, so once
934
+ // CNI comes up the node can no longer reach 10.42.0.1 and every
935
+ // subsequent registry pull dies with "connect: connection refused"
936
+ // (e.g. the in-cluster registry's `registry:2`, applied after the
937
+ // cluster is up — the airgap-bundled system images pull *before* CNI
938
+ // and so sneak through). Move pods to 10.44/service to 10.45.
939
+ "--cluster-cidr=10.44.0.0/16",
940
+ "--service-cidr=10.45.0.0/16",
941
+ "--cluster-dns=10.45.0.10",
942
+ "--flannel-backend=host-gw",
943
+ "--write-kubeconfig-mode=644",
944
+ // kube-proxy in nftables mode: native nftables rules, no
945
+ // xt_comment dependency. Pod→ClusterIP routing works, so
946
+ // CoreDNS / helm-install / anything-talking-to-the-API works.
947
+ // GA in k8s 1.32.
948
+ "--kube-proxy-arg=proxy-mode=nftables",
949
+ ...extra,
950
+ ].join(" ");
951
+ // The service `command` runs under `/bin/sh -c` (see runContainer in
952
+ // daemon.ts), so derive the bridge gateway from the container's default
953
+ // route at start time, write it as the k3s resolv-conf, then exec k3s
954
+ // (exec so it stays the container's main process and signals / the
955
+ // readyCheck behave exactly as before). `/run` is a tmpfs on this
956
+ // service, so the file is writable and never persisted into a snapshot.
957
+ const cmd = "GW=\"$(ip route 2>/dev/null | awk '/^default/{print $3; exit}')\"; " +
958
+ 'if [ -n "$GW" ]; then ' +
959
+ "printf 'nameserver %s\\noptions ndots:0\\n' \"$GW\" > /run/spectest-resolv.conf; " +
960
+ "else echo 'spectest: no default gateway found; k3s pod DNS for peer services will not resolve' >&2; " +
961
+ ": > /run/spectest-resolv.conf; fi; " +
962
+ `exec ${serverArgs}`;
963
+ // Plain /readyz probe. On a warm zot cache the cluster's images are
964
+ // already local, so the first boot completes in seconds; the
965
+ // first-ever boot on a cold-cache host pulls through the mirror and
966
+ // can take a couple of minutes (covered by readyTimeoutSecs).
967
+ const readyCmd = "kubectl get --raw=/readyz >/dev/null 2>&1";
968
+ const def = {
969
+ image: { type: "registry", reference: `rancher/k3s:${version}` },
970
+ command: cmd,
971
+ privileged: true,
972
+ tmpfs: ["/run", "/var/run"],
973
+ cgroupns: "host",
974
+ // 80/443 are advisory — peer services and host code reach them via
975
+ // the k3s container's IP. ServiceLB (klipper-lb) binds them inside
976
+ // the container's netns and forwards to the traefik pod. 5000 is the
977
+ // in-cluster registry (hostNetwork pod bound to the container netns),
978
+ // reached by peers at the cluster's own `<key>.internal:5000` alias.
979
+ ports: registryEnabled ? [80, 443, 6443, K3S_REGISTRY_PORT] : [80, 443, 6443],
980
+ // NOTE: do NOT mount /var/lib/rancher/k3s/agent/containerd as a cache
981
+ // volume. It was tried (to spare a recreated cluster re-pulling its
982
+ // system images on delta restores) and a fresh k3s server against the
983
+ // previous container's containerd store — killed un-cleanly by the
984
+ // teardown's `docker rm -f` — wedged the apiserver minutes in
985
+ // (rollouts never settled, pod listing started failing). The zot
986
+ // mirror already makes those re-pulls cheap; the residual win wasn't
987
+ // worth the recovery semantics of a crash-state store under a fresh
988
+ // cluster db.
989
+ ...(registriesYaml
990
+ ? {
991
+ files: [
992
+ { path: "/etc/rancher/k3s/registries.yaml", content: registriesYaml },
993
+ ],
994
+ }
995
+ : {}),
996
+ readyCheck: {
997
+ type: "exec",
998
+ command: readyCmd,
999
+ timeoutSecs: opts.readyTimeoutSecs ?? 120,
1000
+ },
1001
+ setup: async ({ name, helpers }) => {
1002
+ await setupK3sCluster(name, helpers, {
1003
+ registry: registryEnabled,
1004
+ ingressDomains,
1005
+ });
1006
+ },
1007
+ helpers: async ({ name, exec }) => {
1008
+ // Read the cluster's kubeconfig and address the API server by its
1009
+ // auto-assigned `<name>.internal` hostname on spectest-net. TLS
1010
+ // verification is off (see the K3sHelpers docstring), so the
1011
+ // server's cert SAN list doesn't need to include the .internal
1012
+ // name.
1013
+ const kcRead = await exec(name, ["cat", "/etc/rancher/k3s/k3s.yaml"]);
1014
+ if (kcRead.exitCode !== 0) {
1015
+ throw new Error(`k3s(${name}): failed to read kubeconfig from container: ${kcRead.stderr.trim()}`);
1016
+ }
1017
+ const kubeconfig = new KubeConfig();
1018
+ kubeconfig.loadFromString(kcRead.stdout);
1019
+ const server = `https://${name}.internal:6443`;
1020
+ // Update kc.clusters so any code that reads kubeconfig sees the
1021
+ // right server URL, but the actual request server comes from the
1022
+ // Configuration we build below. `Cluster.server` is typed `readonly`
1023
+ // by @kubernetes/client-node, but the loaded object is a plain mutable
1024
+ // record — write through a mutable view rather than rebuild the config.
1025
+ for (const cluster of kubeconfig.clusters) {
1026
+ cluster.server =
1027
+ server;
1028
+ }
1029
+ const httpApi = new FetchHttpLibrary();
1030
+ const baseServer = new ServerConfiguration(server, {});
1031
+ // eslint-disable-next-line @typescript-eslint/no-explicit-any
1032
+ const config = createConfiguration({
1033
+ baseServer,
1034
+ authMethods: { default: kubeconfig },
1035
+ httpApi: httpApi,
1036
+ });
1037
+ const core = withTagging(new CoreV1Api(config));
1038
+ const apps = withTagging(new AppsV1Api(config));
1039
+ const objects = withTagging(new KubernetesObjectApi(config));
1040
+ const apply = async (manifest) => {
1041
+ const docs = loadAllYaml(manifest);
1042
+ const out = [];
1043
+ for (const doc of docs) {
1044
+ if (!doc || typeof doc !== "object" || !("kind" in doc))
1045
+ continue;
1046
+ // `objects.create` is wrapped by `withTagging`, so each returned
1047
+ // object already carries the back-reference to its create call.
1048
+ const created = await objects.create(doc);
1049
+ out.push(created);
1050
+ }
1051
+ return out;
1052
+ };
1053
+ return {
1054
+ kubeconfig,
1055
+ client: { core, apps, objects },
1056
+ apply,
1057
+ };
1058
+ },
1059
+ };
1060
+ // Wildcard ingress domains → a dnsName(`*.<domain>`, { service: self })
1061
+ // each, attached via provides(). SELF_SERVICE_TOKEN resolves to this
1062
+ // service's key at load time (the component can't know it here). The
1063
+ // resolver then points every host under the domain at the cluster.
1064
+ if (ingressDomains.length === 0)
1065
+ return def;
1066
+ return provides(def, ingressDomains.map((domain) => dnsName(`*.${domain}`, { service: SELF_SERVICE_TOKEN })));
1067
+ }