@specific.dev/spectest 0.39.0 → 0.43.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (105) hide show
  1. package/dist/browser.d.ts +21 -8
  2. package/dist/browser.js +78 -36
  3. package/dist/components/supabase.d.ts +87 -27
  4. package/dist/components/supabase.js +352 -69
  5. package/dist/daemon.d.ts +38 -0
  6. package/dist/daemon.js +464 -987
  7. package/dist/harness/build-context.d.ts +82 -0
  8. package/dist/harness/build-context.js +113 -0
  9. package/dist/harness/buildkit-progress.d.ts +37 -0
  10. package/dist/harness/buildkit-progress.js +66 -0
  11. package/dist/harness/container-run.d.ts +89 -0
  12. package/dist/harness/container-run.js +118 -0
  13. package/dist/harness/file-mounts.d.ts +91 -0
  14. package/dist/harness/file-mounts.js +119 -0
  15. package/dist/harness/hostmatch.d.ts +65 -0
  16. package/dist/harness/hostmatch.js +108 -0
  17. package/dist/harness/http-proxy.d.ts +62 -0
  18. package/dist/harness/http-proxy.js +104 -0
  19. package/dist/harness/ingress-table.d.ts +148 -0
  20. package/dist/harness/ingress-table.js +129 -0
  21. package/dist/harness/log-delta.d.ts +54 -0
  22. package/dist/harness/log-delta.js +83 -0
  23. package/dist/harness/main.d.ts +47 -0
  24. package/dist/harness/main.js +164 -0
  25. package/dist/harness/methods.d.ts +54 -0
  26. package/dist/harness/methods.js +65 -0
  27. package/dist/harness/names-registry.d.ts +63 -0
  28. package/dist/harness/names-registry.js +90 -0
  29. package/dist/harness/protocol.d.ts +88 -0
  30. package/dist/harness/protocol.js +96 -0
  31. package/dist/harness/ready-poll.d.ts +47 -0
  32. package/dist/harness/ready-poll.js +67 -0
  33. package/dist/harness/service-graph.d.ts +29 -0
  34. package/dist/harness/service-graph.js +92 -0
  35. package/dist/harness/volume-paths.d.ts +70 -0
  36. package/dist/harness/volume-paths.js +81 -0
  37. package/dist/index.d.ts +58 -16
  38. package/dist/ingress.d.ts +1 -1
  39. package/dist/mobile.d.ts +9 -5
  40. package/dist/mobile.js +7 -6
  41. package/dist/recorder.d.ts +10 -0
  42. package/dist/resolver.js +5 -8
  43. package/dist/vendor/rrweb-plugin-console-record.umd.js +521 -0
  44. package/dist/vendor/rrweb-record.min.js +5061 -0
  45. package/package.json +7 -1
  46. package/src/aws-sigv4.ts +218 -0
  47. package/src/browser.ts +2095 -0
  48. package/src/components/aws.ts +554 -0
  49. package/src/components/email.ts +398 -0
  50. package/src/components/expo.ts +167 -0
  51. package/src/components/index.ts +81 -0
  52. package/src/components/k3s.ts +2061 -0
  53. package/src/components/postgres.ts +132 -0
  54. package/src/components/replayFake.ts +1015 -0
  55. package/src/components/s3.ts +132 -0
  56. package/src/components/supabase.ts +1699 -0
  57. package/src/daemon.ts +5537 -0
  58. package/src/harness/build-context.test.ts +0 -0
  59. package/src/harness/build-context.ts +146 -0
  60. package/src/harness/buildkit-progress.test.ts +98 -0
  61. package/src/harness/buildkit-progress.ts +74 -0
  62. package/src/harness/container-run.test.ts +209 -0
  63. package/src/harness/container-run.ts +158 -0
  64. package/src/harness/file-mounts.test.ts +185 -0
  65. package/src/harness/file-mounts.ts +145 -0
  66. package/src/harness/hostmatch.test.ts +148 -0
  67. package/src/harness/hostmatch.ts +109 -0
  68. package/src/harness/http-proxy.test.ts +156 -0
  69. package/src/harness/http-proxy.ts +119 -0
  70. package/src/harness/ingress-rebind.test.ts +125 -0
  71. package/src/harness/ingress-table.test.ts +172 -0
  72. package/src/harness/ingress-table.ts +186 -0
  73. package/src/harness/log-delta.test.ts +125 -0
  74. package/src/harness/log-delta.ts +100 -0
  75. package/src/harness/main.test.ts +211 -0
  76. package/src/harness/main.ts +196 -0
  77. package/src/harness/methods.test.ts +63 -0
  78. package/src/harness/methods.ts +92 -0
  79. package/src/harness/names-registry.test.ts +137 -0
  80. package/src/harness/names-registry.ts +108 -0
  81. package/src/harness/protocol.test.ts +148 -0
  82. package/src/harness/protocol.ts +163 -0
  83. package/src/harness/ready-poll.test.ts +172 -0
  84. package/src/harness/ready-poll.ts +93 -0
  85. package/src/harness/service-graph.test.ts +97 -0
  86. package/src/harness/service-graph.ts +97 -0
  87. package/src/harness/volume-paths.test.ts +102 -0
  88. package/src/harness/volume-paths.ts +112 -0
  89. package/src/ids.ts +89 -0
  90. package/src/index.ts +2767 -0
  91. package/src/ingress.ts +305 -0
  92. package/src/inspect.ts +739 -0
  93. package/src/locator.ts +716 -0
  94. package/src/mobile.ts +138 -0
  95. package/src/record-secrets.ts +41 -0
  96. package/src/recorder.ts +856 -0
  97. package/src/redis.ts +202 -0
  98. package/src/replay-bundle.ts +108 -0
  99. package/src/resolver.ts +348 -0
  100. package/src/s3.ts +333 -0
  101. package/src/sql.ts +243 -0
  102. package/src/terminal.ts +740 -0
  103. package/src/url-match.ts +67 -0
  104. package/src/vendor/rrweb-plugin-console-record.umd.js +521 -0
  105. package/src/vendor/rrweb-record.min.js +5061 -0
@@ -0,0 +1,2061 @@
1
+ import { AsyncLocalStorage } from "node:async_hooks";
2
+ import { spawn as nodeSpawn } from "node:child_process";
3
+ import { randomUUID } from "node:crypto";
4
+ import { existsSync, readFileSync } from "node:fs";
5
+ import { readFile, unlink } from "node:fs/promises";
6
+
7
+ import {
8
+ AppsV1Api,
9
+ BatchV1Api,
10
+ CoreV1Api,
11
+ KubeConfig,
12
+ KubernetesObjectApi,
13
+ PatchStrategy,
14
+ ResponseContext,
15
+ ServerConfiguration,
16
+ createConfiguration,
17
+ loadAllYaml,
18
+ type KubernetesListObject,
19
+ type KubernetesObject,
20
+ type RequestContext,
21
+ type V1APIResource,
22
+ type V1DeleteOptions,
23
+ type V1DaemonSetStatus,
24
+ type V1DeploymentStatus,
25
+ type V1Job,
26
+ type V1StatefulSetStatus,
27
+ type V1Status,
28
+ } from "@kubernetes/client-node";
29
+ import { Observable } from "@kubernetes/client-node/dist/gen/rxjsStub.js";
30
+
31
+ import type {
32
+ ServiceDefinition,
33
+ ServiceHelpersContext,
34
+ SpectestContext,
35
+ } from "../index.js";
36
+ import { dnsName, provides, SELF_SERVICE_TOKEN } from "../index.js";
37
+ import { deepUnwrap, readTag, wrap } from "../inspect.js";
38
+ import type { Wrapped } from "../inspect.js";
39
+ import { recorderAnnotate, recorderRemove } from "../recorder.js";
40
+
41
+ export type {
42
+ KubernetesObject,
43
+ KubernetesListObject,
44
+ } from "@kubernetes/client-node";
45
+
46
+ export interface K3sOptions {
47
+ /** Image tag for the official `rancher/k3s` image. Default `"v1.30.6-k3s1"`. */
48
+ version?: string;
49
+ /**
50
+ * Extra arguments appended to `k3s server`. Useful for `--tls-san=...`,
51
+ * additional `--disable=<addon>`, custom CIDRs, etc.
52
+ *
53
+ * Two values are **rejected** rather than silently ignored, because the
54
+ * component owns those decisions and passing them here has no effect:
55
+ * `--disable=traefik` (use `traefik: false`) and `--disable=servicelb`
56
+ * (use `loadBalancer: false`).
57
+ */
58
+ extraArgs?: string[];
59
+ /**
60
+ * Install the component's own Traefik ingress controller (hostNetwork,
61
+ * default IngressClass, wired to the in-VM CA when `ingressDomains` is
62
+ * set). Default `true`.
63
+ *
64
+ * Set `false` to run a cluster with no ingress controller at all — for
65
+ * a project that installs its own (via Helm, an operator, or a raw
66
+ * manifest) and wants the port to itself. The bundled k3s Traefik is
67
+ * disabled either way, so `false` really does mean none.
68
+ */
69
+ traefik?: boolean;
70
+ /**
71
+ * Run k3s's ServiceLB (klipper-lb) so `type: LoadBalancer` Services are
72
+ * assigned an address instead of sitting at `<pending>` forever.
73
+ * Default `true`.
74
+ *
75
+ * The assigned address is the **node IP** — the k3s container's own IP
76
+ * on `spectest-net` — so a LoadBalancer on port 8080 is reachable from
77
+ * a test or a peer service at `<cluster-key>.internal:8080`. Because
78
+ * klipper-lb binds the port on the node, a LoadBalancer that asks for
79
+ * port 80 or 443 collides with the component's hostNetwork Traefik;
80
+ * either pick another port or pass `traefik: false`.
81
+ */
82
+ loadBalancer?: boolean;
83
+ /**
84
+ * Readiness probe timeout in seconds. k3s on a warm image is ready in
85
+ * a few seconds; the first cold start of an env (image pull + cluster
86
+ * bootstrap) can take 30–60s. Default `120`.
87
+ */
88
+ readyTimeoutSecs?: number;
89
+ /**
90
+ * Run an in-cluster OCI registry (CNCF `distribution` / `registry:2`)
91
+ * that the cluster's own containerd trusts. This is the hermetic
92
+ * stand-in for a cloud registry (ECR/GCR/GHCR): a peer service builds
93
+ * an image, pushes it here over plain HTTP, references
94
+ * `<cluster-key>.internal:5000/...` from a Deployment, and the kubelet
95
+ * pulls it straight back. Lets you test a real
96
+ * `build → push → deploy → pull` pipeline with no external registry
97
+ * and no image pre-baking.
98
+ *
99
+ * **The registry's push/pull address is `<cluster-key>.internal:5000`**,
100
+ * where `<cluster-key>` is the key you give this service in the
101
+ * `services` map (e.g. a cluster at `services.cluster` is reachable at
102
+ * `cluster.internal:5000`). That's the cluster service's own
103
+ * unconditional `.internal` alias — peer-reachable and never clobbered
104
+ * by any `hostnames` you set — so it resolves identically from peer
105
+ * containers (push) and the cluster's own containerd (pull). Wire it
106
+ * into your platform, e.g. `env: { REGISTRY_URL: "cluster.internal:5000" }`.
107
+ * Plain HTTP, so configure your push client for an insecure registry.
108
+ *
109
+ * On by default. Set `false` for clusters that only ever run public
110
+ * images — that skips the extra pod.
111
+ */
112
+ registry?: boolean;
113
+ /**
114
+ * Domains to route into this cluster's ingress via **wildcard DNS**. For
115
+ * each `"example.com"`, spectest-resolver answers any `*.example.com`
116
+ * query with the cluster container's IP, where Traefik dispatches by Host
117
+ * to the matching Ingress. This lets a test `kubectl apply` an Ingress for
118
+ * any host under the domain and reach it immediately — no need to
119
+ * pre-declare each hostname in `hostnames`.
120
+ *
121
+ * ```ts
122
+ * services: { k8s: k3s({ ingressDomains: ["example.com"] }) }
123
+ * // a test then applies an Ingress for foo.example.com and fetches it.
124
+ * ```
125
+ *
126
+ * For one-off hosts not under a declared domain, a test can also register
127
+ * dynamically with `ctx.dnsName(host, { service: "k8s" })`.
128
+ *
129
+ * **TLS.** Setting `ingressDomains` also makes those domains reachable
130
+ * over **HTTPS**: Traefik gains a `:443` entrypoint and serves a default
131
+ * certificate minted from the in-VM root CA with SANs `*.<domain>` for
132
+ * each declared domain. The CA is already trusted by the test framework
133
+ * (Node `fetch`, `ctx.browser()`, Python, the system store), so
134
+ * `ctx.fetch("https://foo.example.com")` gets a clean handshake — no
135
+ * per-Ingress `spec.tls` and no `--insecure` needed. Only hosts **under**
136
+ * a declared domain are covered by the cert; static `hostnames` not under
137
+ * one (and one-off `ctx.dnsName` hosts) remain HTTP-only.
138
+ */
139
+ ingressDomains?: string[];
140
+ }
141
+
142
+ /** Port the in-cluster registry listens on (plain HTTP). */
143
+ const K3S_REGISTRY_PORT = 5000;
144
+
145
+ /**
146
+ * `extraArgs` entries the component overrides anyway. Passing one of
147
+ * these looks like it works — the flag really is appended to the k3s
148
+ * command line — but the component's own behaviour is what decides the
149
+ * outcome, so the cluster comes up contradicting the argument. Failing
150
+ * at load time costs a second; discovering it costs a boot cycle.
151
+ */
152
+ const OVERRIDDEN_EXTRA_ARGS: Record<string, string> = {
153
+ traefik:
154
+ "the component installs its own Traefik (hostNetwork) after the cluster is up — " +
155
+ "pass `traefik: false` to have no ingress controller at all",
156
+ servicelb:
157
+ "ServiceLB is controlled by the `loadBalancer` option — pass `loadBalancer: false` to disable it",
158
+ };
159
+
160
+ function assertUsableExtraArgs(extra: readonly string[]): void {
161
+ for (const arg of extra) {
162
+ const m = /^--disable[=\s]+(.+)$/.exec(arg.trim());
163
+ const addon = m?.[1]?.trim();
164
+ if (!addon) continue;
165
+ const why = OVERRIDDEN_EXTRA_ARGS[addon];
166
+ if (why) {
167
+ throw new Error(
168
+ `k3s(): extraArgs cannot control "${addon}" — ${why}. ` +
169
+ `(Remove ${JSON.stringify(arg)} from extraArgs.)`,
170
+ );
171
+ }
172
+ }
173
+ }
174
+
175
+ /**
176
+ * Rewrites a `@kubernetes/client-node` API class so each method's resolved
177
+ * value comes back **inspect-wrapped** ({@link Wrapped}) — exactly what
178
+ * `withTagging` does at runtime. The method *signatures* (argument types) are
179
+ * untouched; only the `Promise<R>` result becomes `Promise<Wrapped<R>>`, so
180
+ * `expect(pod.status.phase)` links to the API call with no cast and you
181
+ * `.unwrap()` before using a value as raw data. Non-method
182
+ * members pass through unchanged. Mirrors {@link RecordingSqlClient} on the
183
+ * postgres side.
184
+ */
185
+ type Tagged<T> = {
186
+ [K in keyof T]: T[K] extends (...args: infer A) => Promise<infer R>
187
+ ? (...args: A) => Promise<Wrapped<R>>
188
+ : T[K];
189
+ };
190
+
191
+ /**
192
+ * `{ apiVersion, kind, metadata: { name } }` — enough to address an
193
+ * existing object. Redeclared here because `@kubernetes/client-node`
194
+ * keeps its equivalent (`KubernetesObjectHeader`) module-private.
195
+ */
196
+ export type KubernetesObjectRef<T extends KubernetesObject = KubernetesObject> =
197
+ Pick<T, "apiVersion" | "kind"> & {
198
+ metadata: { name: string; namespace?: string };
199
+ };
200
+
201
+ /**
202
+ * The dynamic object API, hand-declared rather than derived through
203
+ * {@link Tagged}.
204
+ *
205
+ * `Tagged<T>` infers each method from its *erased* signature, and a mapped
206
+ * type can't reintroduce a type parameter (TypeScript has no higher-kinded
207
+ * types). For `CoreV1Api`/`AppsV1Api` that's invisible — their methods
208
+ * return concrete types — but `KubernetesObjectApi.read<T>`/`list<T>` are
209
+ * generic, so mapping them collapsed `read<MyCustomResource>(…)` to
210
+ * `Promise<Wrapped<KubernetesObject>>`: `Expected 0 type arguments, but got
211
+ * 1`, then `Property 'spec' does not exist on type 'KubernetesObject'` —
212
+ * precisely on the custom resources this API exists to reach. Worse, the
213
+ * degraded value made `expect(...)` resolve to the *locator* overload, so
214
+ * the reported error was a baffling `Property 'toBe' does not exist on type
215
+ * 'LocatorAssertion'` three lines further down.
216
+ *
217
+ * So the generic methods are written out. Arguments match the library's
218
+ * (positional, as it declares them); only the resolved value becomes
219
+ * {@link Wrapped}. See {@link K3sHelpers.list} for an options-object
220
+ * wrapper over the positional `list`.
221
+ */
222
+ export interface TaggedObjectApi {
223
+ create<T extends KubernetesObject>(
224
+ spec: T,
225
+ pretty?: string,
226
+ dryRun?: string,
227
+ fieldManager?: string,
228
+ ): Promise<Wrapped<T>>;
229
+ patch<T extends KubernetesObject>(
230
+ spec: T,
231
+ pretty?: string,
232
+ dryRun?: string,
233
+ fieldManager?: string,
234
+ force?: boolean,
235
+ patchStrategy?: PatchStrategy,
236
+ ): Promise<Wrapped<T>>;
237
+ /** Read one object. **Throws** on 404 — to assert something is *gone*,
238
+ * prefer {@link K3sHelpers.list} with a `metadata.name` field selector
239
+ * and assert the returned `items` are empty (a throw carries no value
240
+ * to assert on, and no provenance link). */
241
+ read<T extends KubernetesObject>(
242
+ spec: KubernetesObjectRef<T>,
243
+ pretty?: string,
244
+ exact?: boolean,
245
+ exportt?: boolean,
246
+ ): Promise<Wrapped<T>>;
247
+ replace<T extends KubernetesObject>(
248
+ spec: T,
249
+ pretty?: string,
250
+ dryRun?: string,
251
+ fieldManager?: string,
252
+ ): Promise<Wrapped<T>>;
253
+ delete(
254
+ spec: KubernetesObject,
255
+ pretty?: string,
256
+ dryRun?: string,
257
+ gracePeriodSeconds?: number,
258
+ orphanDependents?: boolean,
259
+ propagationPolicy?: string,
260
+ body?: V1DeleteOptions,
261
+ ): Promise<Wrapped<V1Status>>;
262
+ list<T extends KubernetesObject>(
263
+ apiVersion: string,
264
+ kind: string,
265
+ namespace?: string,
266
+ pretty?: string,
267
+ exact?: boolean,
268
+ exportt?: boolean,
269
+ fieldSelector?: string,
270
+ labelSelector?: string,
271
+ limit?: number,
272
+ continueToken?: string,
273
+ ): Promise<Wrapped<KubernetesListObject<T>>>;
274
+ }
275
+
276
+ /**
277
+ * Pre-instantiated `@kubernetes/client-node` API clients sharing the
278
+ * same recording HTTP transport. Every method call lands on the test
279
+ * event log as an HTTP event alongside `fetch` calls, and its result is
280
+ * inspect-wrapped (see {@link Tagged}) so assertions on it stay linked.
281
+ */
282
+ export interface K3sClient {
283
+ core: Tagged<CoreV1Api>;
284
+ apps: Tagged<AppsV1Api>;
285
+ batch: Tagged<BatchV1Api>;
286
+ /** Generic object API — `create()`, `read()`, `patch()`, `delete()`,
287
+ * `list()` against any Kubernetes resource, custom resources included.
288
+ * Generic in the resource type: `objects.read<MyCr>({ … })` resolves to
289
+ * `Wrapped<MyCr>`. */
290
+ objects: TaggedObjectApi;
291
+ }
292
+
293
+ /** Options for {@link K3sHelpers.apply}. */
294
+ export interface ApplyOptions {
295
+ /** Namespace for documents that don't declare one (cluster-scoped kinds
296
+ * are unaffected). Defaults to `default`, like `kubectl`. */
297
+ namespace?: string;
298
+ /** Server-side-apply field manager. Default `"spectest"`. Keep it stable
299
+ * across runs — that's what makes re-applying an *update* rather than a
300
+ * conflict. */
301
+ fieldManager?: string;
302
+ /** Take ownership of fields another manager owns, i.e. `kubectl apply
303
+ * --force-conflicts`. Default `true`: a test environment has one owner
304
+ * and a conflict is noise. */
305
+ force?: boolean;
306
+ /** Total budget for converging the whole manifest set, including retries
307
+ * of documents whose CRD/webhook isn't up yet. Default 120 s. */
308
+ timeoutMs?: number;
309
+ }
310
+
311
+ /** A workload to wait on: `"web"`, `"deployment/web"`,
312
+ * `"statefulset/db"`, `"daemonset/agent"`, or an object with
313
+ * `kind`/`metadata` (e.g. an element of {@link K3sHelpers.apply}'s
314
+ * result). Bare names are Deployments. */
315
+ export type RolloutTarget = string | KubernetesObject | KubernetesObjectRef;
316
+
317
+ /** Options for the wait helpers. */
318
+ export interface WaitOptions {
319
+ /** Namespace of the object. Default `default`, or the object's own. */
320
+ namespace?: string;
321
+ /** Give up after this long. Default 120 s. */
322
+ timeoutMs?: number;
323
+ /** Poll interval. Default 500 ms. */
324
+ intervalMs?: number;
325
+ }
326
+
327
+ /** Which pods to read logs from — see {@link K3sHelpers.logs}. */
328
+ export interface LogsOptions {
329
+ /** A single pod by name. */
330
+ pod?: string;
331
+ /** All pods matching a label selector (`"app=web"`). */
332
+ selector?: string;
333
+ namespace?: string;
334
+ /** Container within the pod (required only for multi-container pods). */
335
+ container?: string;
336
+ /** Tail this many lines per pod. Default 200. */
337
+ tailLines?: number;
338
+ }
339
+
340
+ /** Helpers a `k3s(...)` service exposes on `ctx.svc.<name>`. */
341
+ export interface K3sHelpers {
342
+ /**
343
+ * Fully-loaded `KubeConfig`. The cluster server URL is rewritten to
344
+ * `https://<service-name>.internal:6443`, the auto-assigned DNS name
345
+ * for this service on `spectest-net`. TLS verification is off because
346
+ * Bun's fetch doesn't honor an https.Agent's CA option (the client
347
+ * cert from the kubeconfig still flows through for auth), so the
348
+ * server's cert SAN list doesn't need to include the .internal name.
349
+ */
350
+ kubeconfig: KubeConfig;
351
+ /** Pre-built API clients. */
352
+ client: K3sClient;
353
+ /**
354
+ * Apply a (multi-document) YAML manifest with **server-side apply**
355
+ * semantics — the API equivalent of `kubectl apply --server-side
356
+ * --force-conflicts`. Each document is `PATCH`ed with
357
+ * `application/apply-patch+yaml`, so applying is create-or-update:
358
+ * re-applying an edited manifest converges instead of failing
359
+ * `AlreadyExists`, which is what makes it usable for real manifest sets
360
+ * (and for a `dependsOn` child re-applying over its parent's state).
361
+ *
362
+ * Documents that can't land *yet* are retried until `opts.timeoutMs`:
363
+ * a CR whose CRD is in the same manifest, or a resource an admission
364
+ * webhook rejects while its own pod is still coming up, both resolve on
365
+ * a later round instead of failing the run. A document that keeps
366
+ * failing throws with the API server's own message.
367
+ *
368
+ * Returns the API server's response objects in input order — each
369
+ * element inspect-wrapped (the array container itself is plain), so
370
+ * `expect(applied[0]!.metadata.uid)` links to its apply call.
371
+ *
372
+ * Nothing here shells out to `kubectl`; the cluster is driven over its
373
+ * API from the daemon.
374
+ */
375
+ apply(
376
+ manifest: string,
377
+ opts?: ApplyOptions,
378
+ ): Promise<Wrapped<KubernetesObject>[]>;
379
+ /**
380
+ * Wait until a Deployment / StatefulSet / DaemonSet has actually rolled
381
+ * out: the controller has observed the current generation, every replica
382
+ * is updated, and the desired number are Ready (the checks `kubectl
383
+ * rollout status` makes). Throws on timeout with the last-seen status
384
+ * plus pod phases, container reasons and recent Warning events from the
385
+ * namespace, so a stuck image pull or CrashLoop is readable from the
386
+ * failure alone.
387
+ *
388
+ * ```ts
389
+ * await ctx.svc.k8s.apply(manifest);
390
+ * await ctx.svc.k8s.waitForRollout("web");
391
+ * await ctx.svc.k8s.waitForRollout("statefulset/db", { namespace: "data" });
392
+ * ```
393
+ */
394
+ waitForRollout(target: RolloutTarget, opts?: WaitOptions): Promise<void>;
395
+ /**
396
+ * Wait for a Job to finish. Resolves with the completed Job (Complete
397
+ * condition, or `succeeded` at the requested completion count) and
398
+ * throws if it fails — with the failed pods' logs attached, which is
399
+ * the thing you actually want when a migration or seed Job dies.
400
+ *
401
+ * ```ts
402
+ * await ctx.svc.k8s.apply(migrateJob);
403
+ * const job = await ctx.svc.k8s.waitForJob("migrate");
404
+ * expect(job.status?.succeeded).toBe(1);
405
+ * ```
406
+ */
407
+ waitForJob(name: string, opts?: WaitOptions): Promise<Wrapped<V1Job>>;
408
+ /**
409
+ * Read pod logs — one pod by `pod`, or every pod matching `selector`
410
+ * (concatenated, each preceded by a `==> <pod> <==` header). The
411
+ * diagnostic companion to the wait helpers.
412
+ */
413
+ logs(opts: LogsOptions): Promise<Wrapped<string>>;
414
+ /**
415
+ * List objects of any kind by options object rather than the library's
416
+ * ten positional parameters (`list(apiVersion, kind, namespace, pretty,
417
+ * exact, exportt, fieldSelector, …)`). Generic in the resource type, so
418
+ * custom resources keep their shape:
419
+ *
420
+ * ```ts
421
+ * // "the Branch is gone" — without a 404 throw to catch
422
+ * const left = await ctx.svc.k8s.list<Branch>({
423
+ * apiVersion: "xata.io/v1", kind: "Branch",
424
+ * fieldSelector: `metadata.name=${name}`,
425
+ * });
426
+ * expect(left.items).toHaveLength(0);
427
+ * ```
428
+ */
429
+ list<T extends KubernetesObject = KubernetesObject>(opts: {
430
+ apiVersion: string;
431
+ kind: string;
432
+ namespace?: string;
433
+ fieldSelector?: string;
434
+ labelSelector?: string;
435
+ limit?: number;
436
+ }): Promise<Wrapped<KubernetesListObject<T>>>;
437
+ }
438
+
439
+ interface DockerExecResult {
440
+ stdout: string;
441
+ stderr: string;
442
+ code: number;
443
+ }
444
+
445
+ function runProcess(
446
+ cmd: string,
447
+ args: string[],
448
+ timeoutMs = 30_000,
449
+ ): Promise<DockerExecResult> {
450
+ return new Promise((resolve, reject) => {
451
+ const cp = nodeSpawn(cmd, args, { stdio: ["ignore", "pipe", "pipe"] });
452
+ const out: Buffer[] = [];
453
+ const err: Buffer[] = [];
454
+ cp.stdout!.on("data", (c) => out.push(c));
455
+ cp.stderr!.on("data", (c) => err.push(c));
456
+ const t = setTimeout(() => cp.kill("SIGKILL"), timeoutMs);
457
+ cp.on("error", (e) => {
458
+ clearTimeout(t);
459
+ reject(e);
460
+ });
461
+ cp.on("close", (code) => {
462
+ clearTimeout(t);
463
+ resolve({
464
+ stdout: Buffer.concat(out).toString("utf8"),
465
+ stderr: Buffer.concat(err).toString("utf8"),
466
+ code: code ?? -1,
467
+ });
468
+ });
469
+ });
470
+ }
471
+
472
+ // In-VM root CA, generated once into the base snapshot (see
473
+ // control-plane `base.rs`). Trusted everywhere the test framework runs —
474
+ // Node (`NODE_EXTRA_CA_CERTS`), Chromium (NSS DB), Python, the system
475
+ // store — so a leaf signed by it gives `ctx.fetch`/`ctx.browser()` a
476
+ // clean HTTPS handshake. The k3s `setup` hook runs inside the daemon's
477
+ // Bun process (root in the VM), so it can read the CA key and mint
478
+ // directly. These constants are intentionally redeclared here rather than
479
+ // imported from the daemon: the SDK ships to end users and must not
480
+ // depend on daemon internals.
481
+ const CA_PATH = process.env.SPECTEST_CA_PATH ?? "/etc/spectest/ca.crt";
482
+ const CA_KEY_PATH = process.env.SPECTEST_CA_KEY_PATH ?? "/etc/spectest/ca.key";
483
+
484
+ function caPresent(): boolean {
485
+ return existsSync(CA_PATH) && existsSync(CA_KEY_PATH);
486
+ }
487
+
488
+ /**
489
+ * Mint a leaf certificate from the in-VM root CA covering `hostnames`
490
+ * (used here as the SANs of the cluster's wildcard ingress domains).
491
+ * Returns the cert + key as PEM strings. Self-contained openssl shell-out
492
+ * — deliberately not shared with the daemon's own cert minting to keep
493
+ * the distributed SDK decoupled from daemon code.
494
+ */
495
+ async function issueIngressCert(
496
+ hostnames: string[],
497
+ ): Promise<{ cert: string; key: string }> {
498
+ const id = `spectest-k3s-ingress-${randomUUID().slice(0, 8)}`;
499
+ const keyPath = `/tmp/${id}.key`;
500
+ const crtPath = `/tmp/${id}.crt`;
501
+ const sans = hostnames.map((h) => `DNS:${h}`).join(",");
502
+ const r = await runProcess(
503
+ "openssl",
504
+ [
505
+ "req",
506
+ "-newkey",
507
+ "rsa:2048",
508
+ "-nodes",
509
+ "-keyout",
510
+ keyPath,
511
+ "-out",
512
+ crtPath,
513
+ "-x509",
514
+ "-CA",
515
+ CA_PATH,
516
+ "-CAkey",
517
+ CA_KEY_PATH,
518
+ "-days",
519
+ "3650",
520
+ "-subj",
521
+ "/CN=spectest-k3s-ingress",
522
+ "-addext",
523
+ `subjectAltName=${sans}`,
524
+ "-addext",
525
+ "basicConstraints=CA:FALSE",
526
+ "-addext",
527
+ "extendedKeyUsage=serverAuth",
528
+ "-addext",
529
+ "keyUsage=digitalSignature,keyEncipherment",
530
+ ],
531
+ 30_000,
532
+ );
533
+ if (r.code !== 0) {
534
+ throw new Error(
535
+ `k3s ingress cert minting failed (openssl rc=${r.code}): ${
536
+ r.stderr.trim() || r.stdout.trim()
537
+ }`,
538
+ );
539
+ }
540
+ try {
541
+ const [cert, key] = await Promise.all([
542
+ readFile(crtPath, "utf8"),
543
+ readFile(keyPath, "utf8"),
544
+ ]);
545
+ return { cert, key };
546
+ } finally {
547
+ await Promise.all([
548
+ unlink(keyPath).catch(() => {}),
549
+ unlink(crtPath).catch(() => {}),
550
+ ]);
551
+ }
552
+ }
553
+
554
+ // Holds the inspector `sourceSeq` for the most recent HTTP call inside
555
+ // a single API-method invocation. Filled in by `doFetch` after each
556
+ // request, read by the `withTagging` proxy when the method's promise
557
+ // resolves so the returned parsed object carries the back-reference.
558
+ // AsyncLocalStorage is the right scope here — every call to an Api
559
+ // method runs in its own holder and concurrent calls don't race.
560
+ interface CallSlot {
561
+ seq?: number;
562
+ }
563
+ const callContext = new AsyncLocalStorage<CallSlot>();
564
+
565
+ // HTTP transport for `@kubernetes/client-node` that routes through
566
+ // `globalThis.fetch`. Two reasons:
567
+ // 1. The daemon's per-test `installFetchWrapper` already records every
568
+ // `globalThis.fetch` call as an HTTP event — using fetch here gets
569
+ // k8s API calls recorded for free, no library-specific wiring.
570
+ // 2. Bun's fetch needs Bun-shaped TLS options (`tls: { ... }`) for
571
+ // mTLS; node-fetch's `agent` parameter — which the library's
572
+ // default transport relies on — is silently ignored under Bun.
573
+ // We honor the lib's auth flow (`KubeConfig.applySecurityAuthentication`
574
+ // sets an Agent on the request) by extracting cert/key off that
575
+ // agent and passing them via the Bun-shaped option.
576
+ class FetchHttpLibrary {
577
+ send(request: RequestContext): Observable<ResponseContext> {
578
+ const promise = doFetch(request);
579
+ return new Observable(promise);
580
+ }
581
+ }
582
+
583
+ /**
584
+ * Parsed Kubernetes semantics of a single API request, derived purely
585
+ * from the HTTP method + request path. Fed to `recorderAnnotate` to
586
+ * reclassify the generic `http` event the fetch wrapper recorded into a
587
+ * Kubernetes-specific `kube` event.
588
+ */
589
+ interface KubeRequestMeta {
590
+ verb: string;
591
+ group?: string;
592
+ apiVersion?: string;
593
+ resource?: string;
594
+ subresource?: string;
595
+ name?: string;
596
+ namespace?: string;
597
+ }
598
+
599
+ /**
600
+ * Map a Kubernetes API request to its `(verb, group/version, resource,
601
+ * namespace, name, subresource)` from the URL + HTTP method alone.
602
+ *
603
+ * Path grammar (the two API roots):
604
+ * - core group: `/api/<version>/...`
605
+ * - named group: `/apis/<group>/<version>/...`
606
+ * after which the remainder is either a cluster-scoped resource
607
+ * (`nodes`, `namespaces`, …) or `namespaces/<ns>/<resource>...`. The
608
+ * trailing `<resource>[/<name>[/<subresource>]]` shape plus the method
609
+ * (and `?watch=`) yields the verb.
610
+ *
611
+ * Returns `null` for non-resource paths — discovery (`/api`, `/apis`,
612
+ * `/apis/<group>/<version>`), `/version`, `/healthz`, `/openapi/...` —
613
+ * so those stay rendered as plain `http`.
614
+ */
615
+ function describeKubeRequest(
616
+ method: string,
617
+ rawUrl: string,
618
+ ): KubeRequestMeta | null {
619
+ let path: string;
620
+ let query: URLSearchParams;
621
+ try {
622
+ const u = new URL(rawUrl);
623
+ path = u.pathname;
624
+ query = u.searchParams;
625
+ } catch {
626
+ const q = rawUrl.indexOf("?");
627
+ path = q === -1 ? rawUrl : rawUrl.slice(0, q);
628
+ query = new URLSearchParams(q === -1 ? "" : rawUrl.slice(q + 1));
629
+ }
630
+
631
+ const segs = path.split("/").filter((s) => s.length > 0);
632
+ if (segs.length === 0) return null;
633
+
634
+ let group: string | undefined;
635
+ let apiVersion: string | undefined;
636
+ let rest: string[];
637
+ if (segs[0] === "api") {
638
+ group = "";
639
+ apiVersion = segs[1];
640
+ rest = segs.slice(2);
641
+ } else if (segs[0] === "apis") {
642
+ group = segs[1];
643
+ apiVersion = segs[2];
644
+ rest = segs.slice(3);
645
+ } else {
646
+ return null; // /version, /healthz, /openapi, …
647
+ }
648
+ if (!apiVersion) return null; // discovery root (/api, /apis/<group>)
649
+
650
+ // `namespaces/<ns>/<resource>...` is namespaced; everything else
651
+ // (including `namespaces` and `namespaces/<name>` themselves, and
652
+ // cluster-scoped resources like `nodes`) is taken as-is.
653
+ let namespace: string | undefined;
654
+ let resourcePath = rest;
655
+ if (rest[0] === "namespaces" && rest.length >= 3) {
656
+ namespace = rest[1];
657
+ resourcePath = rest.slice(2);
658
+ }
659
+ if (resourcePath.length === 0) return null; // APIResourceList discovery
660
+
661
+ const resource = resourcePath[0];
662
+ const name = resourcePath.length >= 2 ? resourcePath[1] : undefined;
663
+ const subresource = resourcePath.length >= 3 ? resourcePath[2] : undefined;
664
+
665
+ const watchParam = query.get("watch");
666
+ const watch = watchParam === "true" || watchParam === "1";
667
+ const hasName = name !== undefined;
668
+ let verb: string;
669
+ switch (method.toUpperCase()) {
670
+ case "GET":
671
+ case "HEAD":
672
+ verb = hasName ? "get" : watch ? "watch" : "list";
673
+ break;
674
+ case "POST":
675
+ verb = "create";
676
+ break;
677
+ case "PUT":
678
+ verb = "update";
679
+ break;
680
+ case "PATCH":
681
+ verb = "patch";
682
+ break;
683
+ case "DELETE":
684
+ verb = hasName ? "delete" : "deletecollection";
685
+ break;
686
+ default:
687
+ verb = method.toLowerCase();
688
+ }
689
+
690
+ return { verb, group, apiVersion, resource, subresource, name, namespace };
691
+ }
692
+
693
+ /**
694
+ * True for Kubernetes API *discovery* paths — the version/group/resource
695
+ * enumeration endpoints (`/api`, `/api/<version>`, `/apis`, `/apis/<group>`,
696
+ * `/apis/<group>/<version>`) the dynamic client hits to resolve a kind to
697
+ * its resource path. They carry no resource operation (so
698
+ * `describeKubeRequest` returns null), and `doFetch` retracts their events
699
+ * from the timeline. Non-resource paths that are NOT discovery (`/healthz`,
700
+ * `/version`, `/openapi`, …) are deliberately not matched — they stay as
701
+ * `http`.
702
+ */
703
+ function isKubeDiscoveryPath(rawUrl: string): boolean {
704
+ let path: string;
705
+ try {
706
+ path = new URL(rawUrl).pathname;
707
+ } catch {
708
+ const q = rawUrl.indexOf("?");
709
+ path = q === -1 ? rawUrl : rawUrl.slice(0, q);
710
+ }
711
+ const segs = path.split("/").filter((s) => s.length > 0);
712
+ if (segs.length === 0) return false;
713
+ // `/api` + `/api/<version>`; `/apis` + `/apis/<group>` + `/apis/<group>/<version>`.
714
+ // Anything longer carries a resource segment and is handled as `kube`.
715
+ if (segs[0] === "api") return segs.length <= 2;
716
+ if (segs[0] === "apis") return segs.length <= 3;
717
+ return false;
718
+ }
719
+
720
+ async function doFetch(request: RequestContext): Promise<ResponseContext> {
721
+ const url = request.getUrl();
722
+ const method = String(request.getHttpMethod());
723
+ const body = request.getBody();
724
+ const reqHeaders: Record<string, string> = {};
725
+ for (const [k, v] of Object.entries(request.getHeaders())) {
726
+ reqHeaders[k] = String(v);
727
+ }
728
+
729
+ // The library's auth flow puts client cert/key on an https.Agent
730
+ // attached to the request. Pull them out so we can hand them to Bun's
731
+ // fetch via its `tls` option.
732
+ // eslint-disable-next-line @typescript-eslint/no-explicit-any
733
+ const agent = request.getAgent() as any;
734
+ const agentOpts = agent?.options ?? {};
735
+ const tlsOpts: Record<string, unknown> = { rejectUnauthorized: false };
736
+ if (agentOpts.cert) tlsOpts.cert = agentOpts.cert;
737
+ if (agentOpts.key) tlsOpts.key = agentOpts.key;
738
+
739
+ const wrapped = await fetch(url, {
740
+ method,
741
+ headers: reqHeaders,
742
+ body: body as BodyInit | undefined,
743
+ signal: request.getSignal(),
744
+ // Bun-specific TLS shape; under Node this option is ignored.
745
+ // eslint-disable-next-line @typescript-eslint/no-explicit-any
746
+ tls: tlsOpts,
747
+ } as RequestInit);
748
+
749
+ // Before unwrapping, capture the inspector tag the daemon's fetch
750
+ // wrapper installed on the Response. We feed the seq back through
751
+ // AsyncLocalStorage so the API method's eventual return value can
752
+ // re-acquire it — otherwise the chain `core.listNode() →
753
+ // expect(result)` would record assertions with no back-reference.
754
+ const tag = readTag(wrapped);
755
+ const slot = callContext.getStore();
756
+ if (slot && tag) slot.seq = tag.sourceSeq;
757
+
758
+ // Reclassify the `http` event the fetch wrapper just recorded into a
759
+ // Kubernetes-specific `kube` event (verb/resource/namespace/name), so
760
+ // the timeline reads `list pods · default` rather than the raw API URL.
761
+ // A `tag` is only present when recording was active for this call, so
762
+ // this is a no-op outside instrumented test runs.
763
+ if (tag && tag.sourceSeq !== undefined) {
764
+ const meta = describeKubeRequest(method, url);
765
+ if (meta) {
766
+ recorderAnnotate(tag.sourceSeq, { kind: "kube", ...meta });
767
+ } else if (isKubeDiscoveryPath(url)) {
768
+ // The dynamic client (`objects`, KubernetesObjectApi) can't know a
769
+ // kind's resource path ahead of time, so before the real request it
770
+ // GETs the group's resource list (`/apis/<group>/<version>` →
771
+ // APIResourceList) to map e.g. Ingress → `ingresses`/namespaced, then
772
+ // caches it (apiVersionResourceCache). That discovery GET is library
773
+ // plumbing the test author never wrote, and because the cache is
774
+ // per-daemon-process it surfaces non-deterministically across forks
775
+ // (first access pays it; `dependsOn` children inheriting the warm
776
+ // cache don't). Retract it rather than leave a bare `http` row — the
777
+ // real list/read that follows is recorded and reclassified as usual.
778
+ recorderRemove(tag.sourceSeq);
779
+ }
780
+ // Other non-resource paths (/healthz, /version, /openapi, …) fall
781
+ // through and stay rendered as plain `http`.
782
+ }
783
+
784
+ // The daemon's fetch wrapper proxies `Response.status` and similar
785
+ // primitives as carrier objects (so test assertions can fold under
786
+ // the originating HTTP event). The kubernetes/client-node lib calls
787
+ // `httpStatusCode.toString()` which would then return
788
+ // "[object Object]" and the status-code dispatch falls through to
789
+ // "Unknown API Status Code!". Pull out the raw Response.
790
+ const response =
791
+ (wrapped as { unwrap?: () => Response }).unwrap?.() ?? wrapped;
792
+
793
+ const resHeaders: Record<string, string> = {};
794
+ response.headers.forEach((v, k) => {
795
+ resHeaders[k] = v;
796
+ });
797
+ const buf = Buffer.from(await response.arrayBuffer());
798
+ return new ResponseContext(response.status, resHeaders, {
799
+ text: async () => buf.toString("utf8"),
800
+ binary: async () => buf,
801
+ });
802
+ }
803
+
804
+ /**
805
+ * Wrap a `@kubernetes/client-node` Api instance so each method call
806
+ * runs in its own AsyncLocalStorage slot — `doFetch` writes the HTTP
807
+ * event's `sourceSeq` into the slot, and after the lib parses the
808
+ * response we re-attach the seq to the returned object. Downstream
809
+ * `expect(result.items[0].status…)` assertions then fold under that
810
+ * HTTP event in the test event log, the same way `expect(res.status)`
811
+ * does for plain `fetch` calls.
812
+ *
813
+ * Method arguments are deep-unwrapped on the way in so values pulled
814
+ * from a previous API response (still carrying the inspector wrappers)
815
+ * can be passed straight back into another call.
816
+ *
817
+ * Non-function properties pass through untagged.
818
+ */
819
+ function withTagging<T extends object>(api: T): Tagged<T> {
820
+ return new Proxy(api, {
821
+ get(target, prop, receiver) {
822
+ const value = Reflect.get(target, prop, receiver);
823
+ if (typeof value !== "function") return value;
824
+ // Bind the original method to `target` so the lib's internal
825
+ // `this.configuration` accesses keep working.
826
+ // eslint-disable-next-line @typescript-eslint/no-explicit-any
827
+ const method = (value as any).bind(target);
828
+ return (...args: unknown[]): unknown => {
829
+ const unwrappedArgs = args.map(deepUnwrap);
830
+ const slot: CallSlot = {};
831
+ const result = callContext.run(slot, () =>
832
+ method(...unwrappedArgs),
833
+ );
834
+ // Wrap unconditionally — `slot.seq` is undefined when no event was
835
+ // recorded (setup/eval, no active recorder), but the result's type is
836
+ // wrapped, so the value must be wrapped at runtime too (just without a
837
+ // provenance link). Keeps `.unwrap()` available in every context.
838
+ if (result && typeof (result as Promise<unknown>).then === "function") {
839
+ return (result as Promise<unknown>).then((v) => wrap(v, slot.seq));
840
+ }
841
+ return wrap(result, slot.seq);
842
+ };
843
+ },
844
+ }) as unknown as Tagged<T>;
845
+ }
846
+
847
+ /**
848
+ * Image tag for the Traefik we install. Pulled on first cluster boot
849
+ * through the host `zot` mirror (`docker.io` → cache) configured in
850
+ * `registries.yaml`, then captured into the warm-template snapshot so
851
+ * warm starts never pull.
852
+ */
853
+ const TRAEFIK_IMAGE = "rancher/mirrored-library-traefik:3.3.2";
854
+
855
+ /**
856
+ * Names of the in-cluster resources that carry the CA-signed default
857
+ * ingress certificate (created in `setupK3sCluster` when TLS is enabled).
858
+ * The Secret holds the leaf cert+key; the ConfigMap holds the Traefik
859
+ * file-provider snippet that points the `default` TLS store at it.
860
+ */
861
+ const TRAEFIK_TLS_SECRET = "traefik-default-tls";
862
+ const TRAEFIK_DYNAMIC_CONFIGMAP = "traefik-dynamic";
863
+
864
+ /**
865
+ * Traefik file-provider dynamic config: make the in-VM-CA leaf the
866
+ * `default` store certificate, so every router on the `websecure`
867
+ * entrypoint (which we force to TLS) serves it with no per-Ingress
868
+ * `spec.tls` needed.
869
+ */
870
+ const TRAEFIK_DYNAMIC_TLS = `tls:
871
+ stores:
872
+ default:
873
+ defaultCertificate:
874
+ certFile: /certs/tls.crt
875
+ keyFile: /certs/tls.key
876
+ `;
877
+
878
+ /**
879
+ * Traefik manifest applied during setup(). `hostNetwork: true` puts
880
+ * Traefik in the k3s container's netns, so it binds the container's :80
881
+ * (and, with `tls`, :443) directly — no CNI portmap involved (that path
882
+ * still trips on the kernel's missing xt_comment match).
883
+ *
884
+ * When `tls` is set we add a `websecure` :443 entrypoint with TLS forced
885
+ * on (served from the `default` store, i.e. the in-VM-CA leaf mounted
886
+ * from the `traefik-default-tls` Secret via the file provider). HTTPS
887
+ * then works for any routed host under the cluster's `ingressDomains`
888
+ * with zero per-Ingress config; the :80 `web` entrypoint is unchanged.
889
+ */
890
+ function buildTraefikManifest(tls: boolean): string {
891
+ const args = [
892
+ " - --entrypoints.web.address=:80",
893
+ ...(tls
894
+ ? [
895
+ " - --entrypoints.websecure.address=:443",
896
+ " - --entrypoints.websecure.http.tls=true",
897
+ ]
898
+ : []),
899
+ " - --providers.kubernetesingress=true",
900
+ " - --providers.kubernetesingress.ingressclass=traefik",
901
+ ...(tls
902
+ ? [
903
+ " - --providers.file.directory=/dynamic",
904
+ " - --providers.file.watch=true",
905
+ ]
906
+ : []),
907
+ " - --log.level=INFO",
908
+ ].join("\n");
909
+ const ports = [
910
+ " - name: web",
911
+ " containerPort: 80",
912
+ ...(tls
913
+ ? [" - name: websecure", " containerPort: 443"]
914
+ : []),
915
+ ].join("\n");
916
+ const volumeMounts = tls
917
+ ? `
918
+ volumeMounts:
919
+ - name: default-cert
920
+ mountPath: /certs
921
+ readOnly: true
922
+ - name: dynamic
923
+ mountPath: /dynamic
924
+ readOnly: true`
925
+ : "";
926
+ const volumes = tls
927
+ ? `
928
+ volumes:
929
+ - name: default-cert
930
+ secret:
931
+ secretName: ${TRAEFIK_TLS_SECRET}
932
+ - name: dynamic
933
+ configMap:
934
+ name: ${TRAEFIK_DYNAMIC_CONFIGMAP}`
935
+ : "";
936
+ return `apiVersion: v1
937
+ kind: ServiceAccount
938
+ metadata:
939
+ name: traefik
940
+ namespace: kube-system
941
+ ---
942
+ apiVersion: rbac.authorization.k8s.io/v1
943
+ kind: ClusterRole
944
+ metadata:
945
+ name: traefik
946
+ rules:
947
+ - apiGroups: [""]
948
+ resources: ["services", "endpoints", "secrets", "nodes"]
949
+ verbs: ["get", "list", "watch"]
950
+ - apiGroups: ["discovery.k8s.io"]
951
+ resources: ["endpointslices"]
952
+ verbs: ["get", "list", "watch"]
953
+ - apiGroups: ["networking.k8s.io"]
954
+ resources: ["ingresses", "ingressclasses"]
955
+ verbs: ["get", "list", "watch"]
956
+ - apiGroups: ["networking.k8s.io"]
957
+ resources: ["ingresses/status"]
958
+ verbs: ["update"]
959
+ ---
960
+ apiVersion: rbac.authorization.k8s.io/v1
961
+ kind: ClusterRoleBinding
962
+ metadata:
963
+ name: traefik
964
+ roleRef:
965
+ apiGroup: rbac.authorization.k8s.io
966
+ kind: ClusterRole
967
+ name: traefik
968
+ subjects:
969
+ - kind: ServiceAccount
970
+ name: traefik
971
+ namespace: kube-system
972
+ ---
973
+ apiVersion: networking.k8s.io/v1
974
+ kind: IngressClass
975
+ metadata:
976
+ name: traefik
977
+ annotations:
978
+ ingressclass.kubernetes.io/is-default-class: "true"
979
+ spec:
980
+ controller: traefik.io/ingress-controller
981
+ ---
982
+ apiVersion: apps/v1
983
+ kind: Deployment
984
+ metadata:
985
+ name: traefik
986
+ namespace: kube-system
987
+ labels:
988
+ app: traefik
989
+ spec:
990
+ replicas: 1
991
+ selector:
992
+ matchLabels:
993
+ app: traefik
994
+ template:
995
+ metadata:
996
+ labels:
997
+ app: traefik
998
+ spec:
999
+ serviceAccountName: traefik
1000
+ hostNetwork: true
1001
+ dnsPolicy: Default
1002
+ tolerations:
1003
+ - operator: Exists
1004
+ containers:
1005
+ - name: traefik
1006
+ image: ${TRAEFIK_IMAGE}
1007
+ imagePullPolicy: IfNotPresent
1008
+ args:
1009
+ ${args}
1010
+ ports:
1011
+ ${ports}${volumeMounts}${volumes}
1012
+ `;
1013
+ }
1014
+
1015
+ /**
1016
+ * Host-side `zot` pull-through cache layout (local Firecracker provider
1017
+ * only). One zot instance per upstream registry, all bound to the
1018
+ * `spectest-br0` gateway `10.42.0.1` on the ports below — **kept in sync
1019
+ * with `scripts/install-zot.sh`**. We mirror the cluster's containerd
1020
+ * through these so every image pull reuses the shared host cache instead
1021
+ * of hitting the public registry, and we list the canonical upstream as
1022
+ * a fallback endpoint so a missing/cold mirror only ever slows a pull,
1023
+ * never breaks it.
1024
+ */
1025
+ const ZOT_MIRRORS: Array<{ registry: string; port: number; upstream: string }> = [
1026
+ { registry: "docker.io", port: 5000, upstream: "https://registry-1.docker.io" },
1027
+ { registry: "ghcr.io", port: 5001, upstream: "https://ghcr.io" },
1028
+ { registry: "quay.io", port: 5002, upstream: "https://quay.io" },
1029
+ { registry: "registry.k8s.io", port: 5003, upstream: "https://registry.k8s.io" },
1030
+ { registry: "public.ecr.aws", port: 5004, upstream: "https://public.ecr.aws" },
1031
+ { registry: "gcr.io", port: 5005, upstream: "https://gcr.io" },
1032
+ { registry: "mcr.microsoft.com", port: 5006, upstream: "https://mcr.microsoft.com" },
1033
+ ];
1034
+
1035
+ /**
1036
+ * Discover the host-side image cache gateway by reading the same
1037
+ * `registry-mirrors` entry the in-VM dockerd already uses (baked into
1038
+ * the golden rootfs's `/etc/docker/daemon.json`). Returns the
1039
+ * gateway host (`"10.42.0.1"`) when present, or `null` when there's no
1040
+ * host cache, in which case the cluster pulls every image direct. Runs inside the daemon (VM) at `index.ts` load time, so
1041
+ * the result is stable per host and never poisons the warm-template
1042
+ * cache.
1043
+ */
1044
+ function detectHostMirrorGateway(): string | null {
1045
+ try {
1046
+ const cfg = JSON.parse(
1047
+ readFileSync("/etc/docker/daemon.json", "utf8"),
1048
+ ) as { "registry-mirrors"?: string[] };
1049
+ const first = cfg["registry-mirrors"]?.[0];
1050
+ return first ? new URL(first).hostname || null : null;
1051
+ } catch {
1052
+ return null;
1053
+ }
1054
+ }
1055
+
1056
+ /**
1057
+ * Build `/etc/rancher/k3s/registries.yaml`. k3s reads this **once, at
1058
+ * startup**, to configure its embedded containerd — which is why it has
1059
+ * to be seeded via `files` (a pre-start bind mount) rather than a
1060
+ * `setup` hook. Two jobs:
1061
+ * 1. Mirror the cluster's image pulls through the host `zot` cache
1062
+ * (omitted when there's no host cache).
1063
+ * 2. Trust the in-cluster registry, addressed as `<key>.internal:5000`
1064
+ * (the `{{SPECTEST_SERVICE}}` token is expanded to the cluster's
1065
+ * service key when the file is written). Image *references* use
1066
+ * that peer-reachable name, but containerd pulls via the loopback
1067
+ * endpoint `http://127.0.0.1:5000` — the hostNetwork registry pod
1068
+ * shares the node's netns, so this needs no in-container DNS and
1069
+ * can't be broken by a clobbered `hostnames`.
1070
+ * Returns `null` when there's nothing to configure (no host cache and
1071
+ * `registry` disabled), in which case no file is injected.
1072
+ */
1073
+ function buildRegistriesYaml(registryEnabled: boolean): string | null {
1074
+ const gateway = detectHostMirrorGateway();
1075
+ if (!gateway && !registryEnabled) return null;
1076
+
1077
+ const lines: string[] = ["mirrors:"];
1078
+ if (gateway) {
1079
+ for (const { registry, port, upstream } of ZOT_MIRRORS) {
1080
+ lines.push(
1081
+ ` "${registry}":`,
1082
+ ` endpoint:`,
1083
+ ` - "http://${gateway}:${port}"`,
1084
+ ` - "${upstream}"`,
1085
+ );
1086
+ }
1087
+ }
1088
+ if (registryEnabled) {
1089
+ const host = `{{SPECTEST_SERVICE}}.internal:${K3S_REGISTRY_PORT}`;
1090
+ lines.push(
1091
+ ` "${host}":`,
1092
+ ` endpoint:`,
1093
+ ` - "http://127.0.0.1:${K3S_REGISTRY_PORT}"`,
1094
+ "configs:",
1095
+ // The endpoint is plain HTTP; the config (keyed by endpoint host)
1096
+ // makes that explicit and disables any TLS attempt against it.
1097
+ ` "127.0.0.1:${K3S_REGISTRY_PORT}":`,
1098
+ ` tls:`,
1099
+ ` insecure_skip_verify: true`,
1100
+ );
1101
+ }
1102
+ return lines.join("\n") + "\n";
1103
+ }
1104
+
1105
+ /**
1106
+ * In-cluster OCI registry (CNCF `distribution`). `hostNetwork: true`
1107
+ * binds the cluster container's `:5000` directly — the same trick
1108
+ * Traefik uses — so peer services reach it at `<cluster-key>.internal:5000`
1109
+ * (the cluster service's own alias) and the node's own containerd reaches
1110
+ * it at `127.0.0.1:5000`. Storage is an `emptyDir`, so pushed images live
1111
+ * in the cluster and are captured by snapshot / isolated per test fork
1112
+ * like all other in-VM state.
1113
+ */
1114
+ const REGISTRY_MANIFEST = `apiVersion: apps/v1
1115
+ kind: Deployment
1116
+ metadata:
1117
+ name: spectest-registry
1118
+ namespace: kube-system
1119
+ labels:
1120
+ app: spectest-registry
1121
+ spec:
1122
+ replicas: 1
1123
+ selector:
1124
+ matchLabels:
1125
+ app: spectest-registry
1126
+ template:
1127
+ metadata:
1128
+ labels:
1129
+ app: spectest-registry
1130
+ spec:
1131
+ hostNetwork: true
1132
+ dnsPolicy: Default
1133
+ tolerations:
1134
+ - operator: Exists
1135
+ containers:
1136
+ - name: registry
1137
+ image: registry:2
1138
+ imagePullPolicy: IfNotPresent
1139
+ env:
1140
+ - name: REGISTRY_HTTP_ADDR
1141
+ value: ":${K3S_REGISTRY_PORT}"
1142
+ - name: REGISTRY_STORAGE_DELETE_ENABLED
1143
+ value: "true"
1144
+ ports:
1145
+ - name: registry
1146
+ containerPort: ${K3S_REGISTRY_PORT}
1147
+ volumeMounts:
1148
+ - name: data
1149
+ mountPath: /var/lib/registry
1150
+ volumes:
1151
+ - name: data
1152
+ emptyDir: {}
1153
+ `;
1154
+
1155
+ // ────────────────────────────────────────────────────────────────────────
1156
+ // Apply — server-side, converging
1157
+ //
1158
+ // `kubectl apply --server-side --force-conflicts`, over the API. Two
1159
+ // properties make a real manifest set work where plain `create` didn't:
1160
+ //
1161
+ // * create-or-update. SSA PATCHes with `application/apply-patch+yaml`,
1162
+ // so re-applying an edited manifest converges instead of failing
1163
+ // `AlreadyExists` — which is what a `dependsOn` child re-applying over
1164
+ // its parent's state, or a second apply of a shared base, needs.
1165
+ // * ordering by retry, not by sorting. A CR whose CRD is in the same
1166
+ // manifest, or a resource an admission webhook (cert-manager, CNPG)
1167
+ // rejects while its own pod is still starting, fails the first round
1168
+ // and lands on a later one. Only errors that read as "not yet" are
1169
+ // retried; a genuinely malformed object throws immediately.
1170
+ // ────────────────────────────────────────────────────────────────────────
1171
+
1172
+ /** Default SSA field manager. Stable across runs on purpose: server-side
1173
+ * apply keys ownership on it, so a changing value would make every run
1174
+ * conflict with the previous one's fields. */
1175
+ const APPLY_FIELD_MANAGER = "spectest";
1176
+ const APPLY_TIMEOUT_MS = 120_000;
1177
+ const WAIT_TIMEOUT_MS = 120_000;
1178
+ const WAIT_INTERVAL_MS = 500;
1179
+
1180
+ /**
1181
+ * API-server responses that mean "this document can't land *yet*" — the
1182
+ * CRD isn't established, an admission webhook's own pod isn't serving,
1183
+ * the namespace is still being created, or the write raced another
1184
+ * writer. Everything else (a schema violation, a bad field, RBAC) is a
1185
+ * real failure and is thrown on the first round.
1186
+ */
1187
+ const TRANSIENT_APPLY_PATTERNS: readonly RegExp[] = [
1188
+ /no matches for kind/i,
1189
+ // The client's own message when it can't resolve a kind to a resource
1190
+ // path — i.e. the CRD that defines it isn't established yet. This is
1191
+ // the "CR next to its CRD in one manifest" case, so it MUST be retried.
1192
+ /Failed to fetch resource metadata/i,
1193
+ /could not find the requested resource/i,
1194
+ /the server (?:could not find|does not recognize)/i,
1195
+ /failed calling webhook/i,
1196
+ /webhook .* denied the request: .*(?:not ready|unavailable|connection refused)/i,
1197
+ /connection refused/i,
1198
+ /no endpoints available/i,
1199
+ /context deadline exceeded/i,
1200
+ /etcdserver: (?:leader changed|request timed out)/i,
1201
+ /the object has been modified/i,
1202
+ /namespaces? "[^"]+" not found/i,
1203
+ /Internal error occurred/i,
1204
+ /(?:^|\n)HTTP-Code: (?:429|500|503|504)/,
1205
+ ];
1206
+
1207
+ function errorText(err: unknown): string {
1208
+ return (err as Error)?.message ?? String(err);
1209
+ }
1210
+
1211
+ function isTransientApplyError(err: unknown): boolean {
1212
+ const msg = errorText(err);
1213
+ return TRANSIENT_APPLY_PATTERNS.some((re) => re.test(msg));
1214
+ }
1215
+
1216
+ function describeDoc(doc: KubernetesObject): string {
1217
+ const ns = doc.metadata?.namespace;
1218
+ const name = doc.metadata?.name ?? doc.metadata?.generateName ?? "?";
1219
+ return `${doc.kind ?? "?"}/${name}${ns ? ` (ns ${ns})` : ""}`;
1220
+ }
1221
+
1222
+ /** CRDs and Namespaces go first so the common "CR next to its CRD in one
1223
+ * file" case lands in round one rather than costing a retry. Everything
1224
+ * else keeps input order (results are always returned in input order). */
1225
+ function applyRank(doc: KubernetesObject): number {
1226
+ return doc.kind === "CustomResourceDefinition" || doc.kind === "Namespace"
1227
+ ? 0
1228
+ : 1;
1229
+ }
1230
+
1231
+ /**
1232
+ * `KubernetesObjectApi` with its resource-discovery lookup exposed, so
1233
+ * `apply` can tell a namespaced kind from a cluster-scoped one before
1234
+ * defaulting `metadata.namespace` (setting it on a cluster-scoped object
1235
+ * puts a namespace in the body the API server then rejects).
1236
+ */
1237
+ class ResourceAwareObjectApi extends KubernetesObjectApi {
1238
+ async resourceInfo(
1239
+ apiVersion: string,
1240
+ kind: string,
1241
+ ): Promise<V1APIResource | undefined> {
1242
+ try {
1243
+ return await this.resource(apiVersion, kind);
1244
+ } catch {
1245
+ // Discovery failure (the CRD isn't established yet) — the caller
1246
+ // treats "unknown" as "leave the document alone"; the apply that
1247
+ // follows fails transiently and is retried once discovery works.
1248
+ return undefined;
1249
+ }
1250
+ }
1251
+ }
1252
+
1253
+ /** What the apply/wait implementations need from the enclosing cluster. */
1254
+ interface ClusterDeps {
1255
+ clusterName: string;
1256
+ client: K3sClient;
1257
+ objects: TaggedObjectApi;
1258
+ meta: ResourceAwareObjectApi;
1259
+ poll: SpectestContext["poll"];
1260
+ }
1261
+
1262
+ const sleep = (ms: number): Promise<void> =>
1263
+ new Promise((r) => setTimeout(r, ms));
1264
+
1265
+ /** Default a document's namespace, but only for kinds that have one. */
1266
+ async function withDefaultNamespace(
1267
+ deps: ClusterDeps,
1268
+ doc: KubernetesObject,
1269
+ namespace: string | undefined,
1270
+ ): Promise<KubernetesObject> {
1271
+ if (!namespace || doc.metadata?.namespace) return doc;
1272
+ const info = await deps.meta.resourceInfo(doc.apiVersion ?? "v1", doc.kind!);
1273
+ // Unknown kind (CRD not established yet) or cluster-scoped: don't touch it.
1274
+ if (!info?.namespaced) return doc;
1275
+ return { ...doc, metadata: { ...doc.metadata, namespace } };
1276
+ }
1277
+
1278
+ async function applyOne(
1279
+ deps: ClusterDeps,
1280
+ doc: KubernetesObject,
1281
+ opts: { namespace?: string; fieldManager: string; force: boolean },
1282
+ ): Promise<Wrapped<KubernetesObject>> {
1283
+ const spec = await withDefaultNamespace(deps, doc, opts.namespace);
1284
+ // Server-side apply addresses the object by name; a `generateName`-only
1285
+ // document has none, so it can only ever be created.
1286
+ if (!spec.metadata?.name) {
1287
+ return deps.objects.create(spec, undefined, undefined, opts.fieldManager);
1288
+ }
1289
+ return deps.objects.patch(
1290
+ spec,
1291
+ undefined,
1292
+ undefined,
1293
+ opts.fieldManager,
1294
+ opts.force,
1295
+ PatchStrategy.ServerSideApply,
1296
+ );
1297
+ }
1298
+
1299
+ async function applyManifest(
1300
+ deps: ClusterDeps,
1301
+ manifest: string,
1302
+ opts?: ApplyOptions,
1303
+ ): Promise<Wrapped<KubernetesObject>[]> {
1304
+ const docs = (loadAllYaml(manifest) as KubernetesObject[]).filter(
1305
+ (d): d is KubernetesObject =>
1306
+ !!d && typeof d === "object" && typeof d.kind === "string",
1307
+ );
1308
+ const fieldManager = opts?.fieldManager ?? APPLY_FIELD_MANAGER;
1309
+ const force = opts?.force ?? true;
1310
+ const timeoutMs = opts?.timeoutMs ?? APPLY_TIMEOUT_MS;
1311
+ const deadline = Date.now() + timeoutMs;
1312
+ const out: Wrapped<KubernetesObject>[] = new Array(docs.length);
1313
+
1314
+ let pending = docs
1315
+ .map((doc, index) => ({ doc, index }))
1316
+ .sort((a, b) => applyRank(a.doc) - applyRank(b.doc) || a.index - b.index);
1317
+ let backoff = 0;
1318
+ let stalled: string[] = [];
1319
+
1320
+ while (pending.length > 0) {
1321
+ const failed: typeof pending = [];
1322
+ stalled = [];
1323
+ for (const item of pending) {
1324
+ try {
1325
+ out[item.index] = await applyOne(deps, item.doc, {
1326
+ namespace: opts?.namespace,
1327
+ fieldManager,
1328
+ force,
1329
+ });
1330
+ } catch (err) {
1331
+ if (!isTransientApplyError(err)) {
1332
+ throw new Error(
1333
+ `k3s(${deps.clusterName}): applying ${describeDoc(item.doc)} failed: ${errorText(err)}`,
1334
+ );
1335
+ }
1336
+ failed.push(item);
1337
+ stalled.push(`${describeDoc(item.doc)}: ${errorText(err).replace(/\s+/g, " ").slice(0, 300)}`);
1338
+ }
1339
+ }
1340
+ if (failed.length === 0) break;
1341
+ if (Date.now() >= deadline) {
1342
+ throw new Error(
1343
+ `k3s(${deps.clusterName}): apply did not converge within ${timeoutMs / 1000}s — ` +
1344
+ `${failed.length} of ${docs.length} document(s) still failing:\n ${stalled.join("\n ")}`,
1345
+ );
1346
+ }
1347
+ // Nothing landed this round: whatever is missing (a webhook pod, a
1348
+ // CRD's establishment) needs wall-clock, so back off before retrying.
1349
+ // Progress resets the backoff — a long CRD chain shouldn't decay.
1350
+ if (failed.length === pending.length) {
1351
+ await sleep(Math.min(2_000, 250 * 2 ** backoff));
1352
+ backoff += 1;
1353
+ } else {
1354
+ backoff = 0;
1355
+ }
1356
+ pending = failed;
1357
+ }
1358
+ return out;
1359
+ }
1360
+
1361
+ // ────────────────────────────────────────────────────────────────────────
1362
+ // Readiness — "is it actually serving?"
1363
+ // ────────────────────────────────────────────────────────────────────────
1364
+
1365
+ /** A workload reference resolved from a {@link RolloutTarget}. */
1366
+ interface WorkloadRef {
1367
+ kind: "Deployment" | "StatefulSet" | "DaemonSet";
1368
+ name: string;
1369
+ namespace: string;
1370
+ }
1371
+
1372
+ const ROLLOUT_KINDS: Record<string, WorkloadRef["kind"]> = {
1373
+ deployment: "Deployment",
1374
+ deployments: "Deployment",
1375
+ deploy: "Deployment",
1376
+ statefulset: "StatefulSet",
1377
+ statefulsets: "StatefulSet",
1378
+ sts: "StatefulSet",
1379
+ daemonset: "DaemonSet",
1380
+ daemonsets: "DaemonSet",
1381
+ ds: "DaemonSet",
1382
+ };
1383
+
1384
+ function parseRolloutTarget(
1385
+ target: RolloutTarget,
1386
+ namespace: string | undefined,
1387
+ ): WorkloadRef {
1388
+ if (typeof target === "string") {
1389
+ const slash = target.indexOf("/");
1390
+ const kindPart = slash === -1 ? "deployment" : target.slice(0, slash);
1391
+ const name = slash === -1 ? target : target.slice(slash + 1);
1392
+ const kind = ROLLOUT_KINDS[kindPart.toLowerCase()];
1393
+ if (!kind) {
1394
+ throw new Error(
1395
+ `waitForRollout("${target}"): unknown workload kind "${kindPart}" — ` +
1396
+ `use deployment, statefulset or daemonset (a bare name is a Deployment).`,
1397
+ );
1398
+ }
1399
+ if (!name) throw new Error(`waitForRollout("${target}"): missing name.`);
1400
+ return { kind, name, namespace: namespace ?? "default" };
1401
+ }
1402
+ const kind = ROLLOUT_KINDS[String(target.kind ?? "").toLowerCase()];
1403
+ const name = target.metadata?.name;
1404
+ if (!kind || !name) {
1405
+ throw new Error(
1406
+ `waitForRollout(): expected a Deployment/StatefulSet/DaemonSet object with metadata.name, got ` +
1407
+ `${JSON.stringify({ kind: target.kind, name })}.`,
1408
+ );
1409
+ }
1410
+ return {
1411
+ kind,
1412
+ name,
1413
+ namespace: namespace ?? target.metadata?.namespace ?? "default",
1414
+ };
1415
+ }
1416
+
1417
+ /** Rollout state of one workload: settled, or why not. */
1418
+ async function rolloutStatus(
1419
+ deps: ClusterDeps,
1420
+ ref: WorkloadRef,
1421
+ ): Promise<{ done: boolean; reason: string }> {
1422
+ const { name, namespace } = ref;
1423
+ try {
1424
+ if (ref.kind === "Deployment") {
1425
+ const d = (
1426
+ await deps.client.apps.readNamespacedDeployment({ name, namespace })
1427
+ ).unwrap();
1428
+ const want = d.spec?.replicas ?? 1;
1429
+ const st: Partial<V1DeploymentStatus> = d.status ?? {};
1430
+ if ((st.observedGeneration ?? 0) < (d.metadata?.generation ?? 0)) {
1431
+ return { done: false, reason: "controller has not observed the latest update yet" };
1432
+ }
1433
+ if (want === 0) {
1434
+ return {
1435
+ done: (st.replicas ?? 0) === 0,
1436
+ reason: `scaling down: ${st.replicas ?? 0} replica(s) remaining`,
1437
+ };
1438
+ }
1439
+ const updated = st.updatedReplicas ?? 0;
1440
+ const available = st.availableReplicas ?? 0;
1441
+ const total = st.replicas ?? 0;
1442
+ if (updated < want) {
1443
+ return { done: false, reason: `${updated}/${want} replica(s) updated` };
1444
+ }
1445
+ if (total > updated) {
1446
+ return { done: false, reason: `${total - updated} old replica(s) still terminating` };
1447
+ }
1448
+ return {
1449
+ done: available >= want,
1450
+ reason: `${available}/${want} replica(s) available`,
1451
+ };
1452
+ }
1453
+ if (ref.kind === "StatefulSet") {
1454
+ const s = (
1455
+ await deps.client.apps.readNamespacedStatefulSet({ name, namespace })
1456
+ ).unwrap();
1457
+ const want = s.spec?.replicas ?? 1;
1458
+ const st: Partial<V1StatefulSetStatus> = s.status ?? {};
1459
+ if ((st.observedGeneration ?? 0) < (s.metadata?.generation ?? 0)) {
1460
+ return { done: false, reason: "controller has not observed the latest update yet" };
1461
+ }
1462
+ const ready = st.readyReplicas ?? 0;
1463
+ const updated = st.updatedReplicas ?? 0;
1464
+ if (st.updateRevision && st.currentRevision !== st.updateRevision) {
1465
+ return {
1466
+ done: false,
1467
+ reason: `rolling update in progress (${updated}/${want} pod(s) updated)`,
1468
+ };
1469
+ }
1470
+ return { done: ready >= want, reason: `${ready}/${want} pod(s) ready` };
1471
+ }
1472
+ const ds = (
1473
+ await deps.client.apps.readNamespacedDaemonSet({ name, namespace })
1474
+ ).unwrap();
1475
+ const st: Partial<V1DaemonSetStatus> = ds.status ?? {};
1476
+ if ((st.observedGeneration ?? 0) < (ds.metadata?.generation ?? 0)) {
1477
+ return { done: false, reason: "controller has not observed the latest update yet" };
1478
+ }
1479
+ const want = st.desiredNumberScheduled ?? 0;
1480
+ const ready = st.numberReady ?? 0;
1481
+ const updated = st.updatedNumberScheduled ?? 0;
1482
+ if (updated < want) {
1483
+ return { done: false, reason: `${updated}/${want} node(s) updated` };
1484
+ }
1485
+ return { done: want > 0 && ready >= want, reason: `${ready}/${want} node(s) ready` };
1486
+ } catch (err) {
1487
+ const msg = errorText(err);
1488
+ return {
1489
+ done: false,
1490
+ reason: /not found|HTTP-Code: 404/i.test(msg)
1491
+ ? `${ref.kind} ${name} does not exist (yet)`
1492
+ : msg.replace(/\s+/g, " ").slice(0, 300),
1493
+ };
1494
+ }
1495
+ }
1496
+
1497
+ async function waitForRollout(
1498
+ deps: ClusterDeps,
1499
+ target: RolloutTarget,
1500
+ opts?: WaitOptions,
1501
+ ): Promise<void> {
1502
+ const ref = parseRolloutTarget(target, opts?.namespace);
1503
+ const timeoutMs = opts?.timeoutMs ?? WAIT_TIMEOUT_MS;
1504
+ let last = "";
1505
+ try {
1506
+ await deps.poll(
1507
+ `${ref.kind.toLowerCase()}/${ref.name} rolled out`,
1508
+ async () => {
1509
+ const st = await rolloutStatus(deps, ref);
1510
+ last = st.reason;
1511
+ return st.done || null;
1512
+ },
1513
+ { timeoutMs, intervalMs: opts?.intervalMs ?? WAIT_INTERVAL_MS },
1514
+ );
1515
+ } catch (err) {
1516
+ const diag = await collectDiagnostics(deps.client, ref.namespace);
1517
+ throw new Error(
1518
+ `k3s(${deps.clusterName}): ${ref.kind}/${ref.name} did not roll out within ` +
1519
+ `${timeoutMs / 1000}s${last ? ` — ${last}` : ""}.\n${diag}\n(${errorText(err)})`,
1520
+ );
1521
+ }
1522
+ }
1523
+
1524
+ /** Pod names in a namespace, optionally filtered by label selector. */
1525
+ async function podNames(
1526
+ client: K3sClient,
1527
+ namespace: string,
1528
+ labelSelector?: string,
1529
+ ): Promise<string[]> {
1530
+ const pods = (
1531
+ await client.core.listNamespacedPod({
1532
+ namespace,
1533
+ ...(labelSelector ? { labelSelector } : {}),
1534
+ })
1535
+ ).unwrap();
1536
+ return (pods.items ?? [])
1537
+ .map((p) => p.metadata?.name)
1538
+ .filter((n): n is string => !!n);
1539
+ }
1540
+
1541
+ async function readLogs(
1542
+ client: K3sClient,
1543
+ opts: LogsOptions,
1544
+ ): Promise<string> {
1545
+ const namespace = opts.namespace ?? "default";
1546
+ const tailLines = opts.tailLines ?? 200;
1547
+ if (!opts.pod && !opts.selector) {
1548
+ throw new Error("ctx.svc.<k3s>.logs(): pass either `pod` or `selector`.");
1549
+ }
1550
+ const names = opts.pod
1551
+ ? [opts.pod]
1552
+ : await podNames(client, namespace, opts.selector);
1553
+ const chunks: string[] = [];
1554
+ for (const name of names) {
1555
+ try {
1556
+ const text = (
1557
+ await client.core.readNamespacedPodLog({
1558
+ name,
1559
+ namespace,
1560
+ tailLines,
1561
+ ...(opts.container ? { container: opts.container } : {}),
1562
+ })
1563
+ ).unwrap();
1564
+ chunks.push(names.length > 1 ? `==> ${name} <==\n${text}` : String(text));
1565
+ } catch (err) {
1566
+ chunks.push(`==> ${name} <==\n(reading logs failed: ${errorText(err)})`);
1567
+ }
1568
+ }
1569
+ return chunks.join("\n");
1570
+ }
1571
+
1572
+ async function waitForJob(
1573
+ deps: ClusterDeps,
1574
+ name: string,
1575
+ opts?: WaitOptions,
1576
+ ): Promise<Wrapped<V1Job>> {
1577
+ const namespace = opts?.namespace ?? "default";
1578
+ const timeoutMs = opts?.timeoutMs ?? WAIT_TIMEOUT_MS;
1579
+ let last = "";
1580
+ try {
1581
+ // The predicate returns the RAW job; `poll` wraps the winning value, so
1582
+ // the returned object carries provenance to the wait step (double-
1583
+ // wrapping a proxy would not).
1584
+ return (await deps.poll(
1585
+ `job/${name} completed`,
1586
+ async () => {
1587
+ let job: V1Job;
1588
+ try {
1589
+ job = (
1590
+ await deps.client.batch.readNamespacedJob({ name, namespace })
1591
+ ).unwrap();
1592
+ } catch (err) {
1593
+ const msg = errorText(err);
1594
+ if (!/not found|HTTP-Code: 404/i.test(msg)) throw err;
1595
+ last = `Job ${name} does not exist (yet)`;
1596
+ return null;
1597
+ }
1598
+ const conditions = job.status?.conditions ?? [];
1599
+ const failure = conditions.find(
1600
+ (c) => c.type === "Failed" && c.status === "True",
1601
+ );
1602
+ if (failure) {
1603
+ const logs = await readLogs(deps.client, {
1604
+ namespace,
1605
+ selector: `job-name=${name}`,
1606
+ }).catch((e) => `(reading pod logs failed: ${errorText(e)})`);
1607
+ throw new Error(
1608
+ `k3s(${deps.clusterName}): Job ${name} failed` +
1609
+ `${failure.reason ? ` (${failure.reason})` : ""}` +
1610
+ `${failure.message ? `: ${failure.message}` : ""}\n${logs}`,
1611
+ );
1612
+ }
1613
+ const want = job.spec?.completions ?? 1;
1614
+ const succeeded = job.status?.succeeded ?? 0;
1615
+ const complete = conditions.some(
1616
+ (c) => c.type === "Complete" && c.status === "True",
1617
+ );
1618
+ if (complete || succeeded >= want) return job;
1619
+ last = `${succeeded}/${want} completion(s), ${job.status?.active ?? 0} pod(s) active`;
1620
+ return null;
1621
+ },
1622
+ { timeoutMs, intervalMs: opts?.intervalMs ?? WAIT_INTERVAL_MS },
1623
+ )) as Wrapped<V1Job>;
1624
+ } catch (err) {
1625
+ // A Job that failed already carries its pods' logs — don't bury it.
1626
+ if (/Job .* failed/.test(errorText(err))) throw err;
1627
+ const logs = await readLogs(deps.client, {
1628
+ namespace,
1629
+ selector: `job-name=${name}`,
1630
+ }).catch(() => "");
1631
+ throw new Error(
1632
+ `k3s(${deps.clusterName}): Job ${name} did not complete within ${timeoutMs / 1000}s` +
1633
+ `${last ? ` — ${last}` : ""}.\n${logs}\n(${errorText(err)})`,
1634
+ );
1635
+ }
1636
+ }
1637
+
1638
+ /**
1639
+ * Post-Ready setup. Apply the Traefik manifest (hostNetwork) and, when
1640
+ * enabled, the in-cluster registry; wait for each Deployment to come
1641
+ * Ready. Captured by the warm-template snapshot, so warm starts pay none
1642
+ * of this cost.
1643
+ *
1644
+ * When the cluster declares `ingressDomains` (and the in-VM CA is
1645
+ * present), TLS is enabled: we mint a CA-signed leaf covering `*.<domain>`
1646
+ * for each domain, stash it in the `traefik-default-tls` Secret + a
1647
+ * file-provider ConfigMap, and bring Traefik up with a `websecure` :443
1648
+ * entrypoint serving it as the default cert. Those domains are then
1649
+ * reachable over HTTPS with a cert the test framework already trusts.
1650
+ */
1651
+ async function setupK3sCluster(
1652
+ name: string,
1653
+ helpers: K3sHelpers,
1654
+ opts: { registry: boolean; ingressDomains: string[]; traefik: boolean },
1655
+ ): Promise<void> {
1656
+ const tlsEnabled = opts.traefik && opts.ingressDomains.length > 0 && caPresent();
1657
+ if (tlsEnabled) {
1658
+ const { cert, key } = await issueIngressCert(
1659
+ opts.ingressDomains.map((d) => `*.${d}`),
1660
+ );
1661
+ // Apply the cert Secret + dynamic-config ConfigMap before the
1662
+ // Deployment that mounts them. `stringData` lets us hand over plain
1663
+ // PEM; the API server base64-encodes it.
1664
+ await helpers.client.core.createNamespacedSecret({
1665
+ namespace: "kube-system",
1666
+ body: {
1667
+ metadata: { name: TRAEFIK_TLS_SECRET, namespace: "kube-system" },
1668
+ type: "kubernetes.io/tls",
1669
+ stringData: { "tls.crt": cert, "tls.key": key },
1670
+ },
1671
+ });
1672
+ await helpers.client.core.createNamespacedConfigMap({
1673
+ namespace: "kube-system",
1674
+ body: {
1675
+ metadata: {
1676
+ name: TRAEFIK_DYNAMIC_CONFIGMAP,
1677
+ namespace: "kube-system",
1678
+ },
1679
+ data: { "tls.yaml": TRAEFIK_DYNAMIC_TLS },
1680
+ },
1681
+ });
1682
+ }
1683
+ if (opts.traefik) await helpers.apply(buildTraefikManifest(tlsEnabled));
1684
+ if (opts.registry) await helpers.apply(REGISTRY_MANIFEST);
1685
+ // Both rollouts proceed independently inside the cluster — wait on them
1686
+ // concurrently (they used to serialize, wasting up to a rollout's tail).
1687
+ // 250ms polling: these two sit on the cold start's critical path, where
1688
+ // the default 500ms interval wastes real time.
1689
+ const wait = (deployment: string): Promise<void> =>
1690
+ helpers.waitForRollout(deployment, {
1691
+ namespace: "kube-system",
1692
+ timeoutMs: 120_000,
1693
+ intervalMs: 250,
1694
+ });
1695
+ const waits = opts.traefik ? [wait("traefik")] : [];
1696
+ if (opts.registry) waits.push(wait("spectest-registry"));
1697
+ await Promise.all(waits);
1698
+ }
1699
+
1700
+ /**
1701
+ * Snapshot of a namespace's state, dumped whenever a wait times out: pod
1702
+ * phases, the container reasons/messages behind them, and recent Warning
1703
+ * events. That set answers the question a rollout timeout actually raises —
1704
+ * did the pods schedule, did the image pull, did the container crash — so
1705
+ * the failure is diagnosable without a second round-trip to the cluster.
1706
+ */
1707
+ async function collectDiagnostics(
1708
+ client: K3sClient,
1709
+ namespace: string,
1710
+ ): Promise<string> {
1711
+ const lines: string[] = [];
1712
+ try {
1713
+ const pods = await client.core.listNamespacedPod({ namespace });
1714
+ lines.push(`${namespace} pods (${pods.items.length}):`);
1715
+ for (const p of pods.items) {
1716
+ const phase = p.status?.phase ?? "?";
1717
+ const cs = p.status?.containerStatuses ?? [];
1718
+ const reasons = cs
1719
+ .map((c) => c.state?.waiting?.reason ?? c.state?.terminated?.reason ?? "")
1720
+ .filter((s) => s)
1721
+ .join(",");
1722
+ lines.push(
1723
+ ` ${p.metadata?.name ?? "?"}: phase=${phase}${reasons ? ` reasons=${reasons}` : ""}`,
1724
+ );
1725
+ // The waiting `message` carries containerd's actual error — e.g. the
1726
+ // failing endpoint, an upstream `429 Too Many Requests`, or a
1727
+ // `connection refused`. The `reason` alone (`ErrImagePull`) hides all
1728
+ // of that, which is exactly what we need when a pull won't settle.
1729
+ for (const c of cs) {
1730
+ const msg =
1731
+ c.state?.waiting?.message ?? c.state?.terminated?.message ?? "";
1732
+ if (msg) lines.push(` ${c.name}: ${msg.replace(/\s+/g, " ").trim()}`);
1733
+ }
1734
+ }
1735
+ } catch (err) {
1736
+ lines.push(`(listing pods failed: ${(err as Error)?.message ?? String(err)})`);
1737
+ }
1738
+ // Recent Warning events surface pull failures the kubelet emits before a
1739
+ // container status even settles (FailedPull / Failed / BackOff), with the
1740
+ // raw containerd message attached. Best-effort: never let diagnostics throw.
1741
+ try {
1742
+ const events = await client.core.listNamespacedEvent({ namespace });
1743
+ const warnings = (events.items ?? [])
1744
+ .filter((e) => e.type === "Warning")
1745
+ .map((e) => ({
1746
+ obj: e.involvedObject?.name ?? "?",
1747
+ reason: e.reason ?? "?",
1748
+ message: (e.message ?? "").replace(/\s+/g, " ").trim(),
1749
+ }))
1750
+ .filter((e) => e.message);
1751
+ if (warnings.length) {
1752
+ lines.push(`${namespace} Warning events (${warnings.length}):`);
1753
+ // Keep the tail — newest events are appended last by the API.
1754
+ for (const w of warnings.slice(-12)) {
1755
+ lines.push(` ${w.obj} [${w.reason}] ${w.message}`);
1756
+ }
1757
+ }
1758
+ } catch (err) {
1759
+ lines.push(`(listing events failed: ${(err as Error)?.message ?? String(err)})`);
1760
+ }
1761
+ return lines.join("\n");
1762
+ }
1763
+
1764
+ /**
1765
+ * A ready-to-use single-node Kubernetes cluster (k3s). Drop into
1766
+ * `environment.services`:
1767
+ *
1768
+ * ```ts
1769
+ * services: { k8s: k3s() }
1770
+ * ```
1771
+ *
1772
+ * Tests get `@kubernetes/client-node` API objects pre-wired to this
1773
+ * cluster at `ctx.svc.<key>.client` — `core`, `apps`, and a generic
1774
+ * `objects` (`KubernetesObjectApi`). Every API call is recorded on the
1775
+ * test event log alongside `fetch` calls. There's also `apply(yaml)`
1776
+ * sugar for piping a multi-document manifest in.
1777
+ *
1778
+ * **Ingress.** We deploy Traefik ourselves in `hostNetwork` mode
1779
+ * during `setup()`. Traefik binds the cluster container's :80 directly
1780
+ * (no ServiceLB / klipper-lb needed), watches Ingress objects via the
1781
+ * API, and routes incoming requests to pod Endpoints. Any `hostnames`
1782
+ * declared on this service in env.ts therefore route through Traefik:
1783
+ * a peer doing `fetch("http://app.example.com")` resolves the host to
1784
+ * the k3s container's IP (via systest-resolver), lands on Traefik,
1785
+ * and gets dispatched to the matching Ingress rule's backend pods.
1786
+ *
1787
+ * **kube-proxy runs in `nftables` mode** (`--kube-proxy-arg=proxy-mode
1788
+ * =nftables`), not the iptables default. This started as a workaround
1789
+ * for a stock kernel with no `xt_comment` match — kube-proxy's iptables
1790
+ * rules carry `-m comment`, and the kernel rejected every one of them,
1791
+ * breaking pod→ClusterIP routing and any pod talking to the in-cluster
1792
+ * API (helm-install Jobs, CoreDNS, …). We build the guest kernel
1793
+ * ourselves now and it has the match, but nftables mode is the better
1794
+ * path regardless, so it stays. It is GA in k8s 1.32, which is what
1795
+ * pins `DEFAULT_K3S_VERSION`.
1796
+ *
1797
+ * Flannel uses the `host-gw` backend: single-node clusters never route
1798
+ * pod traffic off the node, so VXLAN encapsulation is pure overhead.
1799
+ */
1800
+ /**
1801
+ * Default k3s docker image tag.
1802
+ *
1803
+ * The cluster's system images (the `rancher/k3s` image itself, plus
1804
+ * coredns / local-path-provisioner / pause and the Traefik we deploy)
1805
+ * are pulled on first boot through the host `zot` pull-through cache —
1806
+ * `registries.yaml` (seeded via `files`) mirrors `docker.io` and
1807
+ * `registry.k8s.io` at it. The first-ever cluster boot on a cold-cache
1808
+ * host pays the upstream pull once; thereafter zot serves the blobs
1809
+ * host-wide and the warm-template snapshot captures the booted cluster,
1810
+ * so neither cold-cache nor warm starts re-pull. Any `opts.version`
1811
+ * works — there's no base-snapshot release to keep in sync with.
1812
+ *
1813
+ * **Why v1.32.x:** kube-proxy's `nftables` proxy mode is GA in k8s 1.32
1814
+ * (beta in 1.31, alpha-gated in 1.30). The component runs kube-proxy in
1815
+ * that mode; dropping below 1.31 falls back to the iptables path.
1816
+ */
1817
+ const DEFAULT_K3S_VERSION = "v1.32.1-k3s1";
1818
+
1819
+ export function k3s(opts: K3sOptions = {}) {
1820
+ const version = opts.version ?? DEFAULT_K3S_VERSION;
1821
+ const extra = opts.extraArgs ?? [];
1822
+ assertUsableExtraArgs(extra);
1823
+ const registryEnabled = opts.registry !== false;
1824
+ const traefikEnabled = opts.traefik !== false;
1825
+ const loadBalancerEnabled = opts.loadBalancer !== false;
1826
+ // Wildcard ingress domains. Drives both the `provides(... dnsName)`
1827
+ // wiring below and (when non-empty) the CA-signed TLS default cert that
1828
+ // setupK3sCluster mints so these domains are reachable over HTTPS.
1829
+ const ingressDomains = opts.ingressDomains ?? [];
1830
+ // `/etc/rancher/k3s/registries.yaml` (host-cache mirrors + trust for
1831
+ // the in-cluster registry). Seeded via `files` because k3s reads it
1832
+ // only at startup, before any setup hook could run.
1833
+ const registriesYaml = buildRegistriesYaml(registryEnabled);
1834
+ const serverArgs = [
1835
+ "k3s",
1836
+ "server",
1837
+ // CoreDNS / pod DNS upstream. Without this, k3s sees only the
1838
+ // loopback 127.0.0.11 (Docker's embedded DNS) in the container's
1839
+ // /etc/resolv.conf, decides no usable nameserver exists, and writes a
1840
+ // fallback `nameserver 8.8.8.8` that CoreDNS then forwards to — so
1841
+ // pods reach the public internet but NOT peer services on
1842
+ // spectest-net (`<svc>.internal`, fakes, service-TLS hosts all
1843
+ // NXDOMAIN). We instead point k3s at the container's default gateway
1844
+ // — the spectest-net bridge gateway, where spectest-resolver binds a
1845
+ // second listener for exactly this. The file is written by the
1846
+ // command wrapper below because the gateway IP is only known at
1847
+ // container start.
1848
+ "--resolv-conf=/run/spectest-resolv.conf",
1849
+ // metrics-server isn't useful in a test cluster.
1850
+ "--disable=metrics-server",
1851
+ // The bundled traefik is disabled because we install our own with
1852
+ // hostNetwork (see setupK3sCluster) — ours is wired to the in-VM CA
1853
+ // for `ingressDomains` TLS and to a fixed IngressClass. Opt out of
1854
+ // ours entirely with `k3s({ traefik: false })`; passing
1855
+ // `--disable=traefik` via `extraArgs` does NOT do that (it only
1856
+ // re-disables the bundled one, which is already off) and is rejected
1857
+ // below rather than silently ignored.
1858
+ "--disable=traefik",
1859
+ // ServiceLB (klipper-lb) stays ENABLED, so `type: LoadBalancer`
1860
+ // Services get an address and work. It was disabled for years
1861
+ // because klipper-lb binds its ports with a CNI portmap hostPort,
1862
+ // and portmap's iptables-nft rules need the `xt_comment` netfilter
1863
+ // match — absent from the stock kernel we used to run on, which made
1864
+ // every LoadBalancer hang at <pending> with no svclb DaemonSet. We
1865
+ // build the guest kernel ourselves now and it carries
1866
+ // CONFIG_NETFILTER_XT_MATCH_COMMENT (see
1867
+ // scripts/local-vms-kernel-additions.config), so the constraint is
1868
+ // gone. local-storage stays enabled too: a controller pod that binds
1869
+ // no host ports.
1870
+ ...(loadBalancerEnabled ? [] : ["--disable=servicelb"]),
1871
+ // Pod CIDR MUST avoid 10.42.0.0/16: that's the spectest-br0 host
1872
+ // bridge subnet, whose gateway 10.42.0.1 fronts the host image caches
1873
+ // (zot :5000-5007, buildkitd :1234). k3s's *default* pod CIDR is also
1874
+ // 10.42.0.0/16 — with --flannel-backend=host-gw, flannel programs that
1875
+ // route into the node's own routing table and gives cni0 the subnet's
1876
+ // .1 (10.42.0.1). That shadows the route to the host gateway, so once
1877
+ // CNI comes up the node can no longer reach 10.42.0.1 and every
1878
+ // subsequent registry pull dies with "connect: connection refused"
1879
+ // (e.g. the in-cluster registry's `registry:2`, applied after the
1880
+ // cluster is up — the airgap-bundled system images pull *before* CNI
1881
+ // and so sneak through). Move pods to 10.44/service to 10.45.
1882
+ "--cluster-cidr=10.44.0.0/16",
1883
+ "--service-cidr=10.45.0.0/16",
1884
+ "--cluster-dns=10.45.0.10",
1885
+ "--flannel-backend=host-gw",
1886
+ "--write-kubeconfig-mode=644",
1887
+ // kube-proxy in nftables mode: native nftables rules, no
1888
+ // xt_comment dependency. Pod→ClusterIP routing works, so
1889
+ // CoreDNS / helm-install / anything-talking-to-the-API works.
1890
+ // GA in k8s 1.32.
1891
+ "--kube-proxy-arg=proxy-mode=nftables",
1892
+ ...extra,
1893
+ ].join(" ");
1894
+ // The service `command` runs under `/bin/sh -c` (see runContainer in
1895
+ // daemon.ts), so derive the bridge gateway from the container's default
1896
+ // route at start time, write it as the k3s resolv-conf, then exec k3s
1897
+ // (exec so it stays the container's main process and signals / the
1898
+ // readyCheck behave exactly as before). `/run` is a tmpfs on this
1899
+ // service, so the file is writable and never persisted into a snapshot.
1900
+ const cmd =
1901
+ "GW=\"$(ip route 2>/dev/null | awk '/^default/{print $3; exit}')\"; " +
1902
+ 'if [ -n "$GW" ]; then ' +
1903
+ "printf 'nameserver %s\\noptions ndots:0\\n' \"$GW\" > /run/spectest-resolv.conf; " +
1904
+ "else echo 'spectest: no default gateway found; k3s pod DNS for peer services will not resolve' >&2; " +
1905
+ ": > /run/spectest-resolv.conf; fi; " +
1906
+ // Make the root mount shared. A CSI node plugin bind-mounts volumes
1907
+ // under /var/lib/kubelet and needs that propagation to reach the
1908
+ // kubelet and the workload pod; docker gives a container a private
1909
+ // root, so without this every CSI node plugin fails to publish and
1910
+ // any snapshot/PVC-backed storage test is impossible. `kind` does the
1911
+ // same in its entrypoint for the same reason. Best-effort: on a
1912
+ // kernel/runtime that refuses it, the cluster still boots — only CSI
1913
+ // is affected.
1914
+ "mount --make-rshared / 2>/dev/null || " +
1915
+ "echo 'spectest: could not make / rshared; CSI node plugins may fail to publish volumes' >&2; " +
1916
+ `exec ${serverArgs}`;
1917
+ // Plain /readyz probe. On a warm zot cache the cluster's images are
1918
+ // already local, so the first boot completes in seconds; the
1919
+ // first-ever boot on a cold-cache host pulls through the mirror and
1920
+ // can take a couple of minutes (covered by readyTimeoutSecs).
1921
+ const readyCmd = "kubectl get --raw=/readyz >/dev/null 2>&1";
1922
+ const def = {
1923
+ image: { type: "registry", reference: `rancher/k3s:${version}` },
1924
+ command: cmd,
1925
+ privileged: true,
1926
+ tmpfs: ["/run", "/var/run"],
1927
+ cgroupns: "host",
1928
+ // 80/443 are advisory — peer services and host code reach them via
1929
+ // the k3s container's IP. The component's Traefik binds them directly
1930
+ // (hostNetwork, so it shares the container's netns). 5000 is the
1931
+ // in-cluster registry (hostNetwork pod bound to the container netns),
1932
+ // reached by peers at the cluster's own `<key>.internal:5000` alias.
1933
+ // A `type: LoadBalancer` Service's ports are bound by klipper-lb in
1934
+ // the same netns and are reachable the same way, without appearing
1935
+ // in this list (it's documentation, not a firewall).
1936
+ ports: registryEnabled ? [80, 443, 6443, K3S_REGISTRY_PORT] : [80, 443, 6443],
1937
+ // NOTE: do NOT mount /var/lib/rancher/k3s/agent/containerd as a cache
1938
+ // volume. It was tried (to spare a recreated cluster re-pulling its
1939
+ // system images on delta restores) and a fresh k3s server against the
1940
+ // previous container's containerd store — killed un-cleanly by the
1941
+ // teardown's `docker rm -f` — wedged the apiserver minutes in
1942
+ // (rollouts never settled, pod listing started failing). The zot
1943
+ // mirror already makes those re-pulls cheap; the residual win wasn't
1944
+ // worth the recovery semantics of a crash-state store under a fresh
1945
+ // cluster db.
1946
+ ...(registriesYaml
1947
+ ? {
1948
+ files: [
1949
+ { path: "/etc/rancher/k3s/registries.yaml", content: registriesYaml },
1950
+ ],
1951
+ }
1952
+ : {}),
1953
+ readyCheck: {
1954
+ type: "exec" as const,
1955
+ command: readyCmd,
1956
+ timeoutSecs: opts.readyTimeoutSecs ?? 120,
1957
+ },
1958
+ setup: async ({ name, helpers }: { name: string; helpers: K3sHelpers }) => {
1959
+ await setupK3sCluster(name, helpers, {
1960
+ registry: registryEnabled,
1961
+ ingressDomains,
1962
+ traefik: traefikEnabled,
1963
+ });
1964
+ },
1965
+ helpers: async ({ name, exec, poll }: ServiceHelpersContext): Promise<K3sHelpers> => {
1966
+ // Read the cluster's kubeconfig and address the API server by its
1967
+ // auto-assigned `<name>.internal` hostname on spectest-net. TLS
1968
+ // verification is off (see the K3sHelpers docstring), so the
1969
+ // server's cert SAN list doesn't need to include the .internal
1970
+ // name.
1971
+ const kcRead = await exec(name, ["cat", "/etc/rancher/k3s/k3s.yaml"]);
1972
+ if (kcRead.exitCode !== 0) {
1973
+ throw new Error(
1974
+ `k3s(${name}): failed to read kubeconfig from container: ${kcRead.stderr.trim()}`,
1975
+ );
1976
+ }
1977
+
1978
+ const kubeconfig = new KubeConfig();
1979
+ kubeconfig.loadFromString(kcRead.stdout);
1980
+
1981
+ const server = `https://${name}.internal:6443`;
1982
+ // Update kc.clusters so any code that reads kubeconfig sees the
1983
+ // right server URL, but the actual request server comes from the
1984
+ // Configuration we build below. `Cluster.server` is typed `readonly`
1985
+ // by @kubernetes/client-node, but the loaded object is a plain mutable
1986
+ // record — write through a mutable view rather than rebuild the config.
1987
+ for (const cluster of kubeconfig.clusters) {
1988
+ (cluster as { -readonly [K in keyof typeof cluster]: typeof cluster[K] }).server =
1989
+ server;
1990
+ }
1991
+
1992
+ const httpApi = new FetchHttpLibrary();
1993
+ const baseServer = new ServerConfiguration(server, {});
1994
+ // eslint-disable-next-line @typescript-eslint/no-explicit-any
1995
+ const config = createConfiguration({
1996
+ baseServer,
1997
+ authMethods: { default: kubeconfig as any },
1998
+ httpApi: httpApi as any,
1999
+ });
2000
+
2001
+ const core = withTagging(new CoreV1Api(config));
2002
+ const apps = withTagging(new AppsV1Api(config));
2003
+ const batch = withTagging(new BatchV1Api(config));
2004
+ // The generic dynamic API keeps its type parameters (see
2005
+ // {@link TaggedObjectApi}); `withTagging` only rewrites the runtime
2006
+ // values, so the cast re-states what the proxy actually returns.
2007
+ const objectApi = new ResourceAwareObjectApi(config);
2008
+ const objects = withTagging(objectApi) as unknown as TaggedObjectApi;
2009
+
2010
+ const client: K3sClient = { core, apps, batch, objects };
2011
+ const deps: ClusterDeps = {
2012
+ clusterName: name,
2013
+ client,
2014
+ objects,
2015
+ meta: objectApi,
2016
+ poll,
2017
+ };
2018
+
2019
+ return {
2020
+ kubeconfig,
2021
+ client,
2022
+ apply: (manifest, applyOpts) => applyManifest(deps, manifest, applyOpts),
2023
+ waitForRollout: (target, waitOpts) => waitForRollout(deps, target, waitOpts),
2024
+ waitForJob: (jobName, waitOpts) => waitForJob(deps, jobName, waitOpts),
2025
+ logs: async (logOpts) =>
2026
+ wrap(await readLogs(client, logOpts), undefined) as unknown as Wrapped<string>,
2027
+ list: <T extends KubernetesObject = KubernetesObject>(listOpts: {
2028
+ apiVersion: string;
2029
+ kind: string;
2030
+ namespace?: string;
2031
+ fieldSelector?: string;
2032
+ labelSelector?: string;
2033
+ limit?: number;
2034
+ }): Promise<Wrapped<KubernetesListObject<T>>> =>
2035
+ objects.list<T>(
2036
+ listOpts.apiVersion,
2037
+ listOpts.kind,
2038
+ listOpts.namespace,
2039
+ undefined,
2040
+ undefined,
2041
+ undefined,
2042
+ listOpts.fieldSelector,
2043
+ listOpts.labelSelector,
2044
+ listOpts.limit,
2045
+ ),
2046
+ };
2047
+ },
2048
+ } satisfies ServiceDefinition<K3sHelpers>;
2049
+
2050
+ // Wildcard ingress domains → a dnsName(`*.<domain>`, { service: self })
2051
+ // each, attached via provides(). SELF_SERVICE_TOKEN resolves to this
2052
+ // service's key at load time (the component can't know it here). The
2053
+ // resolver then points every host under the domain at the cluster.
2054
+ if (ingressDomains.length === 0) return def;
2055
+ return provides(
2056
+ def,
2057
+ ingressDomains.map((domain) =>
2058
+ dnsName(`*.${domain}`, { service: SELF_SERVICE_TOKEN }),
2059
+ ),
2060
+ );
2061
+ }