@specific.dev/spectest 0.30.0 → 0.32.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/components/index.d.ts +1 -1
- package/dist/components/k3s.d.ts +175 -12
- package/dist/components/k3s.js +403 -62
- package/dist/daemon.js +182 -109
- package/dist/index.d.ts +251 -217
- package/dist/locator.js +78 -5
- package/package.json +1 -1
package/dist/components/k3s.js
CHANGED
|
@@ -3,7 +3,7 @@ import { spawn as nodeSpawn } from "node:child_process";
|
|
|
3
3
|
import { randomUUID } from "node:crypto";
|
|
4
4
|
import { existsSync, readFileSync } from "node:fs";
|
|
5
5
|
import { readFile, unlink } from "node:fs/promises";
|
|
6
|
-
import { AppsV1Api, CoreV1Api, KubeConfig, KubernetesObjectApi, ResponseContext, ServerConfiguration, createConfiguration, loadAllYaml, } from "@kubernetes/client-node";
|
|
6
|
+
import { AppsV1Api, BatchV1Api, CoreV1Api, KubeConfig, KubernetesObjectApi, PatchStrategy, ResponseContext, ServerConfiguration, createConfiguration, loadAllYaml, } from "@kubernetes/client-node";
|
|
7
7
|
import { Observable } from "@kubernetes/client-node/dist/gen/rxjsStub.js";
|
|
8
8
|
import { dnsName, provides, SELF_SERVICE_TOKEN } from "../index.js";
|
|
9
9
|
import { readRaw, readTag, wrap } from "../inspect.js";
|
|
@@ -696,42 +696,377 @@ spec:
|
|
|
696
696
|
- name: data
|
|
697
697
|
emptyDir: {}
|
|
698
698
|
`;
|
|
699
|
+
// ────────────────────────────────────────────────────────────────────────
|
|
700
|
+
// Apply — server-side, converging
|
|
701
|
+
//
|
|
702
|
+
// `kubectl apply --server-side --force-conflicts`, over the API. Two
|
|
703
|
+
// properties make a real manifest set work where plain `create` didn't:
|
|
704
|
+
//
|
|
705
|
+
// * create-or-update. SSA PATCHes with `application/apply-patch+yaml`,
|
|
706
|
+
// so re-applying an edited manifest converges instead of failing
|
|
707
|
+
// `AlreadyExists` — which is what a `dependsOn` child re-applying over
|
|
708
|
+
// its parent's state, or a second apply of a shared base, needs.
|
|
709
|
+
// * ordering by retry, not by sorting. A CR whose CRD is in the same
|
|
710
|
+
// manifest, or a resource an admission webhook (cert-manager, CNPG)
|
|
711
|
+
// rejects while its own pod is still starting, fails the first round
|
|
712
|
+
// and lands on a later one. Only errors that read as "not yet" are
|
|
713
|
+
// retried; a genuinely malformed object throws immediately.
|
|
714
|
+
// ────────────────────────────────────────────────────────────────────────
|
|
715
|
+
/** Default SSA field manager. Stable across runs on purpose: server-side
|
|
716
|
+
* apply keys ownership on it, so a changing value would make every run
|
|
717
|
+
* conflict with the previous one's fields. */
|
|
718
|
+
const APPLY_FIELD_MANAGER = "spectest";
|
|
719
|
+
const APPLY_TIMEOUT_MS = 120_000;
|
|
720
|
+
const WAIT_TIMEOUT_MS = 120_000;
|
|
721
|
+
const WAIT_INTERVAL_MS = 500;
|
|
699
722
|
/**
|
|
700
|
-
*
|
|
701
|
-
*
|
|
702
|
-
*
|
|
723
|
+
* API-server responses that mean "this document can't land *yet*" — the
|
|
724
|
+
* CRD isn't established, an admission webhook's own pod isn't serving,
|
|
725
|
+
* the namespace is still being created, or the write raced another
|
|
726
|
+
* writer. Everything else (a schema violation, a bad field, RBAC) is a
|
|
727
|
+
* real failure and is thrown on the first round.
|
|
703
728
|
*/
|
|
704
|
-
|
|
729
|
+
const TRANSIENT_APPLY_PATTERNS = [
|
|
730
|
+
/no matches for kind/i,
|
|
731
|
+
// The client's own message when it can't resolve a kind to a resource
|
|
732
|
+
// path — i.e. the CRD that defines it isn't established yet. This is
|
|
733
|
+
// the "CR next to its CRD in one manifest" case, so it MUST be retried.
|
|
734
|
+
/Failed to fetch resource metadata/i,
|
|
735
|
+
/could not find the requested resource/i,
|
|
736
|
+
/the server (?:could not find|does not recognize)/i,
|
|
737
|
+
/failed calling webhook/i,
|
|
738
|
+
/webhook .* denied the request: .*(?:not ready|unavailable|connection refused)/i,
|
|
739
|
+
/connection refused/i,
|
|
740
|
+
/no endpoints available/i,
|
|
741
|
+
/context deadline exceeded/i,
|
|
742
|
+
/etcdserver: (?:leader changed|request timed out)/i,
|
|
743
|
+
/the object has been modified/i,
|
|
744
|
+
/namespaces? "[^"]+" not found/i,
|
|
745
|
+
/Internal error occurred/i,
|
|
746
|
+
/(?:^|\n)HTTP-Code: (?:429|500|503|504)/,
|
|
747
|
+
];
|
|
748
|
+
function errorText(err) {
|
|
749
|
+
return err?.message ?? String(err);
|
|
750
|
+
}
|
|
751
|
+
function isTransientApplyError(err) {
|
|
752
|
+
const msg = errorText(err);
|
|
753
|
+
return TRANSIENT_APPLY_PATTERNS.some((re) => re.test(msg));
|
|
754
|
+
}
|
|
755
|
+
function describeDoc(doc) {
|
|
756
|
+
const ns = doc.metadata?.namespace;
|
|
757
|
+
const name = doc.metadata?.name ?? doc.metadata?.generateName ?? "?";
|
|
758
|
+
return `${doc.kind ?? "?"}/${name}${ns ? ` (ns ${ns})` : ""}`;
|
|
759
|
+
}
|
|
760
|
+
/** CRDs and Namespaces go first so the common "CR next to its CRD in one
|
|
761
|
+
* file" case lands in round one rather than costing a retry. Everything
|
|
762
|
+
* else keeps input order (results are always returned in input order). */
|
|
763
|
+
function applyRank(doc) {
|
|
764
|
+
return doc.kind === "CustomResourceDefinition" || doc.kind === "Namespace"
|
|
765
|
+
? 0
|
|
766
|
+
: 1;
|
|
767
|
+
}
|
|
768
|
+
/**
|
|
769
|
+
* `KubernetesObjectApi` with its resource-discovery lookup exposed, so
|
|
770
|
+
* `apply` can tell a namespaced kind from a cluster-scoped one before
|
|
771
|
+
* defaulting `metadata.namespace` (setting it on a cluster-scoped object
|
|
772
|
+
* puts a namespace in the body the API server then rejects).
|
|
773
|
+
*/
|
|
774
|
+
class ResourceAwareObjectApi extends KubernetesObjectApi {
|
|
775
|
+
async resourceInfo(apiVersion, kind) {
|
|
776
|
+
try {
|
|
777
|
+
return await this.resource(apiVersion, kind);
|
|
778
|
+
}
|
|
779
|
+
catch {
|
|
780
|
+
// Discovery failure (the CRD isn't established yet) — the caller
|
|
781
|
+
// treats "unknown" as "leave the document alone"; the apply that
|
|
782
|
+
// follows fails transiently and is retried once discovery works.
|
|
783
|
+
return undefined;
|
|
784
|
+
}
|
|
785
|
+
}
|
|
786
|
+
}
|
|
787
|
+
const sleep = (ms) => new Promise((r) => setTimeout(r, ms));
|
|
788
|
+
/** Default a document's namespace, but only for kinds that have one. */
|
|
789
|
+
async function withDefaultNamespace(deps, doc, namespace) {
|
|
790
|
+
if (!namespace || doc.metadata?.namespace)
|
|
791
|
+
return doc;
|
|
792
|
+
const info = await deps.meta.resourceInfo(doc.apiVersion ?? "v1", doc.kind);
|
|
793
|
+
// Unknown kind (CRD not established yet) or cluster-scoped: don't touch it.
|
|
794
|
+
if (!info?.namespaced)
|
|
795
|
+
return doc;
|
|
796
|
+
return { ...doc, metadata: { ...doc.metadata, namespace } };
|
|
797
|
+
}
|
|
798
|
+
async function applyOne(deps, doc, opts) {
|
|
799
|
+
const spec = await withDefaultNamespace(deps, doc, opts.namespace);
|
|
800
|
+
// Server-side apply addresses the object by name; a `generateName`-only
|
|
801
|
+
// document has none, so it can only ever be created.
|
|
802
|
+
if (!spec.metadata?.name) {
|
|
803
|
+
return deps.objects.create(spec, undefined, undefined, opts.fieldManager);
|
|
804
|
+
}
|
|
805
|
+
return deps.objects.patch(spec, undefined, undefined, opts.fieldManager, opts.force, PatchStrategy.ServerSideApply);
|
|
806
|
+
}
|
|
807
|
+
async function applyManifest(deps, manifest, opts) {
|
|
808
|
+
const docs = loadAllYaml(manifest).filter((d) => !!d && typeof d === "object" && typeof d.kind === "string");
|
|
809
|
+
const fieldManager = opts?.fieldManager ?? APPLY_FIELD_MANAGER;
|
|
810
|
+
const force = opts?.force ?? true;
|
|
811
|
+
const timeoutMs = opts?.timeoutMs ?? APPLY_TIMEOUT_MS;
|
|
705
812
|
const deadline = Date.now() + timeoutMs;
|
|
706
|
-
|
|
707
|
-
|
|
813
|
+
const out = new Array(docs.length);
|
|
814
|
+
let pending = docs
|
|
815
|
+
.map((doc, index) => ({ doc, index }))
|
|
816
|
+
.sort((a, b) => applyRank(a.doc) - applyRank(b.doc) || a.index - b.index);
|
|
817
|
+
let backoff = 0;
|
|
818
|
+
let stalled = [];
|
|
819
|
+
while (pending.length > 0) {
|
|
820
|
+
const failed = [];
|
|
821
|
+
stalled = [];
|
|
822
|
+
for (const item of pending) {
|
|
823
|
+
try {
|
|
824
|
+
out[item.index] = await applyOne(deps, item.doc, {
|
|
825
|
+
namespace: opts?.namespace,
|
|
826
|
+
fieldManager,
|
|
827
|
+
force,
|
|
828
|
+
});
|
|
829
|
+
}
|
|
830
|
+
catch (err) {
|
|
831
|
+
if (!isTransientApplyError(err)) {
|
|
832
|
+
throw new Error(`k3s(${deps.clusterName}): applying ${describeDoc(item.doc)} failed: ${errorText(err)}`);
|
|
833
|
+
}
|
|
834
|
+
failed.push(item);
|
|
835
|
+
stalled.push(`${describeDoc(item.doc)}: ${errorText(err).replace(/\s+/g, " ").slice(0, 300)}`);
|
|
836
|
+
}
|
|
837
|
+
}
|
|
838
|
+
if (failed.length === 0)
|
|
839
|
+
break;
|
|
840
|
+
if (Date.now() >= deadline) {
|
|
841
|
+
throw new Error(`k3s(${deps.clusterName}): apply did not converge within ${timeoutMs / 1000}s — ` +
|
|
842
|
+
`${failed.length} of ${docs.length} document(s) still failing:\n ${stalled.join("\n ")}`);
|
|
843
|
+
}
|
|
844
|
+
// Nothing landed this round: whatever is missing (a webhook pod, a
|
|
845
|
+
// CRD's establishment) needs wall-clock, so back off before retrying.
|
|
846
|
+
// Progress resets the backoff — a long CRD chain shouldn't decay.
|
|
847
|
+
if (failed.length === pending.length) {
|
|
848
|
+
await sleep(Math.min(2_000, 250 * 2 ** backoff));
|
|
849
|
+
backoff += 1;
|
|
850
|
+
}
|
|
851
|
+
else {
|
|
852
|
+
backoff = 0;
|
|
853
|
+
}
|
|
854
|
+
pending = failed;
|
|
855
|
+
}
|
|
856
|
+
return out;
|
|
857
|
+
}
|
|
858
|
+
const ROLLOUT_KINDS = {
|
|
859
|
+
deployment: "Deployment",
|
|
860
|
+
deployments: "Deployment",
|
|
861
|
+
deploy: "Deployment",
|
|
862
|
+
statefulset: "StatefulSet",
|
|
863
|
+
statefulsets: "StatefulSet",
|
|
864
|
+
sts: "StatefulSet",
|
|
865
|
+
daemonset: "DaemonSet",
|
|
866
|
+
daemonsets: "DaemonSet",
|
|
867
|
+
ds: "DaemonSet",
|
|
868
|
+
};
|
|
869
|
+
function parseRolloutTarget(target, namespace) {
|
|
870
|
+
if (typeof target === "string") {
|
|
871
|
+
const slash = target.indexOf("/");
|
|
872
|
+
const kindPart = slash === -1 ? "deployment" : target.slice(0, slash);
|
|
873
|
+
const name = slash === -1 ? target : target.slice(slash + 1);
|
|
874
|
+
const kind = ROLLOUT_KINDS[kindPart.toLowerCase()];
|
|
875
|
+
if (!kind) {
|
|
876
|
+
throw new Error(`waitForRollout("${target}"): unknown workload kind "${kindPart}" — ` +
|
|
877
|
+
`use deployment, statefulset or daemonset (a bare name is a Deployment).`);
|
|
878
|
+
}
|
|
879
|
+
if (!name)
|
|
880
|
+
throw new Error(`waitForRollout("${target}"): missing name.`);
|
|
881
|
+
return { kind, name, namespace: namespace ?? "default" };
|
|
882
|
+
}
|
|
883
|
+
const kind = ROLLOUT_KINDS[String(target.kind ?? "").toLowerCase()];
|
|
884
|
+
const name = target.metadata?.name;
|
|
885
|
+
if (!kind || !name) {
|
|
886
|
+
throw new Error(`waitForRollout(): expected a Deployment/StatefulSet/DaemonSet object with metadata.name, got ` +
|
|
887
|
+
`${JSON.stringify({ kind: target.kind, name })}.`);
|
|
888
|
+
}
|
|
889
|
+
return {
|
|
890
|
+
kind,
|
|
891
|
+
name,
|
|
892
|
+
namespace: namespace ?? target.metadata?.namespace ?? "default",
|
|
893
|
+
};
|
|
894
|
+
}
|
|
895
|
+
/** Rollout state of one workload: settled, or why not. */
|
|
896
|
+
async function rolloutStatus(deps, ref) {
|
|
897
|
+
const { name, namespace } = ref;
|
|
898
|
+
try {
|
|
899
|
+
if (ref.kind === "Deployment") {
|
|
900
|
+
const d = (await deps.client.apps.readNamespacedDeployment({ name, namespace })).unwrap();
|
|
901
|
+
const want = d.spec?.replicas ?? 1;
|
|
902
|
+
const st = d.status ?? {};
|
|
903
|
+
if ((st.observedGeneration ?? 0) < (d.metadata?.generation ?? 0)) {
|
|
904
|
+
return { done: false, reason: "controller has not observed the latest update yet" };
|
|
905
|
+
}
|
|
906
|
+
if (want === 0) {
|
|
907
|
+
return {
|
|
908
|
+
done: (st.replicas ?? 0) === 0,
|
|
909
|
+
reason: `scaling down: ${st.replicas ?? 0} replica(s) remaining`,
|
|
910
|
+
};
|
|
911
|
+
}
|
|
912
|
+
const updated = st.updatedReplicas ?? 0;
|
|
913
|
+
const available = st.availableReplicas ?? 0;
|
|
914
|
+
const total = st.replicas ?? 0;
|
|
915
|
+
if (updated < want) {
|
|
916
|
+
return { done: false, reason: `${updated}/${want} replica(s) updated` };
|
|
917
|
+
}
|
|
918
|
+
if (total > updated) {
|
|
919
|
+
return { done: false, reason: `${total - updated} old replica(s) still terminating` };
|
|
920
|
+
}
|
|
921
|
+
return {
|
|
922
|
+
done: available >= want,
|
|
923
|
+
reason: `${available}/${want} replica(s) available`,
|
|
924
|
+
};
|
|
925
|
+
}
|
|
926
|
+
if (ref.kind === "StatefulSet") {
|
|
927
|
+
const s = (await deps.client.apps.readNamespacedStatefulSet({ name, namespace })).unwrap();
|
|
928
|
+
const want = s.spec?.replicas ?? 1;
|
|
929
|
+
const st = s.status ?? {};
|
|
930
|
+
if ((st.observedGeneration ?? 0) < (s.metadata?.generation ?? 0)) {
|
|
931
|
+
return { done: false, reason: "controller has not observed the latest update yet" };
|
|
932
|
+
}
|
|
933
|
+
const ready = st.readyReplicas ?? 0;
|
|
934
|
+
const updated = st.updatedReplicas ?? 0;
|
|
935
|
+
if (st.updateRevision && st.currentRevision !== st.updateRevision) {
|
|
936
|
+
return {
|
|
937
|
+
done: false,
|
|
938
|
+
reason: `rolling update in progress (${updated}/${want} pod(s) updated)`,
|
|
939
|
+
};
|
|
940
|
+
}
|
|
941
|
+
return { done: ready >= want, reason: `${ready}/${want} pod(s) ready` };
|
|
942
|
+
}
|
|
943
|
+
const ds = (await deps.client.apps.readNamespacedDaemonSet({ name, namespace })).unwrap();
|
|
944
|
+
const st = ds.status ?? {};
|
|
945
|
+
if ((st.observedGeneration ?? 0) < (ds.metadata?.generation ?? 0)) {
|
|
946
|
+
return { done: false, reason: "controller has not observed the latest update yet" };
|
|
947
|
+
}
|
|
948
|
+
const want = st.desiredNumberScheduled ?? 0;
|
|
949
|
+
const ready = st.numberReady ?? 0;
|
|
950
|
+
const updated = st.updatedNumberScheduled ?? 0;
|
|
951
|
+
if (updated < want) {
|
|
952
|
+
return { done: false, reason: `${updated}/${want} node(s) updated` };
|
|
953
|
+
}
|
|
954
|
+
return { done: want > 0 && ready >= want, reason: `${ready}/${want} node(s) ready` };
|
|
955
|
+
}
|
|
956
|
+
catch (err) {
|
|
957
|
+
const msg = errorText(err);
|
|
958
|
+
return {
|
|
959
|
+
done: false,
|
|
960
|
+
reason: /not found|HTTP-Code: 404/i.test(msg)
|
|
961
|
+
? `${ref.kind} ${name} does not exist (yet)`
|
|
962
|
+
: msg.replace(/\s+/g, " ").slice(0, 300),
|
|
963
|
+
};
|
|
964
|
+
}
|
|
965
|
+
}
|
|
966
|
+
async function waitForRollout(deps, target, opts) {
|
|
967
|
+
const ref = parseRolloutTarget(target, opts?.namespace);
|
|
968
|
+
const timeoutMs = opts?.timeoutMs ?? WAIT_TIMEOUT_MS;
|
|
969
|
+
let last = "";
|
|
970
|
+
try {
|
|
971
|
+
await deps.poll(`${ref.kind.toLowerCase()}/${ref.name} rolled out`, async () => {
|
|
972
|
+
const st = await rolloutStatus(deps, ref);
|
|
973
|
+
last = st.reason;
|
|
974
|
+
return st.done || null;
|
|
975
|
+
}, { timeoutMs, intervalMs: opts?.intervalMs ?? WAIT_INTERVAL_MS });
|
|
976
|
+
}
|
|
977
|
+
catch (err) {
|
|
978
|
+
const diag = await collectDiagnostics(deps.client, ref.namespace);
|
|
979
|
+
throw new Error(`k3s(${deps.clusterName}): ${ref.kind}/${ref.name} did not roll out within ` +
|
|
980
|
+
`${timeoutMs / 1000}s${last ? ` — ${last}` : ""}.\n${diag}\n(${errorText(err)})`);
|
|
981
|
+
}
|
|
982
|
+
}
|
|
983
|
+
/** Pod names in a namespace, optionally filtered by label selector. */
|
|
984
|
+
async function podNames(client, namespace, labelSelector) {
|
|
985
|
+
const pods = (await client.core.listNamespacedPod({
|
|
986
|
+
namespace,
|
|
987
|
+
...(labelSelector ? { labelSelector } : {}),
|
|
988
|
+
})).unwrap();
|
|
989
|
+
return (pods.items ?? [])
|
|
990
|
+
.map((p) => p.metadata?.name)
|
|
991
|
+
.filter((n) => !!n);
|
|
992
|
+
}
|
|
993
|
+
async function readLogs(client, opts) {
|
|
994
|
+
const namespace = opts.namespace ?? "default";
|
|
995
|
+
const tailLines = opts.tailLines ?? 200;
|
|
996
|
+
if (!opts.pod && !opts.selector) {
|
|
997
|
+
throw new Error("ctx.svc.<k3s>.logs(): pass either `pod` or `selector`.");
|
|
998
|
+
}
|
|
999
|
+
const names = opts.pod
|
|
1000
|
+
? [opts.pod]
|
|
1001
|
+
: await podNames(client, namespace, opts.selector);
|
|
1002
|
+
const chunks = [];
|
|
1003
|
+
for (const name of names) {
|
|
708
1004
|
try {
|
|
709
|
-
|
|
710
|
-
|
|
711
|
-
|
|
712
|
-
|
|
713
|
-
|
|
714
|
-
name: deployment,
|
|
715
|
-
namespace: "kube-system",
|
|
1005
|
+
const text = (await client.core.readNamespacedPodLog({
|
|
1006
|
+
name,
|
|
1007
|
+
namespace,
|
|
1008
|
+
tailLines,
|
|
1009
|
+
...(opts.container ? { container: opts.container } : {}),
|
|
716
1010
|
})).unwrap();
|
|
717
|
-
|
|
718
|
-
const want = dep.spec?.replicas ?? 1;
|
|
719
|
-
if (ready >= want && want > 0)
|
|
720
|
-
return;
|
|
721
|
-
lastErr = `${deployment} Deployment exists but only ${ready}/${want} replicas Ready`;
|
|
1011
|
+
chunks.push(names.length > 1 ? `==> ${name} <==\n${text}` : String(text));
|
|
722
1012
|
}
|
|
723
1013
|
catch (err) {
|
|
724
|
-
|
|
725
|
-
lastErr = /not found|404/i.test(msg)
|
|
726
|
-
? `${deployment} Deployment does not exist yet`
|
|
727
|
-
: msg;
|
|
1014
|
+
chunks.push(`==> ${name} <==\n(reading logs failed: ${errorText(err)})`);
|
|
728
1015
|
}
|
|
729
|
-
// 250ms: the two sequential rollout waits in setup sit on the cold
|
|
730
|
-
// start's critical path, and a 1s poll wasted up to ~2s of it.
|
|
731
|
-
await new Promise((r) => setTimeout(r, 250));
|
|
732
1016
|
}
|
|
733
|
-
|
|
734
|
-
|
|
1017
|
+
return chunks.join("\n");
|
|
1018
|
+
}
|
|
1019
|
+
async function waitForJob(deps, name, opts) {
|
|
1020
|
+
const namespace = opts?.namespace ?? "default";
|
|
1021
|
+
const timeoutMs = opts?.timeoutMs ?? WAIT_TIMEOUT_MS;
|
|
1022
|
+
let last = "";
|
|
1023
|
+
try {
|
|
1024
|
+
// The predicate returns the RAW job; `poll` wraps the winning value, so
|
|
1025
|
+
// the returned object carries provenance to the wait step (double-
|
|
1026
|
+
// wrapping a proxy would not).
|
|
1027
|
+
return (await deps.poll(`job/${name} completed`, async () => {
|
|
1028
|
+
let job;
|
|
1029
|
+
try {
|
|
1030
|
+
job = (await deps.client.batch.readNamespacedJob({ name, namespace })).unwrap();
|
|
1031
|
+
}
|
|
1032
|
+
catch (err) {
|
|
1033
|
+
const msg = errorText(err);
|
|
1034
|
+
if (!/not found|HTTP-Code: 404/i.test(msg))
|
|
1035
|
+
throw err;
|
|
1036
|
+
last = `Job ${name} does not exist (yet)`;
|
|
1037
|
+
return null;
|
|
1038
|
+
}
|
|
1039
|
+
const conditions = job.status?.conditions ?? [];
|
|
1040
|
+
const failure = conditions.find((c) => c.type === "Failed" && c.status === "True");
|
|
1041
|
+
if (failure) {
|
|
1042
|
+
const logs = await readLogs(deps.client, {
|
|
1043
|
+
namespace,
|
|
1044
|
+
selector: `job-name=${name}`,
|
|
1045
|
+
}).catch((e) => `(reading pod logs failed: ${errorText(e)})`);
|
|
1046
|
+
throw new Error(`k3s(${deps.clusterName}): Job ${name} failed` +
|
|
1047
|
+
`${failure.reason ? ` (${failure.reason})` : ""}` +
|
|
1048
|
+
`${failure.message ? `: ${failure.message}` : ""}\n${logs}`);
|
|
1049
|
+
}
|
|
1050
|
+
const want = job.spec?.completions ?? 1;
|
|
1051
|
+
const succeeded = job.status?.succeeded ?? 0;
|
|
1052
|
+
const complete = conditions.some((c) => c.type === "Complete" && c.status === "True");
|
|
1053
|
+
if (complete || succeeded >= want)
|
|
1054
|
+
return job;
|
|
1055
|
+
last = `${succeeded}/${want} completion(s), ${job.status?.active ?? 0} pod(s) active`;
|
|
1056
|
+
return null;
|
|
1057
|
+
}, { timeoutMs, intervalMs: opts?.intervalMs ?? WAIT_INTERVAL_MS }));
|
|
1058
|
+
}
|
|
1059
|
+
catch (err) {
|
|
1060
|
+
// A Job that failed already carries its pods' logs — don't bury it.
|
|
1061
|
+
if (/Job .* failed/.test(errorText(err)))
|
|
1062
|
+
throw err;
|
|
1063
|
+
const logs = await readLogs(deps.client, {
|
|
1064
|
+
namespace,
|
|
1065
|
+
selector: `job-name=${name}`,
|
|
1066
|
+
}).catch(() => "");
|
|
1067
|
+
throw new Error(`k3s(${deps.clusterName}): Job ${name} did not complete within ${timeoutMs / 1000}s` +
|
|
1068
|
+
`${last ? ` — ${last}` : ""}.\n${logs}\n(${errorText(err)})`);
|
|
1069
|
+
}
|
|
735
1070
|
}
|
|
736
1071
|
/**
|
|
737
1072
|
* Post-Ready setup. Apply the Traefik manifest (hostNetwork) and, when
|
|
@@ -778,26 +1113,30 @@ async function setupK3sCluster(name, helpers, opts) {
|
|
|
778
1113
|
await helpers.apply(REGISTRY_MANIFEST);
|
|
779
1114
|
// Both rollouts proceed independently inside the cluster — wait on them
|
|
780
1115
|
// concurrently (they used to serialize, wasting up to a rollout's tail).
|
|
781
|
-
|
|
782
|
-
|
|
783
|
-
|
|
784
|
-
|
|
785
|
-
|
|
786
|
-
|
|
1116
|
+
// 250ms polling: these two sit on the cold start's critical path, where
|
|
1117
|
+
// the default 500ms interval wastes real time.
|
|
1118
|
+
const wait = (deployment) => helpers.waitForRollout(deployment, {
|
|
1119
|
+
namespace: "kube-system",
|
|
1120
|
+
timeoutMs: 120_000,
|
|
1121
|
+
intervalMs: 250,
|
|
1122
|
+
});
|
|
1123
|
+
const waits = opts.traefik ? [wait("traefik")] : [];
|
|
1124
|
+
if (opts.registry)
|
|
1125
|
+
waits.push(wait("spectest-registry"));
|
|
787
1126
|
await Promise.all(waits);
|
|
788
1127
|
}
|
|
789
1128
|
/**
|
|
790
|
-
* Snapshot of
|
|
791
|
-
*
|
|
792
|
-
*
|
|
1129
|
+
* Snapshot of a namespace's state, dumped whenever a wait times out: pod
|
|
1130
|
+
* phases, the container reasons/messages behind them, and recent Warning
|
|
1131
|
+
* events. That set answers the question a rollout timeout actually raises —
|
|
1132
|
+
* did the pods schedule, did the image pull, did the container crash — so
|
|
1133
|
+
* the failure is diagnosable without a second round-trip to the cluster.
|
|
793
1134
|
*/
|
|
794
|
-
async function
|
|
1135
|
+
async function collectDiagnostics(client, namespace) {
|
|
795
1136
|
const lines = [];
|
|
796
1137
|
try {
|
|
797
|
-
const pods = await
|
|
798
|
-
|
|
799
|
-
});
|
|
800
|
-
lines.push(`kube-system pods (${pods.items.length}):`);
|
|
1138
|
+
const pods = await client.core.listNamespacedPod({ namespace });
|
|
1139
|
+
lines.push(`${namespace} pods (${pods.items.length}):`);
|
|
801
1140
|
for (const p of pods.items) {
|
|
802
1141
|
const phase = p.status?.phase ?? "?";
|
|
803
1142
|
const cs = p.status?.containerStatuses ?? [];
|
|
@@ -824,9 +1163,7 @@ async function collectTraefikDiagnostics(helpers) {
|
|
|
824
1163
|
// container status even settles (FailedPull / Failed / BackOff), with the
|
|
825
1164
|
// raw containerd message attached. Best-effort: never let diagnostics throw.
|
|
826
1165
|
try {
|
|
827
|
-
const events = await
|
|
828
|
-
namespace: "kube-system",
|
|
829
|
-
});
|
|
1166
|
+
const events = await client.core.listNamespacedEvent({ namespace });
|
|
830
1167
|
const warnings = (events.items ?? [])
|
|
831
1168
|
.filter((e) => e.type === "Warning")
|
|
832
1169
|
.map((e) => ({
|
|
@@ -836,7 +1173,7 @@ async function collectTraefikDiagnostics(helpers) {
|
|
|
836
1173
|
}))
|
|
837
1174
|
.filter((e) => e.message);
|
|
838
1175
|
if (warnings.length) {
|
|
839
|
-
lines.push(
|
|
1176
|
+
lines.push(`${namespace} Warning events (${warnings.length}):`);
|
|
840
1177
|
// Keep the tail — newest events are appended last by the API.
|
|
841
1178
|
for (const w of warnings.slice(-12)) {
|
|
842
1179
|
lines.push(` ${w.obj} [${w.reason}] ${w.message}`);
|
|
@@ -1047,7 +1384,7 @@ export function k3s(opts = {}) {
|
|
|
1047
1384
|
traefik: traefikEnabled,
|
|
1048
1385
|
});
|
|
1049
1386
|
},
|
|
1050
|
-
helpers: async ({ name, exec }) => {
|
|
1387
|
+
helpers: async ({ name, exec, poll }) => {
|
|
1051
1388
|
// Read the cluster's kubeconfig and address the API server by its
|
|
1052
1389
|
// auto-assigned `<name>.internal` hostname on spectest-net. TLS
|
|
1053
1390
|
// verification is off (see the K3sHelpers docstring), so the
|
|
@@ -1079,24 +1416,28 @@ export function k3s(opts = {}) {
|
|
|
1079
1416
|
});
|
|
1080
1417
|
const core = withTagging(new CoreV1Api(config));
|
|
1081
1418
|
const apps = withTagging(new AppsV1Api(config));
|
|
1082
|
-
const
|
|
1083
|
-
|
|
1084
|
-
|
|
1085
|
-
|
|
1086
|
-
|
|
1087
|
-
|
|
1088
|
-
|
|
1089
|
-
|
|
1090
|
-
|
|
1091
|
-
|
|
1092
|
-
|
|
1093
|
-
|
|
1094
|
-
|
|
1419
|
+
const batch = withTagging(new BatchV1Api(config));
|
|
1420
|
+
// The generic dynamic API keeps its type parameters (see
|
|
1421
|
+
// {@link TaggedObjectApi}); `withTagging` only rewrites the runtime
|
|
1422
|
+
// values, so the cast re-states what the proxy actually returns.
|
|
1423
|
+
const objectApi = new ResourceAwareObjectApi(config);
|
|
1424
|
+
const objects = withTagging(objectApi);
|
|
1425
|
+
const client = { core, apps, batch, objects };
|
|
1426
|
+
const deps = {
|
|
1427
|
+
clusterName: name,
|
|
1428
|
+
client,
|
|
1429
|
+
objects,
|
|
1430
|
+
meta: objectApi,
|
|
1431
|
+
poll,
|
|
1095
1432
|
};
|
|
1096
1433
|
return {
|
|
1097
1434
|
kubeconfig,
|
|
1098
|
-
client
|
|
1099
|
-
apply,
|
|
1435
|
+
client,
|
|
1436
|
+
apply: (manifest, applyOpts) => applyManifest(deps, manifest, applyOpts),
|
|
1437
|
+
waitForRollout: (target, waitOpts) => waitForRollout(deps, target, waitOpts),
|
|
1438
|
+
waitForJob: (jobName, waitOpts) => waitForJob(deps, jobName, waitOpts),
|
|
1439
|
+
logs: async (logOpts) => wrap(await readLogs(client, logOpts), undefined),
|
|
1440
|
+
list: (listOpts) => objects.list(listOpts.apiVersion, listOpts.kind, listOpts.namespace, undefined, undefined, undefined, listOpts.fieldSelector, listOpts.labelSelector, listOpts.limit),
|
|
1100
1441
|
};
|
|
1101
1442
|
},
|
|
1102
1443
|
};
|