@skyhook-io/k8s-ui 1.10.4 → 1.11.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/package.json +5 -5
- package/src/components/applications/ApplicationDetail.test.tsx +46 -0
- package/src/components/applications/ApplicationDetail.tsx +14 -15
- package/src/components/issues/IssuesView.tsx +9 -0
- package/src/components/issues/diagnostic.ts +3 -0
- package/src/components/issues/issues.test.ts +11 -0
- package/src/components/issues/severity.ts +3 -0
- package/src/components/issues/types.ts +3 -0
- package/src/components/resources/ResourcesSidebar.tsx +1 -1
- package/src/components/resources/ResourcesView.tsx +259 -30
- package/src/components/resources/index.ts +2 -0
- package/src/components/resources/kyverno-cell-gating.test.ts +93 -0
- package/src/components/resources/kyverno-modern-posture.test.ts +290 -0
- package/src/components/resources/renderers/CNPGClusterRenderer.test.tsx +85 -0
- package/src/components/resources/renderers/CNPGClusterRenderer.tsx +109 -9
- package/src/components/resources/renderers/CNPGPoolerRenderer.tsx +8 -4
- package/src/components/resources/renderers/EventRenderer.test.tsx +23 -0
- package/src/components/resources/renderers/EventRenderer.tsx +10 -18
- package/src/components/resources/renderers/KyvernoCELPolicyRenderers.tsx +317 -0
- package/src/components/resources/renderers/KyvernoExceptionRenderers.tsx +264 -0
- package/src/components/resources/renderers/KyvernoPolicyShared.tsx +212 -0
- package/src/components/resources/renderers/VeleroBSLRenderer.tsx +2 -4
- package/src/components/resources/renderers/VeleroBackupRenderer.tsx +32 -9
- package/src/components/resources/renderers/VeleroRestoreRenderer.tsx +30 -9
- package/src/components/resources/renderers/VeleroScheduleRenderer.tsx +12 -6
- package/src/components/resources/renderers/badge-no-handrolled.test.tsx +1 -1
- package/src/components/resources/renderers/cnpg-cells.tsx +37 -2
- package/src/components/resources/renderers/index.ts +4 -0
- package/src/components/resources/renderers/kyverno-modern-cells.tsx +152 -0
- package/src/components/resources/renderers/velero-cells.tsx +128 -9
- package/src/components/resources/renderers/velero-phase-recovery.test.tsx +79 -0
- package/src/components/resources/resource-utils-cnpg.golden.test.ts +112 -0
- package/src/components/resources/resource-utils-cnpg.test.ts +527 -1
- package/src/components/resources/resource-utils-cnpg.ts +373 -56
- package/src/components/resources/resource-utils-kyverno-exceptions.ts +182 -0
- package/src/components/resources/resource-utils-kyverno-modern.ts +588 -0
- package/src/components/resources/resource-utils-velero.test.ts +237 -0
- package/src/components/resources/resource-utils-velero.ts +214 -22
- package/src/components/resources/resource-utils.ts +70 -3
- package/src/components/shared/ResourceActionsBar.tsx +3 -1
- package/src/components/shared/ResourceRendererDispatch.test.tsx +119 -1
- package/src/components/shared/ResourceRendererDispatch.tsx +116 -22
- package/src/components/topology/K8sResourceNode.tsx +11 -0
- package/src/components/topology/TopologyGraph.tsx +85 -5
- package/src/components/trace/ReachabilityGraph.tsx +517 -0
- package/src/components/trace/ReachabilityView.tsx +1173 -0
- package/src/components/trace/TracePanel.test.ts +172 -0
- package/src/components/trace/TracePanel.tsx +558 -0
- package/src/components/trace/TraceSummary.test.ts +213 -0
- package/src/components/trace/TraceSummary.tsx +168 -0
- package/src/components/trace/inClusterSummary.test.ts +98 -0
- package/src/components/trace/inClusterSummary.ts +81 -0
- package/src/components/trace/index.ts +19 -0
- package/src/components/trace/podReach.test.ts +71 -0
- package/src/components/trace/podReach.ts +37 -0
- package/src/components/trace/probe-display.test.ts +107 -0
- package/src/components/trace/probe-display.ts +37 -0
- package/src/components/trace/problemRows.test.ts +107 -0
- package/src/components/trace/reachFixtures.ts +261 -0
- package/src/components/trace/reachGraphModel.test.ts +1423 -0
- package/src/components/trace/reachGraphModel.ts +1521 -0
- package/src/components/trace/reachInspector.test.ts +892 -0
- package/src/components/trace/reachInspector.ts +797 -0
- package/src/components/trace/reachMarks.test.ts +698 -0
- package/src/components/trace/reachMarks.ts +623 -0
- package/src/components/trace/reachOrigins.test.ts +392 -0
- package/src/components/trace/reachOrigins.ts +449 -0
- package/src/components/trace/reachVerdict.test.ts +537 -0
- package/src/components/trace/reachVerdict.ts +601 -0
- package/src/components/trace/traceFingerprint.test.ts +214 -0
- package/src/components/trace/traceFingerprint.ts +122 -0
- package/src/components/trace/traceToSubgraph.test.ts +623 -0
- package/src/components/trace/traceToSubgraph.ts +463 -0
- package/src/components/trace/types.ts +414 -0
- package/src/components/ui/Badge.tsx +36 -0
- package/src/components/ui/ForceDeleteConfirmDialog.tsx +12 -1
- package/src/components/ui/InClusterConsentDialog.test.tsx +83 -0
- package/src/components/ui/InClusterConsentDialog.tsx +141 -0
- package/src/components/ui/index.ts +1 -0
- package/src/components/workload/WorkloadView.tsx +58 -5
- package/src/components/workload/index.ts +1 -1
- package/src/components/workload/reachabilityTab.test.ts +47 -0
- package/src/index.ts +5 -0
- package/src/theme/components.css +64 -0
- package/src/theme/variables.css +5 -0
- package/src/types/core.ts +8 -0
- package/src/utils/api-resources.ts +2 -0
- package/src/utils/application-history.test.ts +1 -1
- package/src/utils/applications.test.ts +63 -1
- package/src/utils/applications.ts +27 -0
- package/src/utils/inClusterConsent.test.ts +109 -0
- package/src/utils/inClusterConsent.ts +72 -0
- package/src/utils/index.ts +1 -0
- package/src/utils/navigation.test.ts +61 -0
- package/src/utils/navigation.ts +11 -9
- package/src/utils/pluralize.test.ts +80 -1
- package/src/utils/pluralize.ts +34 -0
- package/src/utils/resource-icons.test.ts +33 -0
- package/src/utils/resource-icons.ts +65 -2
|
@@ -0,0 +1,601 @@
|
|
|
1
|
+
import type { Trace, ResourceRef, Finding } from './types'
|
|
2
|
+
import type { StatusTone } from '../ui/status-tone'
|
|
3
|
+
|
|
4
|
+
// VIA_API_SERVER is the ONE operator-facing name for the apiserver-proxy vantage
|
|
5
|
+
// - a real request on the CONTROL-PLANE path (API server → endpoint/kubelet →
|
|
6
|
+
// pod) that bypasses kube-proxy + NetworkPolicy. Named by path, not real/fake.
|
|
7
|
+
// MUST stay identical to the Go const VantageAPIServer (coverage.go); pinned by a
|
|
8
|
+
// test on each side so the two headline generators can't drift.
|
|
9
|
+
export const VIA_API_SERVER = 'via API server'
|
|
10
|
+
|
|
11
|
+
export interface ReachVerdictView {
|
|
12
|
+
icon: '✓' | '⚠' | '✗' | ''
|
|
13
|
+
tone: StatusTone
|
|
14
|
+
text: string
|
|
15
|
+
caveat?: string
|
|
16
|
+
/** The full explanation - shown in a tooltip behind an info icon next to the
|
|
17
|
+
* caveat, so the banner stays a one-liner instead of a wall of text. */
|
|
18
|
+
detail?: string
|
|
19
|
+
}
|
|
20
|
+
|
|
21
|
+
/**
|
|
22
|
+
* reachVerdict maps a trace to a plain-language verdict line. The honesty rule:
|
|
23
|
+
* confidence is a LEVEL, not an alarm - a reached/200 (even via the proxy) reads
|
|
24
|
+
* ✓ with a caveat, NEVER a ⚠. ⚠ is reserved for real problems (server-error,
|
|
25
|
+
* partial, unreachable). Deliberately does NOT reuse the alarm VerdictBanner.
|
|
26
|
+
*/
|
|
27
|
+
/** The host an ExternalName Service aliases to, if the path ends in one. An
|
|
28
|
+
* ExternalName is a DNS CNAME to a host OUTSIDE the cluster - there are no pods
|
|
29
|
+
* or ClusterIP, so there is nothing in-cluster to actively probe. */
|
|
30
|
+
export function externalNameHost(trace: Trace): string | undefined {
|
|
31
|
+
for (const h of trace.downstream ?? []) {
|
|
32
|
+
if (h.resource?.kind === 'ExternalName') return h.resource.name
|
|
33
|
+
}
|
|
34
|
+
return undefined
|
|
35
|
+
}
|
|
36
|
+
|
|
37
|
+
// frontDoorStatus reads whether the EXTERNAL entry (the ingress/gateway host a user
|
|
38
|
+
// actually hits) was reached, couldn't-be-verified-from-here (a skip - e.g. split-
|
|
39
|
+
// horizon DNS), or genuinely failed. Only meaningful when the SUBJECT is the front
|
|
40
|
+
// door (Ingress/Gateway); for a Service subject the host is upstream context, not the
|
|
41
|
+
// thing under test. The host probe lives on the subject's own hop.
|
|
42
|
+
// hostFromTarget extracts the bare host from a probe target ("http://host:80/p"
|
|
43
|
+
// → "host"). Uses linear string ops + an anchored numeric check rather than
|
|
44
|
+
// chained `.replace(/\/.*$/…)` regexes - the target can carry cluster-derived
|
|
45
|
+
// hostnames (uncontrolled), and the chained form trips CodeQL's polynomial-regex
|
|
46
|
+
// (ReDoS) check.
|
|
47
|
+
export function hostFromTarget(target?: string): string {
|
|
48
|
+
let s = target || ''
|
|
49
|
+
const scheme = s.indexOf('://')
|
|
50
|
+
if (scheme >= 0) s = s.slice(scheme + 3) // strip scheme
|
|
51
|
+
s = s.split('/', 1)[0] // strip path
|
|
52
|
+
const colon = s.lastIndexOf(':') // strip trailing :port (numeric only - leave IPv6 alone)
|
|
53
|
+
if (colon >= 0 && /^[0-9]+$/.test(s.slice(colon + 1))) s = s.slice(0, colon)
|
|
54
|
+
return s
|
|
55
|
+
}
|
|
56
|
+
|
|
57
|
+
function frontDoorStatus(trace: Trace): { state: 'reached' | 'partial' | 'partial-untested' | 'partial-reached' | 'http-reached' | 'transport' | 'server-error' | 'skipped' | 'failed' | 'none'; host?: string; detail?: string; dnsSkip?: boolean } {
|
|
58
|
+
const k = trace.subject.kind
|
|
59
|
+
if (k !== 'Ingress' && k !== 'Gateway') return { state: 'none' }
|
|
60
|
+
const hop = (trace.downstream ?? []).find((h) => h.resource?.kind === k && h.resource?.name === trace.subject.name) ?? (trace.downstream ?? [])[0]
|
|
61
|
+
const probes = hop?.probes ?? []
|
|
62
|
+
if (probes.length === 0) return { state: 'none' }
|
|
63
|
+
const hostOf = (p: { target?: string }) => hostFromTarget(p.target)
|
|
64
|
+
const host = probes.map(hostOf).find(Boolean)
|
|
65
|
+
const live = probes.filter((p) => !p.skipped)
|
|
66
|
+
// Only an HTTP 2xx (tone 'healthy') is a CONFIRMED front door. A bare TCP/TLS
|
|
67
|
+
// connect or a 5xx is not - an LB can answer the SYN or return 502 while the
|
|
68
|
+
// app is down, so claiming "confirmed" off transport would overclaim health.
|
|
69
|
+
const http2xx = live.find((p) => p.layer === 'http' && p.ok && p.tone === 'healthy')
|
|
70
|
+
// A failure on a DIFFERENT host is a real partial; a not-ok probe on the SAME host as
|
|
71
|
+
// the 2xx (a lower-layer retry that flapped) must not invent a second failed host.
|
|
72
|
+
// When there is no 2xx, every failure still counts (the no-2xx branches below).
|
|
73
|
+
const failed = live.find((p) => !p.ok && (!http2xx || hostOf(p) !== hostOf(http2xx)))
|
|
74
|
+
if (http2xx) {
|
|
75
|
+
// A 2xx confirms ONE host, but a multi-host entry can have a sibling host fail
|
|
76
|
+
// (connection-refused on another hostname). Don't claim a blanket "confirmed"
|
|
77
|
+
// then - surface the partial so the failed host isn't silently dropped.
|
|
78
|
+
if (failed) return { state: 'partial', host, detail: failed.detail || failed.error }
|
|
79
|
+
// A sibling host can return a 5xx (ok=true, tone 'degraded') - not a transport
|
|
80
|
+
// 'failed', so it slips past the check above. Surface it as partial too, or the
|
|
81
|
+
// banner would claim "confirmed" while a declared host is erroring.
|
|
82
|
+
const sibling5xx = live.find((p) => p.layer === 'http' && p.ok && p.tone === 'degraded')
|
|
83
|
+
if (sibling5xx) return { state: 'partial', host, detail: sibling5xx.detail || sibling5xx.error }
|
|
84
|
+
// Symmetric to the FAILED case: a declared SIBLING host whose probes all
|
|
85
|
+
// SKIPPED (split-horizon / internal host that didn't resolve from this vantage)
|
|
86
|
+
// was never tested. A green "confirmed" over an untested host overclaims - name
|
|
87
|
+
// the gap instead and keep the ✓ only when every declared host was confirmed.
|
|
88
|
+
const okHost = hostOf(http2xx)
|
|
89
|
+
// A sibling host can return a 3xx/4xx (ok=true, tone 'reached') - it WAS reached
|
|
90
|
+
// at the HTTP layer but not verified end-to-end. The single-host path refuses to
|
|
91
|
+
// confirm an identical 3xx/4xx (http-reached → neutral), so a blanket "confirmed"
|
|
92
|
+
// here would overclaim. Name the unverified sibling instead.
|
|
93
|
+
const siblingReached = live.find((p) => p.layer === 'http' && p.ok && p.tone === 'reached' && hostOf(p) && hostOf(p) !== okHost)
|
|
94
|
+
if (siblingReached) return { state: 'partial-reached', host: okHost, detail: siblingReached.detail || siblingReached.error }
|
|
95
|
+
const skippedOther = probes.find((p) => p.skipped && hostOf(p) && hostOf(p) !== okHost)
|
|
96
|
+
if (skippedOther) return { state: 'partial-untested', host: okHost, detail: skippedOther.reason }
|
|
97
|
+
return { state: 'reached', host, detail: http2xx.detail }
|
|
98
|
+
}
|
|
99
|
+
const http5xx = live.find((p) => p.layer === 'http' && p.ok && p.tone === 'degraded')
|
|
100
|
+
if (http5xx) return { state: 'server-error', host, detail: http5xx.detail }
|
|
101
|
+
if (failed) return { state: 'failed', host, detail: failed.detail || failed.error }
|
|
102
|
+
// An HTTP 3xx/4xx (tone 'reached', ok=true) means the front door RETURNED an HTTP
|
|
103
|
+
// response - it was reached at the HTTP layer, just not verified end-to-end. Check
|
|
104
|
+
// this BEFORE the TCP/TLS transport branch: a front door answering 301/404 almost
|
|
105
|
+
// always has a preceding TCP-ok probe, so testing transport first would mislabel a
|
|
106
|
+
// real HTTP response as "transport only - the HTTP response wasn’t checked".
|
|
107
|
+
const httpReached = live.find((p) => p.layer === 'http' && p.ok && p.tone === 'reached')
|
|
108
|
+
if (httpReached) return { state: 'http-reached', host, detail: httpReached.detail }
|
|
109
|
+
// Transport connected (TCP/TLS) but no HTTP-layer confirmation - reached, not verified.
|
|
110
|
+
const transport = live.find((p) => p.ok && (p.layer === 'tcp' || p.layer === 'tls'))
|
|
111
|
+
if (transport) return { state: 'transport', host, detail: transport.detail }
|
|
112
|
+
// Nothing live succeeded - a skip. Surface the most relevant skip reason and
|
|
113
|
+
// whether the DNS layer specifically was the skip (so the caveat doesn't claim
|
|
114
|
+
// "didn't resolve" when DNS resolved fine and only TCP/HTTP were skipped).
|
|
115
|
+
const dnsSkipProbe = probes.find((p) => p.skipped && p.layer === 'dns')
|
|
116
|
+
return { state: 'skipped', host, detail: (dnsSkipProbe ?? probes[0])?.reason, dnsSkip: Boolean(dnsSkipProbe) }
|
|
117
|
+
}
|
|
118
|
+
|
|
119
|
+
// worstFindingMessage returns the plain one-line message of the worst static
|
|
120
|
+
// finding (critical first, then warning) across all hops - the operator-facing
|
|
121
|
+
// sentence, not the forensic cause/summary.
|
|
122
|
+
// A finding that describes a backend pod's own runtime state (crash/OOM/pull/init/
|
|
123
|
+
// readiness), as opposed to a config, controller, or front-door concern.
|
|
124
|
+
function isPodStateFinding(f: Finding): boolean {
|
|
125
|
+
const c = f.code ?? ''
|
|
126
|
+
return c.startsWith('problem:') || c.startsWith('pods:') || c.startsWith('svc:no-ready') || f.resource?.kind === 'Pod'
|
|
127
|
+
}
|
|
128
|
+
|
|
129
|
+
function worstFinding(trace: Trace): Finding | undefined {
|
|
130
|
+
const downstream = (trace.downstream ?? []).flatMap((h) => h.findings ?? [])
|
|
131
|
+
// Upstream missing_ref findings concern a SIBLING backend route, not this
|
|
132
|
+
// subject - drop them so a sibling's broken backend doesn't headline an
|
|
133
|
+
// otherwise-healthy path (mirrors TracePanel HopRow's upstream scoping).
|
|
134
|
+
const upstream = (trace.upstreams ?? []).flatMap((h) => h.findings ?? []).filter((x) => !(x.code ?? '').startsWith('missing_ref:'))
|
|
135
|
+
const all = [...downstream, ...upstream]
|
|
136
|
+
return all.find((x) => x.severity === 'critical') ?? all.find((x) => x.severity === 'warning')
|
|
137
|
+
}
|
|
138
|
+
|
|
139
|
+
function worstFindingMessage(trace: Trace): string | undefined {
|
|
140
|
+
const f = worstFinding(trace)
|
|
141
|
+
if (!f) return undefined
|
|
142
|
+
// Pod-state findings carry a raw kubelet message ("back-off 5m0s restarting failed
|
|
143
|
+
// container=c pod=…(uid)"); podDiagnosis put a clean, container-naming string in
|
|
144
|
+
// `cause` - prefer it so the headline reads in plain English, not kubelet noise.
|
|
145
|
+
return isPodStateFinding(f) && f.cause ? f.cause : f.message
|
|
146
|
+
}
|
|
147
|
+
|
|
148
|
+
// entryControllerProblem returns the plain message of an ingress-controller
|
|
149
|
+
// PROBLEM on the entry hop (no controller installed, or its pods unready) - the
|
|
150
|
+
// real front-door blocker, which should headline the verdict.
|
|
151
|
+
function entryControllerProblem(trace: Trace): string | undefined {
|
|
152
|
+
const entry = (trace.downstream ?? [])[0]
|
|
153
|
+
const f = (entry?.findings ?? []).find((x) => x.code === 'ingress:no-controller' || x.code === 'ingress:controller-unready')
|
|
154
|
+
return f?.message
|
|
155
|
+
}
|
|
156
|
+
|
|
157
|
+
// backendDown detects the "wired but no live backend" state: a downstream hop
|
|
158
|
+
// whose pods are SELECTED but NONE are ready (ready === 0, selected > 0). That's
|
|
159
|
+
// a DEFINITIVE break known from endpoint readiness - not a probe limitation - so
|
|
160
|
+
// the verdict must headline the real cause (the crashing/unpullable app), credit
|
|
161
|
+
// the parts that ARE confirmed (the Service is wired to its workload), and point
|
|
162
|
+
// at the backend. Scaled-to-0 (selected === 0) is excluded - it's deliberate
|
|
163
|
+
// dormancy with its own benign read, not a backend that failed.
|
|
164
|
+
//
|
|
165
|
+
// meta.ready counts endpoints the DATAPLANE routes to. For a
|
|
166
|
+
// publishNotReadyAddresses Service that is the PUBLISHED count
|
|
167
|
+
// (publishedEndpointCount, entries.go) - every selected pod that has an IP and
|
|
168
|
+
// isn't Succeeded/Failed; PNR waives the READINESS condition only, not those
|
|
169
|
+
// rules. So when pods are Running, ready === selected and neither this gate nor
|
|
170
|
+
// the "N of M pods ready" wording misfires on readiness alone. But an all-Pending
|
|
171
|
+
// (or IP-less) PNR Service publishes nothing: ready === 0 < selected, so this gate
|
|
172
|
+
// CAN fire - and honestly, because zero published endpoints means no routable
|
|
173
|
+
// backend ("0 of N pods ready" is truthful). Per-pod kubelet state stays visible
|
|
174
|
+
// in the pod grid regardless.
|
|
175
|
+
function backendDown(trace: Trace): { reason: string; raw?: string; cause?: string; pod?: ResourceRef; ready: number; selected: number; transitional: boolean } | undefined {
|
|
176
|
+
const hops = trace.downstream ?? []
|
|
177
|
+
// If any backend can't be judged from pod readiness, we can't claim "no healthy
|
|
178
|
+
// backend" - an unseen sibling might be serving. Unverifiable backends: a read
|
|
179
|
+
// failure (RBAC-redacted / pod-lister error, or meta.endpointSource === 'unknown'
|
|
180
|
+
// with no finding) AND terminal backends with no pod readiness to inspect -
|
|
181
|
+
// selectorless (manual endpoints) or ExternalName (an external host). Stay
|
|
182
|
+
// conservative and fall through to the partial/unknown route logic.
|
|
183
|
+
const unverifiableBackend = hops.some((h) => {
|
|
184
|
+
if (h.meta?.endpointSource === 'unknown' || h.meta?.selectorless === true || h.resource?.kind === 'ExternalName') return true
|
|
185
|
+
return (h.findings ?? []).some((f) => {
|
|
186
|
+
const c = f.code ?? ''
|
|
187
|
+
return c.startsWith('rbac:') || c === 'pods:lister-error' || c === 'svc:selectorless' || c === 'svc:external-name'
|
|
188
|
+
})
|
|
189
|
+
})
|
|
190
|
+
if (unverifiableBackend) return undefined
|
|
191
|
+
// Consider only hops that actually have backend pods to judge (selector matched
|
|
192
|
+
// ≥1 pod). Scaled-to-0 (selected === 0) is deliberate dormancy, not a failure.
|
|
193
|
+
const backendHops = hops.filter((h) => typeof h.meta?.selected === 'number' && (h.meta.selected as number) > 0)
|
|
194
|
+
if (backendHops.length === 0) return undefined
|
|
195
|
+
// "No healthy backend" is only honest when EVERY backend is down. A multi-backend
|
|
196
|
+
// Ingress/Gateway with one healthy backend must NOT be condemned - that path can
|
|
197
|
+
// still serve, so fall through to the partial-route logic below.
|
|
198
|
+
if (backendHops.some((h) => typeof h.meta?.ready === 'number' && (h.meta.ready as number) > 0)) return undefined
|
|
199
|
+
// Pick the most SPECIFIC cause for the headline. The "0/N selected pods ready"
|
|
200
|
+
// symptom now carries a Pod ref too (linkNoReadyToCulprit), so it must be ranked
|
|
201
|
+
// BELOW the real pod-failure findings (CrashLoopBackOff, ImagePullBackOff, …) -
|
|
202
|
+
// otherwise the headline degrades to the generic "the backend pods aren't ready"
|
|
203
|
+
// instead of naming the actual failure. Fall back to the symptom only if no
|
|
204
|
+
// specific cause is present.
|
|
205
|
+
const isSymptom = (code?: string) =>
|
|
206
|
+
code === 'svc:no-ready-endpoints' || /^problem:0\/[1-9]\d* selected pods ready$/.test(code ?? '')
|
|
207
|
+
const problems = hops.flatMap((h) => h.findings ?? []).filter((f) => (f.code ?? '').startsWith('problem:') || f.code === 'svc:no-ready-endpoints')
|
|
208
|
+
const causes = problems.filter((f) => !isSymptom(f.code))
|
|
209
|
+
const realCause =
|
|
210
|
+
causes.find((f) => f.resource?.kind === 'Pod' && f.severity === 'critical') ??
|
|
211
|
+
causes.find((f) => f.severity === 'critical')
|
|
212
|
+
// Transitional: 0 ready but NO root-cause failure finding - only the generic
|
|
213
|
+
// "0/N selected pods ready" symptom, which ALSO fires for Pending pods mid-rollout.
|
|
214
|
+
// Without a corroborating crash/pull/OOM, this is "not ready YET", not a terminal
|
|
215
|
+
// break - must read amber "may be starting", never hard-✗ (a fresh rollout isn't
|
|
216
|
+
// an outage). A real failure finding makes it definitive.
|
|
217
|
+
const transitional = !realCause
|
|
218
|
+
const culprit = realCause ?? problems.find((f) => f.severity === 'critical')
|
|
219
|
+
// Read the pod count from the SAME hop that owns the named culprit, so a
|
|
220
|
+
// multi-backend trace never prints one backend's "M selected" beside another
|
|
221
|
+
// backend's pod name. Falls back to the first backend hop when the culprit is the
|
|
222
|
+
// hop-agnostic "0/N selected" symptom (which lives on the Service entry hop).
|
|
223
|
+
const culpritHop = (culprit && backendHops.find((h) => (h.findings ?? []).includes(culprit))) || backendHops[0]
|
|
224
|
+
const ready = (culpritHop.meta?.ready as number) ?? 0
|
|
225
|
+
const selected = (culpritHop.meta?.selected as number) ?? 0
|
|
226
|
+
const raw = culprit ? (culprit.code ?? '').replace(/^problem:/, '') : undefined
|
|
227
|
+
return { reason: plainPodReason(raw), raw, cause: culprit?.cause, pod: culprit?.resource, ready, selected, transitional }
|
|
228
|
+
}
|
|
229
|
+
|
|
230
|
+
// plainPodReason renders a backend pod's failure reason in operator language -
|
|
231
|
+
// the k8s reason stays as a tag in the headline for those who recognize it.
|
|
232
|
+
function plainPodReason(raw?: string): string {
|
|
233
|
+
switch (raw) {
|
|
234
|
+
case 'CrashLoopBackOff':
|
|
235
|
+
return 'a backend container keeps crashing'
|
|
236
|
+
case 'ImagePullBackOff':
|
|
237
|
+
case 'ErrImagePull':
|
|
238
|
+
case 'InvalidImageName':
|
|
239
|
+
return "the backend pod can't pull its image"
|
|
240
|
+
case 'OOMKilled':
|
|
241
|
+
return 'a backend container ran out of memory'
|
|
242
|
+
default:
|
|
243
|
+
return "the backend pods aren't ready"
|
|
244
|
+
}
|
|
245
|
+
}
|
|
246
|
+
|
|
247
|
+
const REACH_TONE_RANK: Record<string, number> = { healthy: 0, neutral: 1, unknown: 1, degraded: 2, alert: 3, unhealthy: 4 }
|
|
248
|
+
|
|
249
|
+
// reachVerdict floors the reachability read at the backend's static verdict. The
|
|
250
|
+
// probe only sees what it reaches: a static degrade/break it can't observe - a
|
|
251
|
+
// missing TLS secret, a pending LoadBalancer, any config-only finding - must not
|
|
252
|
+
// be painted over by a positive "backend reachable". When the static verdict is
|
|
253
|
+
// worse than the probe read, escalate the tone and surface the worst finding so
|
|
254
|
+
// the headline says why (it never DOWNGRADES - healthy/by-design reads are kept).
|
|
255
|
+
export function reachVerdict(trace: Trace, probed?: boolean): ReachVerdictView {
|
|
256
|
+
const view = reachVerdictBase(trace, probed)
|
|
257
|
+
let floor: StatusTone | undefined =
|
|
258
|
+
trace.verdict === 'broken' ? 'unhealthy' : trace.verdict === 'degraded' ? 'degraded' : undefined
|
|
259
|
+
// A transitional 0-ready backend (no failure finding - pods may be starting) must
|
|
260
|
+
// not be floored to a hard ✗ by a static 'broken' verdict; cap it at degraded so
|
|
261
|
+
// the honest "may be starting" amber survives (a fresh rollout isn't an outage).
|
|
262
|
+
if (floor === 'unhealthy' && backendDown(trace)?.transitional) floor = 'degraded'
|
|
263
|
+
let result = view
|
|
264
|
+
if (floor && (REACH_TONE_RANK[view.tone] ?? 0) < REACH_TONE_RANK[floor]) {
|
|
265
|
+
const why = worstFindingMessage(trace)
|
|
266
|
+
const icon = floor === 'unhealthy' ? '✗' : '⚠'
|
|
267
|
+
// The probe could only see what it reached; a static break/degrade it couldn't
|
|
268
|
+
// observe must HEADLINE the verdict, not sit as a footnote under an optimistic
|
|
269
|
+
// "reached" line. Lead with the problem; demote the probe read to a caveat so
|
|
270
|
+
// the headline never reads "reached/healthy" on a broken/degraded path.
|
|
271
|
+
result = why && why !== view.text
|
|
272
|
+
? { ...view, tone: floor, icon, text: why, caveat: view.text, detail: view.caveat ?? view.detail }
|
|
273
|
+
: { ...view, tone: floor, icon, caveat: why ?? view.caveat }
|
|
274
|
+
}
|
|
275
|
+
// Partial readiness (some pods up, some down) is a degraded-but-reached state. Lead
|
|
276
|
+
// the headline with "N of M pods ready", but only when the headline is itself the
|
|
277
|
+
// backend-pod concern (0-ready is handled by backendDown). If the headline is a
|
|
278
|
+
// controller, front-door, or config concern, prefixing a pod count would misdirect,
|
|
279
|
+
// so require result.text to be exactly the pod finding's message.
|
|
280
|
+
const wf = worstFinding(trace)
|
|
281
|
+
const podLed = !!wf && isPodStateFinding(wf) && result.text === worstFindingMessage(trace)
|
|
282
|
+
// Read the count from the SAME hop that owns the headline finding - a multi-backend
|
|
283
|
+
// trace has >1 Pods hop, and the first hop's count could mismatch the diagnosed one.
|
|
284
|
+
const wfHop = wf ? [...(trace.downstream ?? []), ...(trace.upstreams ?? [])].find((h) => (h.findings ?? []).includes(wf)) : undefined
|
|
285
|
+
const ready = typeof wfHop?.meta?.ready === 'number' ? (wfHop.meta.ready as number) : undefined
|
|
286
|
+
const selected = typeof wfHop?.meta?.selected === 'number' ? (wfHop.meta.selected as number) : wfHop?.config?.podNames?.length
|
|
287
|
+
// Skip when the headline already carries a count ("N of M ready", "N/M") so it's never doubled.
|
|
288
|
+
const alreadyCounted = /pods? ready/i.test(result.text) || /\b\d+\s*(?:of|\/)\s*\d+\b/.test(result.text)
|
|
289
|
+
if (podLed && ready !== undefined && selected !== undefined && ready > 0 && ready < selected && !alreadyCounted) {
|
|
290
|
+
result = { ...result, text: `${ready} of ${selected} pods ready - ${result.text}` }
|
|
291
|
+
}
|
|
292
|
+
return result
|
|
293
|
+
}
|
|
294
|
+
|
|
295
|
+
|
|
296
|
+
function reachVerdictBase(trace: Trace, probed?: boolean): ReachVerdictView {
|
|
297
|
+
// ExternalName - a DNS alias to a host outside the cluster. It IS testable:
|
|
298
|
+
// the probe DNS-resolves + HTTP-reaches the external host, producing a real
|
|
299
|
+
// route. Un-probed, surface the alias + invite the test; once probed, fall
|
|
300
|
+
// through to the normal route-based verdict (which reflects the live result).
|
|
301
|
+
const ext = externalNameHost(trace)
|
|
302
|
+
if (ext && !probed && (trace.routes ?? []).length === 0) {
|
|
303
|
+
return { icon: '', tone: 'neutral', text: `Resolves to ${ext} - external DNS alias`, caveat: 'run the test to check its reachability' }
|
|
304
|
+
}
|
|
305
|
+
const routes = trace.routes ?? []
|
|
306
|
+
const okCount = routes.filter((r) => r.outcome === 'verified' || r.outcome === 'reached').length
|
|
307
|
+
const total = routes.length
|
|
308
|
+
const unreach = routes.filter((r) => r.outcome === 'unreachable' && !r.benign)
|
|
309
|
+
// A proxy-only (indirect) unreachable never tested the real path - it must not
|
|
310
|
+
// count as a CONFIRMED backend failure (that would false-condemn an untested
|
|
311
|
+
// route), mirroring traceToSubgraph's isIndirectUnreach. The broken/onlyIndirect
|
|
312
|
+
// branch below still uses the full `unreach` (it needs to SEE the indirect ones).
|
|
313
|
+
const realUnreach = unreach.filter((r) => r.confidence !== 'indirect')
|
|
314
|
+
const anyIndirectUnreach = unreach.some((r) => r.confidence === 'indirect')
|
|
315
|
+
const any5xx = routes.some((r) => r.outcome === 'server-error')
|
|
316
|
+
const allReal = okCount > 0 && routes.every((r) => !(r.outcome === 'verified' || r.outcome === 'reached') || r.confidence === 'real')
|
|
317
|
+
const backendReachable = okCount > 0 && realUnreach.length === 0 && !any5xx
|
|
318
|
+
|
|
319
|
+
// When the Ingress CONTROLLER itself is the problem (none installed, or pods
|
|
320
|
+
// unready), that IS the front-door answer - lead with it. Otherwise the
|
|
321
|
+
// neutral "backend reachable, entry not tested" below would bury the real
|
|
322
|
+
// blocker (the operator's host can't be served at all, regardless of DNS).
|
|
323
|
+
const ctrlProblem = entryControllerProblem(trace)
|
|
324
|
+
if (ctrlProblem) {
|
|
325
|
+
return { icon: '⚠', tone: 'degraded', text: ctrlProblem }
|
|
326
|
+
}
|
|
327
|
+
|
|
328
|
+
// No live backend (pods wired but none ready) - lead with the real cause and
|
|
329
|
+
// credit what IS confirmed, instead of "unreachable via API server"
|
|
330
|
+
// (which frames a diagnosed backend failure as an untested route). The path is
|
|
331
|
+
// wired correctly up to the workload; the break is in the pods, not the network.
|
|
332
|
+
const bd = backendDown(trace)
|
|
333
|
+
if (bd) {
|
|
334
|
+
const noun = bd.selected === 1 ? 'pod' : 'pods'
|
|
335
|
+
if (bd.transitional) {
|
|
336
|
+
// 0 ready but no failure detected - pods may still be starting (fresh rollout
|
|
337
|
+
// / Pending). Amber + "not ready yet", never the terminal red of a real break.
|
|
338
|
+
return {
|
|
339
|
+
icon: '⚠',
|
|
340
|
+
tone: 'degraded',
|
|
341
|
+
text: `No ready backend yet - ${bd.ready}/${bd.selected} ${noun} ready`,
|
|
342
|
+
caveat: 'No failure detected on the pods - they may still be starting (a fresh rollout or pending pods). Re-run once they settle.',
|
|
343
|
+
}
|
|
344
|
+
}
|
|
345
|
+
// Lead with the pod diagnosis (bd.cause) - it names the EXACT failing container
|
|
346
|
+
// and reason (a crashing sidecar, a stalled init container, a failing readiness
|
|
347
|
+
// probe), so the headline can't assert "the backend app" when a side component is
|
|
348
|
+
// the real culprit. Fall back to the generic reason only when the pod state was
|
|
349
|
+
// unclassifiable (cause empty).
|
|
350
|
+
const tag = bd.raw && !bd.raw.includes('ready') ? ` (${bd.raw})` : ''
|
|
351
|
+
const lead = bd.cause || `${bd.reason}${tag}`
|
|
352
|
+
return {
|
|
353
|
+
icon: '✗',
|
|
354
|
+
tone: 'unhealthy',
|
|
355
|
+
text: `No healthy backend - ${lead}`,
|
|
356
|
+
caveat: `The Service is wired to its workload (selector matches), so the break is in the pods, not the network. ${bd.ready}/${bd.selected} ${noun} ready.`,
|
|
357
|
+
detail: lead === bd.cause ? undefined : bd.cause,
|
|
358
|
+
}
|
|
359
|
+
}
|
|
360
|
+
|
|
361
|
+
// Ingress/Gateway SUBJECT: the FRONT DOOR (the external host a user hits) is the
|
|
362
|
+
// journey under test - state it explicitly, instead of letting backend health speak
|
|
363
|
+
// for a path it never proves. Honesty cuts both ways: a host actually reached from
|
|
364
|
+
// outside is a first-class ✓, while a host that couldn't be verified from here (a
|
|
365
|
+
// skip, e.g. split-horizon DNS) is named - never condemned, never papered over.
|
|
366
|
+
const fd = frontDoorStatus(trace)
|
|
367
|
+
// backendUntested distinguishes "no backend route was probed" from "a backend
|
|
368
|
+
// route failed" - asserting "backend has issues" for an UNtested backend would
|
|
369
|
+
// false-condemn it.
|
|
370
|
+
const backendUntested = okCount === 0 && realUnreach.length === 0 && !any5xx
|
|
371
|
+
// backendReachable folds only the directly-failed signals (realUnreach, 5xx); it
|
|
372
|
+
// ignores routes that were skipped / not-tested / only-indirect-unreachable. A
|
|
373
|
+
// passed sibling can therefore mark the backend "reachable" while another backend
|
|
374
|
+
// route was never confirmed. Surface that gap as a caveat (the front door stays a
|
|
375
|
+
// genuine ✓), instead of a silent green.
|
|
376
|
+
const backendGap = (trace.coverage?.skipped ?? 0) > 0 || (trace.notTested?.length ?? 0) > 0 || anyIndirectUnreach
|
|
377
|
+
const backendCaveat = backendReachable
|
|
378
|
+
? (backendGap ? 'some backend routes weren’t confirmed - see the path below' : undefined)
|
|
379
|
+
: backendUntested ? 'backend not separately tested - see the path below' : 'backend has issues - see the path below'
|
|
380
|
+
if (fd.state === 'reached') {
|
|
381
|
+
// A 2xx to the front door (typically '/') does NOT prove a backend route that
|
|
382
|
+
// actually FAILED. When a backend route is unreachable (not merely untested),
|
|
383
|
+
// downgrade off the green ✓ - the path is partly broken, not confirmed healthy.
|
|
384
|
+
const backendFailed = !backendReachable && !backendUntested
|
|
385
|
+
return {
|
|
386
|
+
icon: backendFailed ? '⚠' : '✓',
|
|
387
|
+
tone: backendFailed ? 'degraded' : 'healthy',
|
|
388
|
+
text: `Front door confirmed from outside${fd.detail ? ` - ${fd.detail}` : ''}`,
|
|
389
|
+
caveat: backendCaveat,
|
|
390
|
+
}
|
|
391
|
+
}
|
|
392
|
+
if (fd.state === 'partial') {
|
|
393
|
+
// One host was confirmed (2xx) but another front-door host failed - don't
|
|
394
|
+
// claim a blanket "confirmed from outside" that drops the failed host.
|
|
395
|
+
return { icon: '⚠', tone: 'degraded', text: `Front door partially reachable - one host confirmed, another failed${fd.detail ? ` (${fd.detail})` : ''}`, caveat: backendCaveat }
|
|
396
|
+
}
|
|
397
|
+
if (fd.state === 'partial-untested') {
|
|
398
|
+
// One host confirmed (2xx) but another DECLARED host's probes all skipped from
|
|
399
|
+
// this vantage (split-horizon / internal host) - never tested. Stay neutral and
|
|
400
|
+
// name the gap; a green ✓ here would claim "confirmed" over an untested host.
|
|
401
|
+
return { icon: '', tone: 'neutral', text: `Front door: one host confirmed, another not tested from here${fd.host ? ` - ${fd.host}` : ''}`, caveat: fd.detail ? `another declared host wasn’t tested: ${fd.detail}` : 'another declared host wasn’t tested from this vantage', detail: backendUntested ? undefined : backendCaveat }
|
|
402
|
+
}
|
|
403
|
+
if (fd.state === 'partial-reached') {
|
|
404
|
+
// One host confirmed (2xx) but another DECLARED host only returned a 3xx/4xx -
|
|
405
|
+
// reached at the HTTP layer, not verified end-to-end. Stay neutral and name the
|
|
406
|
+
// gap; a green ✓ here would claim "confirmed" over a host the single-host path
|
|
407
|
+
// itself refuses to confirm (it routes the identical 3xx/4xx to http-reached).
|
|
408
|
+
return { icon: '', tone: 'neutral', text: `Front door: one host confirmed, another reached but not verified${fd.host ? ` - ${fd.host}` : ''}`, caveat: fd.detail ? `another declared host returned ${fd.detail} (3xx/4xx) - reachable, but not verified end-to-end` : 'another declared host returned an HTTP 3xx/4xx - reachable, but not verified end-to-end', detail: backendUntested ? undefined : backendCaveat }
|
|
409
|
+
}
|
|
410
|
+
if (fd.state === 'server-error') {
|
|
411
|
+
// This fires only for a front-door 502/504 (HTTP-layer degraded). An app's own
|
|
412
|
+
// 5xx now reads as "reached" (classifyHTTPStatus), so a degraded HTTP front door
|
|
413
|
+
// means the front door answered but could NOT reach its backend - a network-path
|
|
414
|
+
// fault, not an app error. Do not send the operator to app logs; fd.detail
|
|
415
|
+
// ("the front door couldn't reach the backend") names the hop.
|
|
416
|
+
return { icon: '⚠', tone: 'degraded', text: `Front door reached but couldn't reach the backend${fd.host ? `, ${fd.host}` : ''}${fd.detail ? ` (${fd.detail})` : ''}` }
|
|
417
|
+
}
|
|
418
|
+
if (fd.state === 'http-reached') {
|
|
419
|
+
// The front door RETURNED an HTTP response (3xx/4xx - e.g. a 301 redirect to https,
|
|
420
|
+
// or a 404 at '/' on a path-routed app), so it WAS reached at the HTTP layer. Not
|
|
421
|
+
// verified end-to-end (the route behind it wasn't confirmed) - but never claim the
|
|
422
|
+
// dishonest "transport only / the HTTP response wasn’t checked".
|
|
423
|
+
return { icon: '', tone: 'neutral', text: `Front door reached - HTTP responded, route not verified${fd.host ? ` - ${fd.host}` : ''}`, caveat: fd.detail ? `the front door returned ${fd.detail} (3xx/4xx) - reachable, but the route wasn’t verified end-to-end` : 'the front door returned an HTTP 3xx/4xx - reachable, but the route wasn’t verified end-to-end', detail: backendUntested ? undefined : backendCaveat }
|
|
424
|
+
}
|
|
425
|
+
if (fd.state === 'transport') {
|
|
426
|
+
// TCP/TLS connected but no HTTP response was verified - never claim "confirmed".
|
|
427
|
+
return { icon: '', tone: 'neutral', text: `Front door reachable (transport only) - not verified${fd.host ? ` - ${fd.host}` : ''}`, caveat: 'a TCP/TLS connect succeeded but the HTTP response wasn’t checked', detail: backendUntested ? undefined : backendCaveat }
|
|
428
|
+
}
|
|
429
|
+
if (fd.state === 'failed') {
|
|
430
|
+
return { icon: '✗', tone: 'unhealthy', text: `Front door unreachable${fd.host ? ` - ${fd.host}` : ''}${fd.detail ? ` (${fd.detail})` : ''}` }
|
|
431
|
+
}
|
|
432
|
+
if (fd.state === 'skipped' && backendReachable) {
|
|
433
|
+
// Two facts a developer must connect into ONE conclusion: the backend is up,
|
|
434
|
+
// but the Ingress entry path itself was NOT tested (its host didn't resolve
|
|
435
|
+
// where Radar ran). Headline states the relationship ("reachable - but…");
|
|
436
|
+
// the caveat says how the backend WAS reached (so "via API server" reads as an
|
|
437
|
+
// implication, not jargon) and gives the action that actually works for a
|
|
438
|
+
// host that only resolves elsewhere - test it from where it resolves, NOT the
|
|
439
|
+
// self-contradicting "from outside" when it may be cluster-internal.
|
|
440
|
+
const host = fd.host || 'the host'
|
|
441
|
+
// "Confirmed" requires a real-traffic route that actually VERIFIED (2xx), not
|
|
442
|
+
// just any real-confidence reach: a real in-cluster dial that got 3xx/4xx is
|
|
443
|
+
// reached, not verified end-to-end, so it must not claim confirmation.
|
|
444
|
+
const realVerified = routes.some((r) => r.confidence === 'real' && r.outcome === 'verified')
|
|
445
|
+
const how = realVerified
|
|
446
|
+
? 'confirmed by real in-cluster traffic'
|
|
447
|
+
: allReal
|
|
448
|
+
? 'reached over real in-cluster traffic but not verified end-to-end'
|
|
449
|
+
: 'reached through the API server, not the real ingress path'
|
|
450
|
+
// The skip has two distinct causes: DNS didn't resolve here (split-horizon),
|
|
451
|
+
// OR DNS resolved but TCP/HTTP were skipped (e.g. the ingress IP isn't routable
|
|
452
|
+
// from a laptop). Only say "didn't resolve" when the DNS layer was the skip -
|
|
453
|
+
// otherwise describe the actual gap and give the right next action.
|
|
454
|
+
const gap = fd.dnsSkip
|
|
455
|
+
? `${host} didn’t resolve from where Radar ran`
|
|
456
|
+
: `the entry path to ${host} couldn’t be reached from where Radar ran`
|
|
457
|
+
const fix = fd.dnsSkip
|
|
458
|
+
? `request ${host} from a client that can resolve it (inside the cluster if it’s internal)`
|
|
459
|
+
: `request ${host} from a client on a network that can reach the ingress (inside the cluster if it’s internal)`
|
|
460
|
+
return {
|
|
461
|
+
icon: '',
|
|
462
|
+
tone: 'neutral',
|
|
463
|
+
// Don't assert a flat "reachable" on API-server-only evidence - name HOW the
|
|
464
|
+
// backend was reached so the word doesn't overclaim the real data path.
|
|
465
|
+
text: allReal
|
|
466
|
+
? 'Backend reachable - but the Ingress entry point wasn’t tested'
|
|
467
|
+
: `Backend answered ${VIA_API_SERVER} - but the Ingress entry point wasn’t tested`,
|
|
468
|
+
// One short line stays on screen; the full why + fix lives in the tooltip.
|
|
469
|
+
caveat: `${gap} - only the backend was checked.`,
|
|
470
|
+
detail: `${gap} (it may be internal to the cluster), so requests coming in through the Ingress weren’t checked - only the backend, ${how}. To confirm the entry works, ${fix}.`,
|
|
471
|
+
}
|
|
472
|
+
}
|
|
473
|
+
|
|
474
|
+
// Benign-only - every probed route is an intentional scale-to-0 (dormant by
|
|
475
|
+
// design, not an outage). 'unreach' already excludes benign, so without this it
|
|
476
|
+
// would fall through to the neutral "Config looks healthy" fallback and
|
|
477
|
+
// contradict TraceSummary's amber 'No running backends'. Mirror that here.
|
|
478
|
+
if (total > 0 && okCount === 0 && unreach.length === 0 && !any5xx && routes.some((r) => r.benign)) {
|
|
479
|
+
return {
|
|
480
|
+
icon: '⚠',
|
|
481
|
+
tone: 'degraded',
|
|
482
|
+
text: trace.headline || 'No running backends (scaled to 0)',
|
|
483
|
+
caveat: 'the backing workload is intentionally scaled to 0 - nothing to reach',
|
|
484
|
+
}
|
|
485
|
+
}
|
|
486
|
+
|
|
487
|
+
// Broken - nothing reachable. A 5xx route DID reach a backend (it answered),
|
|
488
|
+
// so a mix of unreachable + server-error must degrade (⚠), not hard-condemn.
|
|
489
|
+
if (!any5xx && (trace.verdict === 'broken' || (unreach.length > 0 && okCount === 0))) {
|
|
490
|
+
// An apiserver-proxy-only (indirect) failure must NEVER hard-condemn the real
|
|
491
|
+
// path - the proxy is "indirect, never the headline". When every unreachable
|
|
492
|
+
// route is indirect AND the verdict isn't independently broken (a real/static
|
|
493
|
+
// structural break), fall to neutral/unknown keeping the honest headline the
|
|
494
|
+
// Go singleRouteHeadline already wrote, instead of overriding it with red.
|
|
495
|
+
const onlyIndirectUnreach = unreach.length > 0 && unreach.every((r) => r.confidence === 'indirect')
|
|
496
|
+
if (onlyIndirectUnreach && trace.verdict !== 'broken') {
|
|
497
|
+
return { icon: '', tone: 'unknown', text: trace.headline || 'Unreachable via API server - real path not confirmed' }
|
|
498
|
+
}
|
|
499
|
+
// Prefer the verdict reason for the headline: on an entry-path break the coverage
|
|
500
|
+
// headline can be optimistic ("Reachable - verified" from the healthy backend's
|
|
501
|
+
// own port probes) while the verdict is broken from the failed upstream entries -
|
|
502
|
+
// leading with trace.headline would put that green text beside the red ✗.
|
|
503
|
+
return { icon: '✗', tone: 'unhealthy', text: trace.reason || trace.headline || 'Unreachable - traffic can’t reach the backend' }
|
|
504
|
+
}
|
|
505
|
+
// Partial - some routes reachable, some REALLY unreachable (a real problem → ⚠
|
|
506
|
+
// honest). A proxy-only (indirect) unreachable is excluded - its real path was
|
|
507
|
+
// never tested, so it isn't a confirmed failure to condemn here.
|
|
508
|
+
if (realUnreach.length > 0 && okCount > 0) {
|
|
509
|
+
return { icon: '⚠', tone: 'degraded', text: trace.headline || `${okCount} of ${total} routes reachable · ${realUnreach.length} unreachable` }
|
|
510
|
+
}
|
|
511
|
+
// A degraded route is named by the layer that failed: "upstream" is a front-door
|
|
512
|
+
// 502/504 (HTTP was reached, but the gateway couldn't reach its backend), "tls" is
|
|
513
|
+
// a certificate failure. It is NEVER an app's own 5xx (that reads "reached"). Name
|
|
514
|
+
// the exact layer; prefer the Go headline when it already wrote the specific fault.
|
|
515
|
+
if (any5xx) {
|
|
516
|
+
const degraded = routes.filter((r) => r.outcome === 'server-error')
|
|
517
|
+
const tls = degraded.some((r) => r.failedLayer === 'tls')
|
|
518
|
+
const upstream = degraded.some((r) => r.failedLayer === 'upstream')
|
|
519
|
+
const fallback =
|
|
520
|
+
tls && !upstream ? 'Reached, but the TLS handshake failed (certificate problem) - check the certificate.'
|
|
521
|
+
: upstream && !tls ? "Reached the front door, but it couldn't reach its backend (502/504) - the upstream is unavailable."
|
|
522
|
+
: 'Reached, but a route is degraded (a TLS certificate or a front-door 502/504) - see the detail.'
|
|
523
|
+
return { icon: '⚠', tone: 'degraded', text: trace.headline || fallback }
|
|
524
|
+
}
|
|
525
|
+
// Reachable. A proxy-only reach is NEUTRAL, not a green ✓ - the green check makes a
|
|
526
|
+
// non-expert read "it works, done" and skip the real limitation. It is still NEVER a
|
|
527
|
+
// ⚠ alarm (the path did reach). Only real-traffic verified earns the green ✓.
|
|
528
|
+
if (okCount > 0) {
|
|
529
|
+
// Gate the green ✓ on POSITIVE proof: any ok route that isn't provably
|
|
530
|
+
// real-confidence (indirect OR undefined) must read neutral with a caveat -
|
|
531
|
+
// undefined confidence is not real-traffic confirmation.
|
|
532
|
+
if (!allReal) {
|
|
533
|
+
// An ExternalName was dialed directly by Radar - say that, not "via the API
|
|
534
|
+
// proxy" (which would contradict the matrix's "direct from your machine").
|
|
535
|
+
if (ext) {
|
|
536
|
+
return {
|
|
537
|
+
icon: '',
|
|
538
|
+
tone: 'neutral',
|
|
539
|
+
text: `Reached ${ext} directly from Radar`,
|
|
540
|
+
caveat: 'in-cluster clients may resolve or route it differently - not confirmed from inside the cluster',
|
|
541
|
+
}
|
|
542
|
+
}
|
|
543
|
+
return {
|
|
544
|
+
icon: '',
|
|
545
|
+
tone: 'neutral',
|
|
546
|
+
// Lead with the GAP, not the mechanism: the backend answered, but only on
|
|
547
|
+
// the control-plane path - the real network path isn't confirmed.
|
|
548
|
+
text: `Backend answered ${VIA_API_SERVER} - the real network path isn’t confirmed`,
|
|
549
|
+
caveat: 'control-plane path: bypasses kube-proxy + NetworkPolicy - run the in-cluster test to confirm the real data path',
|
|
550
|
+
}
|
|
551
|
+
}
|
|
552
|
+
// allReal is true here, but 'real' confidence only means the LIVE path was
|
|
553
|
+
// exercised - a route that merely 'reached' (transport-only TCP/TLS, or an
|
|
554
|
+
// HTTP 3xx/4xx) is NOT verified end-to-end. Green-checking it would overclaim,
|
|
555
|
+
// contradicting coverageBannerTone's realPass and reachVerdict's own
|
|
556
|
+
// indirect-reached neutral. Require a verified real route for the green ✓.
|
|
557
|
+
const realVerified = routes.some((r) => r.confidence === 'real' && r.outcome === 'verified')
|
|
558
|
+
if (!realVerified) {
|
|
559
|
+
return { icon: '', tone: 'neutral', text: trace.headline || 'Reached - route not verified', caveat: 'the route responded but wasn’t verified end-to-end - see the path below' }
|
|
560
|
+
}
|
|
561
|
+
// Real-traffic confirmed. But if coverage is partial (routes still not tested),
|
|
562
|
+
// a green tone overclaims full success - down-rank to neutral while keeping the
|
|
563
|
+
// honest (coverage-aware) headline text.
|
|
564
|
+
// A proxy-only (indirect) unreachable route is an UNTESTED real path, not a
|
|
565
|
+
// pass - count it as a coverage gap so a sibling real pass doesn't green-light
|
|
566
|
+
// the whole subject over a route whose real path was never confirmed.
|
|
567
|
+
const hasGap = (trace.coverage?.skipped ?? 0) > 0 || (trace.notTested?.length ?? 0) > 0 || anyIndirectUnreach
|
|
568
|
+
if (hasGap) {
|
|
569
|
+
return { icon: '', tone: 'neutral', text: trace.headline || 'Reachable - some routes not yet tested', caveat: 'some routes weren’t tested - see the path below' }
|
|
570
|
+
}
|
|
571
|
+
return { icon: '✓', tone: 'healthy', text: trace.headline || 'Reachable - confirmed from inside the cluster' }
|
|
572
|
+
}
|
|
573
|
+
// Unknown - couldn’t determine (RBAC, cache cold, API absent). Not clean, not broken.
|
|
574
|
+
if (trace.verdict === 'unknown') {
|
|
575
|
+
return { icon: '', tone: 'unknown', text: trace.reason || trace.headline || 'Couldn’t determine reachability' }
|
|
576
|
+
}
|
|
577
|
+
// Not probed yet. Having passed the broken/degraded checks above, the STATIC
|
|
578
|
+
// config is valid - a positive state, not an empty "not tested". The full
|
|
579
|
+
// static path + config render below regardless; this is just the headline.
|
|
580
|
+
// Never surface the backend's "Configuration only - not yet tested" string.
|
|
581
|
+
if ((trace.routes ?? []).length === 0) {
|
|
582
|
+
// Un-probed but the STATIC verdict found a config warning (e.g. targetPort
|
|
583
|
+
// mismatch) - that is a real ⚠, never "✓ valid". The cause shows on the path below.
|
|
584
|
+
if (trace.verdict === 'degraded') {
|
|
585
|
+
// Prefer the worst finding's PLAIN message over diagnosis.summary (which can
|
|
586
|
+
// be the forensic cause, e.g. "No IngressClass resolves…"). Mirrors the
|
|
587
|
+
// drawer (TraceSummary) so the banner - the loudest surface - never leaks
|
|
588
|
+
// the jargon the message was written to replace.
|
|
589
|
+
return { icon: '⚠', tone: 'degraded', text: worstFindingMessage(trace) || trace.diagnosis?.summary || trace.headline || 'Configuration issue - see the path below' }
|
|
590
|
+
}
|
|
591
|
+
// The probe RAN but produced no testable route (every route skipped from this
|
|
592
|
+
// vantage, or nothing probeable). Acknowledge it - clicking Run must not read
|
|
593
|
+
// as a no-op - instead of telling the user to "run the test" they already ran.
|
|
594
|
+
if (probed) {
|
|
595
|
+
return { icon: '', tone: 'neutral', text: 'Tested - no route could be actively probed from here', caveat: 'the config looks valid; nothing was reachable to test from this vantage' }
|
|
596
|
+
}
|
|
597
|
+
return { icon: '', tone: 'neutral', text: 'Config valid - not yet tested', caveat: 'run the test to check live reachability' }
|
|
598
|
+
}
|
|
599
|
+
// Probed, but nothing conclusive (every route skipped from this vantage).
|
|
600
|
+
return { icon: '', tone: 'neutral', text: 'Config looks healthy - not confirmed from this vantage' }
|
|
601
|
+
}
|