@skyhook-io/k8s-ui 1.10.4 → 1.11.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/package.json +5 -5
- package/src/components/applications/ApplicationDetail.test.tsx +46 -0
- package/src/components/applications/ApplicationDetail.tsx +14 -15
- package/src/components/issues/IssuesView.tsx +9 -0
- package/src/components/issues/diagnostic.ts +3 -0
- package/src/components/issues/issues.test.ts +11 -0
- package/src/components/issues/severity.ts +3 -0
- package/src/components/issues/types.ts +3 -0
- package/src/components/resources/ResourcesSidebar.tsx +1 -1
- package/src/components/resources/ResourcesView.tsx +259 -30
- package/src/components/resources/index.ts +2 -0
- package/src/components/resources/kyverno-cell-gating.test.ts +93 -0
- package/src/components/resources/kyverno-modern-posture.test.ts +290 -0
- package/src/components/resources/renderers/CNPGClusterRenderer.test.tsx +85 -0
- package/src/components/resources/renderers/CNPGClusterRenderer.tsx +109 -9
- package/src/components/resources/renderers/CNPGPoolerRenderer.tsx +8 -4
- package/src/components/resources/renderers/EventRenderer.test.tsx +23 -0
- package/src/components/resources/renderers/EventRenderer.tsx +10 -18
- package/src/components/resources/renderers/KyvernoCELPolicyRenderers.tsx +317 -0
- package/src/components/resources/renderers/KyvernoExceptionRenderers.tsx +264 -0
- package/src/components/resources/renderers/KyvernoPolicyShared.tsx +212 -0
- package/src/components/resources/renderers/VeleroBSLRenderer.tsx +2 -4
- package/src/components/resources/renderers/VeleroBackupRenderer.tsx +32 -9
- package/src/components/resources/renderers/VeleroRestoreRenderer.tsx +30 -9
- package/src/components/resources/renderers/VeleroScheduleRenderer.tsx +12 -6
- package/src/components/resources/renderers/badge-no-handrolled.test.tsx +1 -1
- package/src/components/resources/renderers/cnpg-cells.tsx +37 -2
- package/src/components/resources/renderers/index.ts +4 -0
- package/src/components/resources/renderers/kyverno-modern-cells.tsx +152 -0
- package/src/components/resources/renderers/velero-cells.tsx +128 -9
- package/src/components/resources/renderers/velero-phase-recovery.test.tsx +79 -0
- package/src/components/resources/resource-utils-cnpg.golden.test.ts +112 -0
- package/src/components/resources/resource-utils-cnpg.test.ts +527 -1
- package/src/components/resources/resource-utils-cnpg.ts +373 -56
- package/src/components/resources/resource-utils-kyverno-exceptions.ts +182 -0
- package/src/components/resources/resource-utils-kyverno-modern.ts +588 -0
- package/src/components/resources/resource-utils-velero.test.ts +237 -0
- package/src/components/resources/resource-utils-velero.ts +214 -22
- package/src/components/resources/resource-utils.ts +70 -3
- package/src/components/shared/ResourceActionsBar.tsx +3 -1
- package/src/components/shared/ResourceRendererDispatch.test.tsx +119 -1
- package/src/components/shared/ResourceRendererDispatch.tsx +116 -22
- package/src/components/topology/K8sResourceNode.tsx +11 -0
- package/src/components/topology/TopologyGraph.tsx +85 -5
- package/src/components/trace/ReachabilityGraph.tsx +517 -0
- package/src/components/trace/ReachabilityView.tsx +1173 -0
- package/src/components/trace/TracePanel.test.ts +172 -0
- package/src/components/trace/TracePanel.tsx +558 -0
- package/src/components/trace/TraceSummary.test.ts +213 -0
- package/src/components/trace/TraceSummary.tsx +168 -0
- package/src/components/trace/inClusterSummary.test.ts +98 -0
- package/src/components/trace/inClusterSummary.ts +81 -0
- package/src/components/trace/index.ts +19 -0
- package/src/components/trace/podReach.test.ts +71 -0
- package/src/components/trace/podReach.ts +37 -0
- package/src/components/trace/probe-display.test.ts +107 -0
- package/src/components/trace/probe-display.ts +37 -0
- package/src/components/trace/problemRows.test.ts +107 -0
- package/src/components/trace/reachFixtures.ts +261 -0
- package/src/components/trace/reachGraphModel.test.ts +1423 -0
- package/src/components/trace/reachGraphModel.ts +1521 -0
- package/src/components/trace/reachInspector.test.ts +892 -0
- package/src/components/trace/reachInspector.ts +797 -0
- package/src/components/trace/reachMarks.test.ts +698 -0
- package/src/components/trace/reachMarks.ts +623 -0
- package/src/components/trace/reachOrigins.test.ts +392 -0
- package/src/components/trace/reachOrigins.ts +449 -0
- package/src/components/trace/reachVerdict.test.ts +537 -0
- package/src/components/trace/reachVerdict.ts +601 -0
- package/src/components/trace/traceFingerprint.test.ts +214 -0
- package/src/components/trace/traceFingerprint.ts +122 -0
- package/src/components/trace/traceToSubgraph.test.ts +623 -0
- package/src/components/trace/traceToSubgraph.ts +463 -0
- package/src/components/trace/types.ts +414 -0
- package/src/components/ui/Badge.tsx +36 -0
- package/src/components/ui/ForceDeleteConfirmDialog.tsx +12 -1
- package/src/components/ui/InClusterConsentDialog.test.tsx +83 -0
- package/src/components/ui/InClusterConsentDialog.tsx +141 -0
- package/src/components/ui/index.ts +1 -0
- package/src/components/workload/WorkloadView.tsx +58 -5
- package/src/components/workload/index.ts +1 -1
- package/src/components/workload/reachabilityTab.test.ts +47 -0
- package/src/index.ts +5 -0
- package/src/theme/components.css +64 -0
- package/src/theme/variables.css +5 -0
- package/src/types/core.ts +8 -0
- package/src/utils/api-resources.ts +2 -0
- package/src/utils/application-history.test.ts +1 -1
- package/src/utils/applications.test.ts +63 -1
- package/src/utils/applications.ts +27 -0
- package/src/utils/inClusterConsent.test.ts +109 -0
- package/src/utils/inClusterConsent.ts +72 -0
- package/src/utils/index.ts +1 -0
- package/src/utils/navigation.test.ts +61 -0
- package/src/utils/navigation.ts +11 -9
- package/src/utils/pluralize.test.ts +80 -1
- package/src/utils/pluralize.ts +34 -0
- package/src/utils/resource-icons.test.ts +33 -0
- package/src/utils/resource-icons.ts +65 -2
|
@@ -0,0 +1,463 @@
|
|
|
1
|
+
import type { Trace, RouteResult, RouteOutcome, RouteConfidence, ResourceRef, Hop, Verdict } from './types'
|
|
2
|
+
import type { Topology, TopologyNode, TopologyEdge, HealthStatus } from '../../types/core'
|
|
3
|
+
|
|
4
|
+
// outcomeToStatus maps a route outcome onto the subject NODE's health. A node's
|
|
5
|
+
// color is the resource's OWN health - did it respond - NOT how we reached it. So a
|
|
6
|
+
// route that was reached (verified OR reached, real traffic OR only via the apiserver
|
|
7
|
+
// proxy) means the resource ANSWERED → healthy/green; the "real in-cluster traffic not
|
|
8
|
+
// confirmed" caveat is a property of the PATH and lives on the EDGE (dashed green) +
|
|
9
|
+
// the verdict, it never recolors the node (painting a healthy Service amber just
|
|
10
|
+
// because the vantage was the proxy is the misattribution this avoids). Only a real
|
|
11
|
+
// own-health problem downgrades it: a 5xx (the app erroring) or an unreachable break.
|
|
12
|
+
export function outcomeToStatus(outcome: RouteOutcome, confidence?: RouteConfidence): HealthStatus {
|
|
13
|
+
switch (outcome) {
|
|
14
|
+
case 'verified':
|
|
15
|
+
case 'reached':
|
|
16
|
+
return 'healthy'
|
|
17
|
+
case 'server-error':
|
|
18
|
+
return 'degraded'
|
|
19
|
+
case 'unreachable':
|
|
20
|
+
// A proxy-only (indirect) unreachable never condemns the real path - the
|
|
21
|
+
// real in-cluster path was never tested. Fail toward silence (neutral),
|
|
22
|
+
// mirroring reachVerdict's onlyIndirectUnreach guard, instead of red.
|
|
23
|
+
return confidence === 'indirect' ? 'unknown' : 'unhealthy'
|
|
24
|
+
case 'not-tested':
|
|
25
|
+
default:
|
|
26
|
+
return 'unknown'
|
|
27
|
+
}
|
|
28
|
+
}
|
|
29
|
+
|
|
30
|
+
function verdictToStatus(v: Verdict): HealthStatus {
|
|
31
|
+
switch (v) {
|
|
32
|
+
case 'healthy':
|
|
33
|
+
return 'healthy'
|
|
34
|
+
case 'degraded':
|
|
35
|
+
return 'degraded'
|
|
36
|
+
case 'broken':
|
|
37
|
+
return 'unhealthy'
|
|
38
|
+
default:
|
|
39
|
+
return 'unknown'
|
|
40
|
+
}
|
|
41
|
+
}
|
|
42
|
+
|
|
43
|
+
// A NetworkPolicy finding describes the PATH into a pod (who may reach it), not
|
|
44
|
+
// the pod's OWN health - the pod can be perfectly healthy and answering while a
|
|
45
|
+
// rule restricts the route. So policy findings never recolor the node, the same
|
|
46
|
+
// way a probe vantage doesn't. They surface in the hop detail + edge, not the
|
|
47
|
+
// node dot. (Same reasoning as the nodeOwnStatus doc below.)
|
|
48
|
+
const isOwnHealthFinding = (f: { code?: string }): boolean => !(f.code ?? '').startsWith('netpol:')
|
|
49
|
+
|
|
50
|
+
// A Service's "0/N selected pods ready" is a SYMPTOM of its backing pods' health,
|
|
51
|
+
// not the Service's OWN health - the Service object is correctly configured. The
|
|
52
|
+
// real fault (crashloop, image pull, OOM) is carried red on the downstream Pods
|
|
53
|
+
// node. So this finding caps the Service at DEGRADED (amber: "wired, no live
|
|
54
|
+
// backend") instead of the false red "the Service itself is broken". N > 0 here
|
|
55
|
+
// guarantees a Pods hop exists to carry the red, so the break is never hidden.
|
|
56
|
+
const isBackendSymptomFinding = (f: { code?: string }): boolean =>
|
|
57
|
+
f.code === 'svc:no-ready-endpoints' || /^problem:0\/[1-9]\d* selected pods ready$/.test(f.code ?? '')
|
|
58
|
+
|
|
59
|
+
// hopHasReadyEndpoints reports whether a backend hop still has ready endpoints
|
|
60
|
+
// (meta.ready > 0) - the Service serves traffic to the ready pods.
|
|
61
|
+
const hopHasReadyEndpoints = (hop: Hop): boolean => typeof hop.meta?.ready === 'number' && (hop.meta.ready as number) > 0
|
|
62
|
+
|
|
63
|
+
function findingStatus(hop: Hop): HealthStatus | undefined {
|
|
64
|
+
const f = (hop.findings ?? []).filter(isOwnHealthFinding)
|
|
65
|
+
const crit = f.filter((x) => x.severity === 'critical')
|
|
66
|
+
if (crit.length > 0) {
|
|
67
|
+
// A critical paints the node red ONLY when it's a real own-health break AND the
|
|
68
|
+
// backend has no ready endpoints. A backend that still serves (ready > 0 - one
|
|
69
|
+
// replica crashing while siblings serve) or a backend-symptom critical (0/N
|
|
70
|
+
// ready, whose red lives on the downstream Pods node) caps at degraded - never
|
|
71
|
+
// the false red "all down" (mirrors backend hopReachSeverity).
|
|
72
|
+
const hardRed = !hopHasReadyEndpoints(hop) && crit.some((x) => !isBackendSymptomFinding(x))
|
|
73
|
+
return hardRed ? 'unhealthy' : 'degraded'
|
|
74
|
+
}
|
|
75
|
+
if (f.some((x) => x.severity === 'warning')) return 'degraded'
|
|
76
|
+
return undefined
|
|
77
|
+
}
|
|
78
|
+
|
|
79
|
+
// worseStatus returns the more severe of two health statuses (unhealthy >
|
|
80
|
+
// degraded/alert > healthy > neutral/unknown).
|
|
81
|
+
function worseStatus(a: HealthStatus, b: HealthStatus): HealthStatus {
|
|
82
|
+
const rank: Record<string, number> = { unhealthy: 4, degraded: 3, alert: 3, healthy: 2, neutral: 1, unknown: 0 }
|
|
83
|
+
return (rank[b] ?? 0) > (rank[a] ?? 0) ? b : a
|
|
84
|
+
}
|
|
85
|
+
|
|
86
|
+
function outcomeRank(o: RouteOutcome): number {
|
|
87
|
+
const order: Record<RouteOutcome, number> = { unreachable: 0, 'server-error': 1, 'not-tested': 2, reached: 3, verified: 4 }
|
|
88
|
+
return order[o] ?? 5
|
|
89
|
+
}
|
|
90
|
+
|
|
91
|
+
const refId = (r: ResourceRef): string => `${r.kind}/${r.namespace ?? ''}/${r.name || 'pods'}`
|
|
92
|
+
|
|
93
|
+
const isRbac = (hop: Hop): boolean => (hop.findings ?? []).some((f) => f.code === 'rbac:cross-namespace-redacted')
|
|
94
|
+
// A NetworkPolicy finding is a PREDICTION, not a confirmed reachability break - it
|
|
95
|
+
// must never mark an edge unreachable and cascade-block downstream (the node logic
|
|
96
|
+
// already excludes netpol: via isOwnHealthFinding). Mirror that exclusion here.
|
|
97
|
+
// A backend that still has ready endpoints is reachable - a per-pod crash among
|
|
98
|
+
// serving replicas must NOT mark the edge unreachable and cascade-block downstream
|
|
99
|
+
// (mirrors hopReachSeverity). A 0-ready backend keeps its critical (real break).
|
|
100
|
+
const hasCritical = (hop: Hop): boolean =>
|
|
101
|
+
!hopHasReadyEndpoints(hop) && (hop.findings ?? []).some((f) => f.severity === 'critical' && isOwnHealthFinding(f))
|
|
102
|
+
|
|
103
|
+
// A NODE's status reflects the resource's OWN health (its findings/probes), NEVER a
|
|
104
|
+
// route's outcome - so a healthy router Ingress is not painted red just because one of
|
|
105
|
+
// its routes is broken (that per-route truth lives on the edges). Critically, the
|
|
106
|
+
// probe VANTAGE never recolors the node: a resource REACHED only via the apiserver
|
|
107
|
+
// proxy still responded, so its own health is fine - the "real in-cluster traffic not
|
|
108
|
+
// confirmed" caveat is a property of the PATH and lives on the edge (dashed green),
|
|
109
|
+
// not on the node. Only a real own-health signal downgrades it: a finding, a transport
|
|
110
|
+
// failure to the resource, or a 5xx (the app itself erroring).
|
|
111
|
+
function nodeOwnStatus(hop: Hop): HealthStatus {
|
|
112
|
+
if (isRbac(hop)) return 'unknown'
|
|
113
|
+
const finding = findingStatus(hop)
|
|
114
|
+
if (finding) return finding
|
|
115
|
+
const live = (hop.probes ?? []).filter((p) => !p.skipped)
|
|
116
|
+
if (live.length === 0) return 'unknown'
|
|
117
|
+
// A FAILED apiserver-proxy probe is indirect evidence: the proxy path itself may
|
|
118
|
+
// be what failed, and the real network path was never tested - so it must never
|
|
119
|
+
// paint the node red (mirrors the route-level isIndirectUnreach guard: a proxy
|
|
120
|
+
// 2xx is positive evidence the resource answered, a proxy failure is only a
|
|
121
|
+
// localization hint). Drop failed proxy probes before judging own health; if
|
|
122
|
+
// they were the ONLY live evidence, own health is untested → unknown.
|
|
123
|
+
const considered = live.filter((p) => !(p.path === 'apiserver' && (!p.ok || p.tone === 'unhealthy')))
|
|
124
|
+
if (considered.length === 0) return 'unknown'
|
|
125
|
+
const bad = considered.filter((p) => !p.ok || p.tone === 'unhealthy')
|
|
126
|
+
// A DNS success only proves the NAME resolves - it is not transport evidence
|
|
127
|
+
// that the resource answered, so it must never soften a real break: an
|
|
128
|
+
// ExternalName whose every port failed is unhealthy, not "partially healthy"
|
|
129
|
+
// off its DNS row. DNS FAILURES stay in `bad` - a name that doesn't resolve
|
|
130
|
+
// is a real reachability break.
|
|
131
|
+
const good = considered.filter((p) => p.ok && p.tone !== 'unhealthy' && p.layer !== 'dns')
|
|
132
|
+
// Partial own-health: a multi-port Service where one port answered and another
|
|
133
|
+
// didn't is DEGRADED (amber), not fully dead (red) - red over-claims "nothing
|
|
134
|
+
// works" when port 80 served HTTP 200.
|
|
135
|
+
if (bad.length > 0 && good.length > 0) return 'degraded'
|
|
136
|
+
if (bad.length > 0) return 'unhealthy'
|
|
137
|
+
if (considered.some((p) => p.tone === 'degraded')) return 'degraded' // app answered 5xx - a real own-health problem, not a vantage caveat
|
|
138
|
+
// DNS resolving a name to an IP is not own-health/reachability. When the only
|
|
139
|
+
// live probe is a DNS success (transport/HTTP skipped - ExternalName host, or a
|
|
140
|
+
// non-HTTP backend port), there's no evidence the resource was actually reached:
|
|
141
|
+
// fail toward silence (unknown), never paint it green 'healthy'.
|
|
142
|
+
if (!considered.some((p) => p.layer !== 'dns')) return 'unknown'
|
|
143
|
+
return 'healthy'
|
|
144
|
+
}
|
|
145
|
+
|
|
146
|
+
// An EDGE's reachOutcome = did traffic flow along THIS route/hop. A proxy-only
|
|
147
|
+
// "verified" reads 'reached' (green is fine; the node/verdict carry the not-real caveat).
|
|
148
|
+
function edgeReach(hop: Hop, route?: RouteResult): TopologyEdge['reachOutcome'] {
|
|
149
|
+
if (hasCritical(hop)) return 'unreachable'
|
|
150
|
+
if (route) {
|
|
151
|
+
switch (route.outcome) {
|
|
152
|
+
case 'verified':
|
|
153
|
+
return route.confidence === 'real' ? 'verified' : 'reached'
|
|
154
|
+
case 'reached':
|
|
155
|
+
case 'server-error':
|
|
156
|
+
// A 5xx REACHED the backend - traffic flowed; the app erroring is the
|
|
157
|
+
// node's own degraded health, not a broken edge. Green-dashed reached, not
|
|
158
|
+
// a red "route break" (which would falsely say traffic never arrived).
|
|
159
|
+
return 'reached'
|
|
160
|
+
case 'unreachable':
|
|
161
|
+
// A benign scale-to-0 is dormant by DESIGN, not a break. A proxy-only
|
|
162
|
+
// (indirect) unreachable never tested the real path. Neither must color
|
|
163
|
+
// the edge red or cascade-block the pods behind it (the node stays neutral
|
|
164
|
+
// via its own logic). Treat both as not-tested for edge purposes.
|
|
165
|
+
return route.benign || route.confidence === 'indirect' ? 'not-tested' : 'unreachable'
|
|
166
|
+
case 'not-tested':
|
|
167
|
+
return 'not-tested'
|
|
168
|
+
}
|
|
169
|
+
}
|
|
170
|
+
// No route on this hop (e.g. the Service→Pods edge): the edge answers "did any
|
|
171
|
+
// traffic get through" from this hop's own probes.
|
|
172
|
+
return probeReach(hop)
|
|
173
|
+
}
|
|
174
|
+
|
|
175
|
+
// probeReach classifies an edge purely from a hop's OWN probe evidence, ignoring
|
|
176
|
+
// findings. If ANY non-skipped TRANSPORT probe reached, the edge reached - a
|
|
177
|
+
// multi-port pod where port 80 served 200 but 9090 refused is REACHED (the
|
|
178
|
+
// per-port failure shows in the node detail), never a hard-blocked red edge. A
|
|
179
|
+
// DNS success is NOT traffic-flow evidence (resolving a name reaches nothing), so
|
|
180
|
+
// it can neither mark the edge reached nor mask every transport port failing; a
|
|
181
|
+
// DNS failure IS a real break. Findings are deliberately NOT consulted so a
|
|
182
|
+
// sibling route's break on the SAME entry never condemns this edge.
|
|
183
|
+
function probeReach(hop: Hop): TopologyEdge['reachOutcome'] {
|
|
184
|
+
const live = (hop.probes ?? []).filter((p) => !p.skipped)
|
|
185
|
+
if (live.length === 0) return 'not-tested'
|
|
186
|
+
const transport = live.filter((p) => p.layer !== 'dns')
|
|
187
|
+
if (transport.length === 0) {
|
|
188
|
+
// DNS-only evidence: a failed resolve is a real break; a resolve alone
|
|
189
|
+
// proves nothing about traffic - not-tested, mirroring nodeOwnStatus's
|
|
190
|
+
// DNS-only unknown.
|
|
191
|
+
return live.some((p) => !p.ok) ? 'unreachable' : 'not-tested'
|
|
192
|
+
}
|
|
193
|
+
if (transport.some((p) => p.ok)) return 'reached'
|
|
194
|
+
// Every transport failure came through the apiserver proxy - the real path was
|
|
195
|
+
// never tested (mirrors the indirect-unreachable route rule above): not-tested,
|
|
196
|
+
// never a red edge that would also cascade-block everything downstream.
|
|
197
|
+
return transport.every((p) => p.path === 'apiserver') ? 'not-tested' : 'unreachable'
|
|
198
|
+
}
|
|
199
|
+
|
|
200
|
+
// nodeSubtitle builds the diagram node's functional one-liner from the hop's
|
|
201
|
+
// declared config + live meta - the same facts the Tree shows as pills: the
|
|
202
|
+
// service port→targetPort mapping (or a Pod's container ports) and pod readiness.
|
|
203
|
+
// This is what gives the diagram parity with the Tree instead of a bare "ClusterIP".
|
|
204
|
+
// Falls back to the caller's status/route string when there's no port/readiness.
|
|
205
|
+
function nodeSubtitle(hop: Hop | undefined, fallback: string): string {
|
|
206
|
+
const cfg = hop?.config
|
|
207
|
+
const parts: string[] = []
|
|
208
|
+
const ports = cfg?.ports ?? []
|
|
209
|
+
if (ports.length > 0) {
|
|
210
|
+
const shown = ports.slice(0, 2).map((p) => `${p.port} → :${p.targetPort ?? p.port}`)
|
|
211
|
+
parts.push(shown.join(', ') + (ports.length > 2 ? ` +${ports.length - 2}` : ''))
|
|
212
|
+
}
|
|
213
|
+
const cps = cfg?.containerPorts ?? []
|
|
214
|
+
if (ports.length === 0 && cps.length > 0) {
|
|
215
|
+
parts.push(':' + [...new Set(cps.map((c) => c.port))].slice(0, 2).join(', :'))
|
|
216
|
+
}
|
|
217
|
+
const ready = hop?.meta?.['ready']
|
|
218
|
+
const selected = hop?.meta?.['selected']
|
|
219
|
+
if (typeof ready === 'number' && typeof selected === 'number') {
|
|
220
|
+
// For a publishNotReadyAddresses Service, meta.ready counts PUBLISHED
|
|
221
|
+
// endpoints (every selected pod, readiness not required) - "N/M ready"
|
|
222
|
+
// would claim crashlooping pods are ready. Say what the count actually is.
|
|
223
|
+
parts.push(
|
|
224
|
+
hop?.meta?.publishNotReadyAddresses
|
|
225
|
+
? `${ready} published (readiness not required)`
|
|
226
|
+
: `${ready}/${selected} ready`,
|
|
227
|
+
)
|
|
228
|
+
}
|
|
229
|
+
return parts.length ? parts.join(' · ') : fallback
|
|
230
|
+
}
|
|
231
|
+
|
|
232
|
+
/**
|
|
233
|
+
* traceToSubgraph projects a reachability Trace into the topology-graph shape so the
|
|
234
|
+
* existing TopologyGraph can render the path HONESTLY:
|
|
235
|
+
* - it BRANCHES by hop-edge semantics - backend Services hang off the entry, Pods off
|
|
236
|
+
* THEIR parent Service (a multi-backend Ingress fans out; N ingresses converge), never
|
|
237
|
+
* a linear chain;
|
|
238
|
+
* - per-route truth lives on the EDGE (reachOutcome) with a route LABEL - a router node
|
|
239
|
+
* stays neutral unless ALL its routes fail (no over-attribution);
|
|
240
|
+
* - nothing downstream of a real break is "reached": from any broken node/edge, descendant
|
|
241
|
+
* edges become 'blocked' (dashed) - never a green edge flowing past a break.
|
|
242
|
+
* Frontend-only; probe detail stays out of the graph (it lives in the detail/tree).
|
|
243
|
+
*/
|
|
244
|
+
// routeEdgeLabel renders the path on a route edge. By default it's the route's
|
|
245
|
+
// DECLARED path. When the operator overrides the tested path (probePath, e.g.
|
|
246
|
+
// "/test"), the edge echoes what was actually tested so the graph agrees with
|
|
247
|
+
// the banner/probe rows - but we NEVER overwrite the route identity: the label
|
|
248
|
+
// shows the tested path and the tooltip keeps the declared route, so we never
|
|
249
|
+
// imply the Ingress declares a rule it doesn't.
|
|
250
|
+
function routeEdgeLabel(declared: string | undefined, probePath?: string): { label?: string; labelTitle?: string } {
|
|
251
|
+
if (!declared) return {}
|
|
252
|
+
const ov = probePath?.trim()
|
|
253
|
+
// Active override only: a non-default path that differs from the declared one,
|
|
254
|
+
// on a route that actually has a path component.
|
|
255
|
+
if (!ov || ov === '/' || ov === declared || !declared.includes('/')) return { label: declared }
|
|
256
|
+
const i = declared.indexOf('/')
|
|
257
|
+
const tested = declared.slice(0, i) + ov // keep any host prefix, swap the path
|
|
258
|
+
return { label: tested, labelTitle: `declared ${declared} · testing ${ov}` }
|
|
259
|
+
}
|
|
260
|
+
|
|
261
|
+
export function traceToSubgraph(trace: Trace, probePath?: string): Topology {
|
|
262
|
+
const nodes: TopologyNode[] = []
|
|
263
|
+
const edges: TopologyEdge[] = []
|
|
264
|
+
const byId = new Map<string, TopologyNode>()
|
|
265
|
+
const add = (n: TopologyNode): string => {
|
|
266
|
+
if (!byId.has(n.id)) {
|
|
267
|
+
byId.set(n.id, n)
|
|
268
|
+
nodes.push(n)
|
|
269
|
+
}
|
|
270
|
+
return n.id
|
|
271
|
+
}
|
|
272
|
+
|
|
273
|
+
const routes = trace.routes ?? []
|
|
274
|
+
// Match a route to a backend by EXACT name, not substring - a substring match
|
|
275
|
+
// would pair backend "api" with a route targeting "api-v2" (wrong edge color/label).
|
|
276
|
+
// The target is "<name>" or "<name>:<port>", so compare the part before the colon.
|
|
277
|
+
// When the route carries a target namespace (cross-namespace Gateway API
|
|
278
|
+
// backendRef), it must also match the ref's namespace so two same-named backends
|
|
279
|
+
// across namespaces don't swap labels/outcomes.
|
|
280
|
+
const routeMatchesRef = (r: RouteResult, ref: ResourceRef): boolean => {
|
|
281
|
+
const nameMatch = (r.target ?? '').split(':')[0] === ref.name || r.route === ref.name
|
|
282
|
+
if (!nameMatch) return false
|
|
283
|
+
if (r.targetNamespace && ref.namespace && r.targetNamespace !== ref.namespace) return false
|
|
284
|
+
// An UNSET targetNamespace means the route targets the subject's namespace - so a
|
|
285
|
+
// hop in a DIFFERENT namespace (a cross-ns backendRef) must not match it, or a
|
|
286
|
+
// same-ns route's outcome would attach to a cross-ns same-named backend hop.
|
|
287
|
+
if (!r.targetNamespace && ref.namespace && trace.subject.namespace && ref.namespace !== trace.subject.namespace) return false
|
|
288
|
+
return true
|
|
289
|
+
}
|
|
290
|
+
// Pick the WORST matching route, not the first - an Ingress with two paths to the
|
|
291
|
+
// same Service (/web verified, /api unreachable) must not hide the failure behind
|
|
292
|
+
// a verified-first ordering on the single backend edge.
|
|
293
|
+
const routeFor = (ref: ResourceRef): RouteResult | undefined => {
|
|
294
|
+
const matches = routes.filter((r) => routeMatchesRef(r, ref))
|
|
295
|
+
if (matches.length === 0) return undefined
|
|
296
|
+
return matches.slice().sort((a, b) => outcomeRank(a.outcome) - outcomeRank(b.outcome))[0]
|
|
297
|
+
}
|
|
298
|
+
|
|
299
|
+
// Subject = the entry. A ROUTER (Ingress/Gateway) stays NEUTRAL unless ALL routes fail -
|
|
300
|
+
// its per-route truth is on the edges, not its node color.
|
|
301
|
+
const subjectId = refId(trace.subject)
|
|
302
|
+
const isRouter = trace.subject.kind === 'Ingress' || trace.subject.kind === 'Gateway'
|
|
303
|
+
// The reused Service node defaults its type subtitle to "ClusterIP" when none is
|
|
304
|
+
// passed - a lie for an ExternalName Service (a DNS alias, no ClusterIP). Surface
|
|
305
|
+
// the real type so the node reads "ExternalName" instead of the false "ClusterIP".
|
|
306
|
+
const isExternalName = (trace.downstream ?? []).some((h) => h.resource?.kind === 'ExternalName')
|
|
307
|
+
const okCount = routes.filter((r) => r.outcome === 'verified' || r.outcome === 'reached').length
|
|
308
|
+
// A proxy-only (indirect) unreachable is NOT a real failure - the real path was
|
|
309
|
+
// never tested. Exclude it everywhere a failure would redden/degrade the node,
|
|
310
|
+
// mirroring reachVerdict's onlyIndirectUnreach guard.
|
|
311
|
+
const isIndirectUnreach = (r: RouteResult): boolean => r.outcome === 'unreachable' && r.confidence === 'indirect'
|
|
312
|
+
const failCount = routes.filter((r) => (r.outcome === 'unreachable' && !isIndirectUnreach(r)) || r.outcome === 'server-error').length
|
|
313
|
+
// A ROUTER (Ingress/Gateway) node = the front door's OWN health. It goes red
|
|
314
|
+
// ONLY when the door genuinely can't pass traffic on every real route (all
|
|
315
|
+
// non-benign routes are hard unreachable). A server-error means traffic DID
|
|
316
|
+
// flow - the backend answered 5xx - so the front door works: that's amber, not
|
|
317
|
+
// red. A benign scale-to-zero route isn't a failure at all. Counting either as
|
|
318
|
+
// a hard fail would paint a working front door red.
|
|
319
|
+
const hardUnreach = routes.filter((r) => r.outcome === 'unreachable' && !r.benign && !isIndirectUnreach(r)).length
|
|
320
|
+
const nonBenignRoutes = routes.filter((r) => !(r.outcome === 'unreachable' && (r.benign || isIndirectUnreach(r)))).length
|
|
321
|
+
// "All failed" = every real (non-benign) route is a genuine unreachable. A
|
|
322
|
+
// server-error doesn't count (traffic flowed to a 5xx backend), nor does a
|
|
323
|
+
// benign scale-to-zero - so neither paints the front door red or the edge
|
|
324
|
+
// "unreachable".
|
|
325
|
+
const allHardFail = nonBenignRoutes > 0 && hardUnreach === nonBenignRoutes
|
|
326
|
+
const anySoftFail = hardUnreach > 0 || routes.some((r) => r.outcome === 'server-error')
|
|
327
|
+
const anyTrafficFlowed = okCount > 0 || routes.some((r) => r.outcome === 'server-error')
|
|
328
|
+
const worst = routes.slice().sort((a, b) => outcomeRank(a.outcome) - outcomeRank(b.outcome))[0]
|
|
329
|
+
// worstReal ignores benign scale-to-zero so a dormant Service backend doesn't
|
|
330
|
+
// read as a hard (red) failure - benign-only fails fall through to amber below.
|
|
331
|
+
const worstReal = routes
|
|
332
|
+
.filter((r) => !(r.outcome === 'unreachable' && (r.benign || isIndirectUnreach(r))))
|
|
333
|
+
.sort((a, b) => outcomeRank(a.outcome) - outcomeRank(b.outcome))[0]
|
|
334
|
+
// onlyIndirectUnreach: every failure is a proxy-only unreachable and nothing
|
|
335
|
+
// passed - the node must stay neutral/unknown, never red, mirroring
|
|
336
|
+
// reachVerdict.ts's onlyIndirectUnreach guard.
|
|
337
|
+
const onlyIndirectUnreach =
|
|
338
|
+
routes.some((r) => isIndirectUnreach(r) && !r.benign) &&
|
|
339
|
+
!routes.some(
|
|
340
|
+
(r) =>
|
|
341
|
+
(r.outcome === 'unreachable' && !r.benign && !isIndirectUnreach(r)) ||
|
|
342
|
+
r.outcome === 'server-error' ||
|
|
343
|
+
r.outcome === 'verified' ||
|
|
344
|
+
r.outcome === 'reached',
|
|
345
|
+
)
|
|
346
|
+
const subjectHop = (trace.downstream ?? []).find((h) => refId(h.resource) === subjectId)
|
|
347
|
+
const routeDerivedSubj: HealthStatus = onlyIndirectUnreach
|
|
348
|
+
? 'unknown'
|
|
349
|
+
: isRouter
|
|
350
|
+
? allHardFail ? 'unhealthy' : anySoftFail ? 'degraded' : 'unknown'
|
|
351
|
+
: okCount > 0 && failCount > 0 ? 'degraded' // partial - some ports/routes reach, some don't → amber, not a red "all dead"
|
|
352
|
+
: worstReal ? outcomeToStatus(worstReal.outcome, worstReal.confidence)
|
|
353
|
+
: worst ? 'degraded' // only benign (scaled-to-zero) failures remain → dormant, amber not red
|
|
354
|
+
: verdictToStatus(trace.verdict)
|
|
355
|
+
// Fold the subject hop's OWN health (its findings - e.g. ingress:no-controller /
|
|
356
|
+
// controller-unready - and a failed front-door probe) in as a FLOOR. Without
|
|
357
|
+
// this the router node stays neutral while reachVerdict headlines degraded /
|
|
358
|
+
// unreachable, under-representing a real front-door problem. Only a genuine own
|
|
359
|
+
// problem (degraded/unhealthy) floors; a healthy/unknown own status never
|
|
360
|
+
// downgrades the route-derived status.
|
|
361
|
+
const ownFloor = subjectHop ? nodeOwnStatus(subjectHop) : 'unknown'
|
|
362
|
+
const subjStatus: HealthStatus =
|
|
363
|
+
ownFloor === 'unhealthy' || ownFloor === 'degraded' ? worseStatus(routeDerivedSubj, ownFloor) : routeDerivedSubj
|
|
364
|
+
const subjType = isExternalName ? 'ExternalName' : subjectHop?.config?.serviceType
|
|
365
|
+
const subjFallback = routes.length > 1 ? `${okCount} of ${routes.length} routes reachable` : (worst?.evidence ?? trace.headline ?? '')
|
|
366
|
+
add({
|
|
367
|
+
id: subjectId,
|
|
368
|
+
kind: trace.subject.kind,
|
|
369
|
+
name: trace.subject.name,
|
|
370
|
+
status: subjStatus,
|
|
371
|
+
data: { ref: trace.subject, ...(subjType ? { type: subjType } : {}), subtitleOverride: nodeSubtitle(subjectHop, subjFallback), hop: subjectHop, routes },
|
|
372
|
+
})
|
|
373
|
+
|
|
374
|
+
const subjReach: TopologyEdge['reachOutcome'] = allHardFail ? 'unreachable' : anyTrafficFlowed ? 'reached' : 'not-tested'
|
|
375
|
+
// Entry paths converging on a Service subject. An upstream is an ENTRY that routes
|
|
376
|
+
// here - we test the SUBJECT's reachability, not each upstream's path individually. So
|
|
377
|
+
// the upstream NODE stays NEUTRAL: it is never colored by its own/other-route findings
|
|
378
|
+
// (that was the over-attribution bug - e.g. `multi` painted red on echo's view because
|
|
379
|
+
// of multi's UNRELATED /api→ghost break), and a neutral upstream also never triggers
|
|
380
|
+
// the frontier block-cascade through the shared subject to its pods. RBAC-redacted
|
|
381
|
+
// upstreams keep their distinct "no access" label.
|
|
382
|
+
for (const up of trace.upstreams ?? []) {
|
|
383
|
+
const id = add({ id: refId(up.resource), kind: up.resource.kind, name: up.resource.name, status: 'unknown', data: { ref: up.resource, subtitleOverride: isRbac(up) ? 'no access (RBAC)' : nodeSubtitle(up, ''), hop: up } })
|
|
384
|
+
// When this upstream WAS probed at its own front door, color the edge from its
|
|
385
|
+
// OWN probe evidence (findings ignored, so a sibling route's break never
|
|
386
|
+
// condemns it) - a genuinely-failed entry shows unreachable instead of
|
|
387
|
+
// inheriting the subject's reach. An unprobed upstream falls back to the
|
|
388
|
+
// subject's reach (the thing we actually tested).
|
|
389
|
+
const upProbed = (up.probes ?? []).some((p) => !p.skipped)
|
|
390
|
+
edges.push({ id: `e:${id}->${subjectId}`, source: id, target: subjectId, type: 'routes-to', reachOutcome: upProbed ? probeReach(up) : subjReach })
|
|
391
|
+
}
|
|
392
|
+
|
|
393
|
+
// Downstream - BRANCH: backends (X->Service) hang off the subject; Pods (Service->Pods)
|
|
394
|
+
// hang off their parent Service (the last-seen Service while walking).
|
|
395
|
+
let lastService = subjectId
|
|
396
|
+
for (const dn of trace.downstream ?? []) {
|
|
397
|
+
const baseId = refId(dn.resource)
|
|
398
|
+
if (baseId === subjectId) {
|
|
399
|
+
lastService = subjectId
|
|
400
|
+
continue
|
|
401
|
+
}
|
|
402
|
+
const isPods = dn.resource.kind === 'Pods' || /pods/i.test(dn.edge)
|
|
403
|
+
const parent = isPods ? lastService : subjectId
|
|
404
|
+
// An empty-name Pods hop's refId collapses to "Pods/<ns>/pods" for every
|
|
405
|
+
// backend; without parent-scoping, a multi-backend Ingress drops the second
|
|
406
|
+
// pod group (add() dedups by id) and both Service→Pods edges merge into one
|
|
407
|
+
// node. Scope the id to the owning Service so each backend keeps its own pods.
|
|
408
|
+
const id = isPods && !dn.resource.name ? `${parent}::pods` : baseId
|
|
409
|
+
const route = isPods ? undefined : routeFor(dn.resource)
|
|
410
|
+
const subtitleOverride = isRbac(dn) ? 'no access (RBAC)' : hasCritical(dn) ? (dn.findings[0]?.message ?? '') : nodeSubtitle(dn, '')
|
|
411
|
+
add({ id, kind: dn.resource.kind || 'Pods', name: dn.resource.name || 'Pods', status: nodeOwnStatus(dn), data: { ref: dn.resource, subtitleOverride, hop: dn, ...(route ? { routes: [route] } : {}) } })
|
|
412
|
+
const { label: edgeLabel, labelTitle } = routeEdgeLabel(route?.route, probePath)
|
|
413
|
+
edges.push({ id: `e:${parent}->${id}`, source: parent, target: id, type: 'routes-to', label: edgeLabel, labelTitle, reachOutcome: edgeReach(dn, route) })
|
|
414
|
+
if (!isPods) lastService = id
|
|
415
|
+
}
|
|
416
|
+
|
|
417
|
+
// Frontier rule: nothing downstream of a real break is "reached". From any broken NODE
|
|
418
|
+
// (a no-endpoints Service, a down Pod) or unreachable EDGE, mark descendant edges 'blocked'
|
|
419
|
+
// (dashed gray) and their nodes "blocked upstream" - never a green/reached edge past a break.
|
|
420
|
+
const out = new Map<string, TopologyEdge[]>()
|
|
421
|
+
for (const e of edges) {
|
|
422
|
+
const arr = out.get(e.source) ?? []
|
|
423
|
+
arr.push(e)
|
|
424
|
+
out.set(e.source, arr)
|
|
425
|
+
}
|
|
426
|
+
const blockFrom = (start: string) => {
|
|
427
|
+
const stack = [start]
|
|
428
|
+
const seen = new Set<string>()
|
|
429
|
+
while (stack.length) {
|
|
430
|
+
const cur = stack.pop()
|
|
431
|
+
if (cur === undefined || seen.has(cur)) continue
|
|
432
|
+
seen.add(cur)
|
|
433
|
+
for (const nx of out.get(cur) ?? []) {
|
|
434
|
+
if (nx.reachOutcome !== 'unreachable' && nx.reachOutcome !== 'blocked') {
|
|
435
|
+
nx.reachOutcome = 'blocked'
|
|
436
|
+
const n = byId.get(nx.target)
|
|
437
|
+
if (n && n.status !== 'unhealthy') {
|
|
438
|
+
n.status = 'unknown'
|
|
439
|
+
;(n.data as Record<string, unknown>).subtitleOverride = 'not reached - break upstream'
|
|
440
|
+
}
|
|
441
|
+
}
|
|
442
|
+
stack.push(nx.target)
|
|
443
|
+
}
|
|
444
|
+
}
|
|
445
|
+
}
|
|
446
|
+
// A backend reached by ANOTHER route/port must not be cascade-blocked just because
|
|
447
|
+
// the WORST matching route colored its edge unreachable (e.g. /web→svc:80 verified,
|
|
448
|
+
// /api→svc:9090 down - port 80 reached the service AND its pods). The worst-route is
|
|
449
|
+
// honest for the edge LABEL/color, but blocking descendants a sibling route reached
|
|
450
|
+
// would false-condemn them. Skip the cascade when the edge's backend has any reach.
|
|
451
|
+
const backendReachedByAnother = (targetId: string): boolean => {
|
|
452
|
+
const ref = (byId.get(targetId)?.data as { ref?: ResourceRef } | undefined)?.ref
|
|
453
|
+
if (!ref) return false
|
|
454
|
+
// server-error counts as backend-reach: a 5xx route DID reach the backend
|
|
455
|
+
// (edgeReach maps server-error to 'reached'), so it must not cascade-block the
|
|
456
|
+
// pods that route actually reached.
|
|
457
|
+
return routes.some((r) => routeMatchesRef(r, ref) && (r.outcome === 'verified' || r.outcome === 'reached' || r.outcome === 'server-error'))
|
|
458
|
+
}
|
|
459
|
+
for (const n of nodes) if (n.status === 'unhealthy') blockFrom(n.id)
|
|
460
|
+
for (const e of edges) if (e.reachOutcome === 'unreachable' && !backendReachedByAnother(e.target)) blockFrom(e.target)
|
|
461
|
+
|
|
462
|
+
return { nodes, edges }
|
|
463
|
+
}
|