@skyhook-io/k8s-ui 1.10.4 → 1.11.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/package.json +5 -5
- package/src/components/applications/ApplicationDetail.test.tsx +46 -0
- package/src/components/applications/ApplicationDetail.tsx +14 -15
- package/src/components/issues/IssuesView.tsx +9 -0
- package/src/components/issues/diagnostic.ts +3 -0
- package/src/components/issues/issues.test.ts +11 -0
- package/src/components/issues/severity.ts +3 -0
- package/src/components/issues/types.ts +3 -0
- package/src/components/resources/ResourcesSidebar.tsx +1 -1
- package/src/components/resources/ResourcesView.tsx +259 -30
- package/src/components/resources/index.ts +2 -0
- package/src/components/resources/kyverno-cell-gating.test.ts +93 -0
- package/src/components/resources/kyverno-modern-posture.test.ts +290 -0
- package/src/components/resources/renderers/CNPGClusterRenderer.test.tsx +85 -0
- package/src/components/resources/renderers/CNPGClusterRenderer.tsx +109 -9
- package/src/components/resources/renderers/CNPGPoolerRenderer.tsx +8 -4
- package/src/components/resources/renderers/EventRenderer.test.tsx +23 -0
- package/src/components/resources/renderers/EventRenderer.tsx +10 -18
- package/src/components/resources/renderers/KyvernoCELPolicyRenderers.tsx +317 -0
- package/src/components/resources/renderers/KyvernoExceptionRenderers.tsx +264 -0
- package/src/components/resources/renderers/KyvernoPolicyShared.tsx +212 -0
- package/src/components/resources/renderers/VeleroBSLRenderer.tsx +2 -4
- package/src/components/resources/renderers/VeleroBackupRenderer.tsx +32 -9
- package/src/components/resources/renderers/VeleroRestoreRenderer.tsx +30 -9
- package/src/components/resources/renderers/VeleroScheduleRenderer.tsx +12 -6
- package/src/components/resources/renderers/badge-no-handrolled.test.tsx +1 -1
- package/src/components/resources/renderers/cnpg-cells.tsx +37 -2
- package/src/components/resources/renderers/index.ts +4 -0
- package/src/components/resources/renderers/kyverno-modern-cells.tsx +152 -0
- package/src/components/resources/renderers/velero-cells.tsx +128 -9
- package/src/components/resources/renderers/velero-phase-recovery.test.tsx +79 -0
- package/src/components/resources/resource-utils-cnpg.golden.test.ts +112 -0
- package/src/components/resources/resource-utils-cnpg.test.ts +527 -1
- package/src/components/resources/resource-utils-cnpg.ts +373 -56
- package/src/components/resources/resource-utils-kyverno-exceptions.ts +182 -0
- package/src/components/resources/resource-utils-kyverno-modern.ts +588 -0
- package/src/components/resources/resource-utils-velero.test.ts +237 -0
- package/src/components/resources/resource-utils-velero.ts +214 -22
- package/src/components/resources/resource-utils.ts +70 -3
- package/src/components/shared/ResourceActionsBar.tsx +3 -1
- package/src/components/shared/ResourceRendererDispatch.test.tsx +119 -1
- package/src/components/shared/ResourceRendererDispatch.tsx +116 -22
- package/src/components/topology/K8sResourceNode.tsx +11 -0
- package/src/components/topology/TopologyGraph.tsx +85 -5
- package/src/components/trace/ReachabilityGraph.tsx +517 -0
- package/src/components/trace/ReachabilityView.tsx +1173 -0
- package/src/components/trace/TracePanel.test.ts +172 -0
- package/src/components/trace/TracePanel.tsx +558 -0
- package/src/components/trace/TraceSummary.test.ts +213 -0
- package/src/components/trace/TraceSummary.tsx +168 -0
- package/src/components/trace/inClusterSummary.test.ts +98 -0
- package/src/components/trace/inClusterSummary.ts +81 -0
- package/src/components/trace/index.ts +19 -0
- package/src/components/trace/podReach.test.ts +71 -0
- package/src/components/trace/podReach.ts +37 -0
- package/src/components/trace/probe-display.test.ts +107 -0
- package/src/components/trace/probe-display.ts +37 -0
- package/src/components/trace/problemRows.test.ts +107 -0
- package/src/components/trace/reachFixtures.ts +261 -0
- package/src/components/trace/reachGraphModel.test.ts +1423 -0
- package/src/components/trace/reachGraphModel.ts +1521 -0
- package/src/components/trace/reachInspector.test.ts +892 -0
- package/src/components/trace/reachInspector.ts +797 -0
- package/src/components/trace/reachMarks.test.ts +698 -0
- package/src/components/trace/reachMarks.ts +623 -0
- package/src/components/trace/reachOrigins.test.ts +392 -0
- package/src/components/trace/reachOrigins.ts +449 -0
- package/src/components/trace/reachVerdict.test.ts +537 -0
- package/src/components/trace/reachVerdict.ts +601 -0
- package/src/components/trace/traceFingerprint.test.ts +214 -0
- package/src/components/trace/traceFingerprint.ts +122 -0
- package/src/components/trace/traceToSubgraph.test.ts +623 -0
- package/src/components/trace/traceToSubgraph.ts +463 -0
- package/src/components/trace/types.ts +414 -0
- package/src/components/ui/Badge.tsx +36 -0
- package/src/components/ui/ForceDeleteConfirmDialog.tsx +12 -1
- package/src/components/ui/InClusterConsentDialog.test.tsx +83 -0
- package/src/components/ui/InClusterConsentDialog.tsx +141 -0
- package/src/components/ui/index.ts +1 -0
- package/src/components/workload/WorkloadView.tsx +58 -5
- package/src/components/workload/index.ts +1 -1
- package/src/components/workload/reachabilityTab.test.ts +47 -0
- package/src/index.ts +5 -0
- package/src/theme/components.css +64 -0
- package/src/theme/variables.css +5 -0
- package/src/types/core.ts +8 -0
- package/src/utils/api-resources.ts +2 -0
- package/src/utils/application-history.test.ts +1 -1
- package/src/utils/applications.test.ts +63 -1
- package/src/utils/applications.ts +27 -0
- package/src/utils/inClusterConsent.test.ts +109 -0
- package/src/utils/inClusterConsent.ts +72 -0
- package/src/utils/index.ts +1 -0
- package/src/utils/navigation.test.ts +61 -0
- package/src/utils/navigation.ts +11 -9
- package/src/utils/pluralize.test.ts +80 -1
- package/src/utils/pluralize.ts +34 -0
- package/src/utils/resource-icons.test.ts +33 -0
- package/src/utils/resource-icons.ts +65 -2
|
@@ -0,0 +1,797 @@
|
|
|
1
|
+
import type { Trace, RouteResult, ResourceRef } from './types'
|
|
2
|
+
import type { Mark, SevTone } from './reachMarks'
|
|
3
|
+
import { routeMark, routeChip, routeTone, routeAsSeenFrom, originRouteEvidence, routeForOrigin, traceInClusterRunnable, markHelp } from './reachMarks'
|
|
4
|
+
import type { Origin, OriginId } from './reachOrigins'
|
|
5
|
+
import { strongestGap, actionableGap, originSkipReason, originInformationalReason } from './reachOrigins'
|
|
6
|
+
import { hopEvidenceFor, originProducedEvidence, type GraphNode } from './reachGraphModel'
|
|
7
|
+
|
|
8
|
+
// 'run-probes' re-runs the reachability probes. It is deliberately NOT
|
|
9
|
+
// 'refresh': the panel that offers it promises fresh evidence, and a static
|
|
10
|
+
// refetch collects none.
|
|
11
|
+
export type InspectorAction = 'run-in-cluster' | 'run-probes' | 'open-resource' | 'copy-command'
|
|
12
|
+
|
|
13
|
+
export interface InspectorCTA {
|
|
14
|
+
text: string
|
|
15
|
+
primary?: boolean
|
|
16
|
+
action?: InspectorAction
|
|
17
|
+
ref?: ResourceRef
|
|
18
|
+
command?: string
|
|
19
|
+
/** Set when the CTA describes something Radar cannot do - rendered inert. */
|
|
20
|
+
disabledReason?: string
|
|
21
|
+
}
|
|
22
|
+
|
|
23
|
+
/** Only nodes are selectable now. Edges carry no segment-local evidence, so
|
|
24
|
+
* clicking one could only ever repeat the path result - see `Sidebar`. */
|
|
25
|
+
export type Selection = string | undefined
|
|
26
|
+
|
|
27
|
+
/**
|
|
28
|
+
* What the sidebar shows.
|
|
29
|
+
*
|
|
30
|
+
* `path` is ALWAYS present: whether traffic got through, from where, with what
|
|
31
|
+
* caveats, and what to do next. That question must never require a click - it
|
|
32
|
+
* is the reason the tab exists. `resource` is ADDITIVE, appearing when a node
|
|
33
|
+
* is selected, and never replaces the diagnosis.
|
|
34
|
+
*/
|
|
35
|
+
export interface Sidebar {
|
|
36
|
+
path: {
|
|
37
|
+
chipTone: SevTone
|
|
38
|
+
chipText: string
|
|
39
|
+
title: string
|
|
40
|
+
/** The concrete path under test. Always visible: it was inside the collapsed
|
|
41
|
+
* details, so the panel described a result without ever naming what was
|
|
42
|
+
* requested. */
|
|
43
|
+
request?: string
|
|
44
|
+
body: string
|
|
45
|
+
scope: { k: string; v: string }[]
|
|
46
|
+
evidence: { mark: Mark; text: string }[]
|
|
47
|
+
notProve: string[]
|
|
48
|
+
next: { header: string; body: string; blocked?: string; ctas: InspectorCTA[] }
|
|
49
|
+
}
|
|
50
|
+
/** Every hop on the selected path, in order - the whole story for this path
|
|
51
|
+
* seen from this vantage, rather than a summary plus whichever node was last
|
|
52
|
+
* clicked. Reading beat clicking: understanding a path used to take three
|
|
53
|
+
* clicks whose panel meant something different after each one. */
|
|
54
|
+
hops: HopReport[]
|
|
55
|
+
/** Configured-but-bypassed resources: on screen for orientation, outside the
|
|
56
|
+
* journey. The label says WHY they are not part of this vantage's path. */
|
|
57
|
+
context?: { label: string; hops: HopReport[] }
|
|
58
|
+
}
|
|
59
|
+
|
|
60
|
+
export interface HopDetail {
|
|
61
|
+
kind: string
|
|
62
|
+
name: string
|
|
63
|
+
chipTone: SevTone
|
|
64
|
+
chipText: string
|
|
65
|
+
body: string
|
|
66
|
+
facts: { k: string; v: string }[]
|
|
67
|
+
rows?: { mark: Mark; name: string; detail: string }[]
|
|
68
|
+
moreRows?: number
|
|
69
|
+
anomalies?: { mark: Mark; text: string }[]
|
|
70
|
+
notProve: string[]
|
|
71
|
+
openRef?: ResourceRef
|
|
72
|
+
}
|
|
73
|
+
|
|
74
|
+
/** Configured-but-bypassed resources, grouped under one label naming WHY they
|
|
75
|
+
* are outside this vantage's journey. Their config sections stay readable;
|
|
76
|
+
* their journey-state words do not apply. */
|
|
77
|
+
function contextGroup(ctx: Ctx, byId: Map<string, GraphNode>, sel: Selection): Sidebar['context'] {
|
|
78
|
+
const nodes = (ctx.contextNodeIds ?? []).map((id) => byId.get(id)).filter((n): n is GraphNode => !!n)
|
|
79
|
+
if (nodes.length === 0) return undefined
|
|
80
|
+
const label =
|
|
81
|
+
ctx.origin.id === 'apiserver'
|
|
82
|
+
? 'CONFIGURED ENTRY — KUBERNETES RELAYED THE REQUEST PAST THIS'
|
|
83
|
+
: ctx.origin.id === 'incluster' || ctx.origin.id === 'radar-incluster'
|
|
84
|
+
? 'CONFIGURED ENTRY — THE PROBE DIALS THE SERVICE DIRECTLY'
|
|
85
|
+
: 'PARALLEL ENTRY — NOT ON THIS ROUTE\u2019S PATH'
|
|
86
|
+
return {
|
|
87
|
+
label,
|
|
88
|
+
// Context hops are collapsed by default - they are not the journey. But the
|
|
89
|
+
// SELECTED one opens: the entry-problem row's whole purpose is "show me the
|
|
90
|
+
// cause", and landing on a still-collapsed section makes the reader spend a
|
|
91
|
+
// second click on the thing they just asked for.
|
|
92
|
+
hops: nodes.map((n) => ({
|
|
93
|
+
...resourceSection(n),
|
|
94
|
+
id: n.id,
|
|
95
|
+
state: 'plain' as const,
|
|
96
|
+
chipText: '',
|
|
97
|
+
expanded: sel === n.id,
|
|
98
|
+
})),
|
|
99
|
+
}
|
|
100
|
+
}
|
|
101
|
+
|
|
102
|
+
/**
|
|
103
|
+
* A hop's place in the story.
|
|
104
|
+
*
|
|
105
|
+
* - `break` where the request is known to have stopped. Expanded by default;
|
|
106
|
+
* it is the answer.
|
|
107
|
+
* - `before` reached before that. Collapsed - it worked, so it is context.
|
|
108
|
+
* - `after` never tried, because something earlier stopped. Collapsed, and
|
|
109
|
+
* labelled so its silence is not read as health.
|
|
110
|
+
* - `plain` no break to order around.
|
|
111
|
+
*/
|
|
112
|
+
export type HopState = 'before' | 'break' | 'after' | 'plain'
|
|
113
|
+
|
|
114
|
+
export interface HopReport extends HopDetail {
|
|
115
|
+
id: string
|
|
116
|
+
state: HopState
|
|
117
|
+
expanded: boolean
|
|
118
|
+
/** Set on entry hops when they are parallel members of one entry stage - the
|
|
119
|
+
* note reads "one of N parallel entry points" instead of implying sequence. */
|
|
120
|
+
parallelCount?: number
|
|
121
|
+
}
|
|
122
|
+
|
|
123
|
+
const NOT_DATAPLANE = 'Nothing about the normal network path. Kubernetes relayed this for us, so routing, network policy and the mesh were all skipped.'
|
|
124
|
+
const SYNTHETIC_IDENTITY =
|
|
125
|
+
'That your app can reach it. This test ran from a throwaway Pod under the namespace’s default account with no token mounted — not as your application — so anything that checks who is calling may answer differently.'
|
|
126
|
+
|
|
127
|
+
function originScope(o: Origin, trace: Trace, request?: string): { k: string; v: string }[] {
|
|
128
|
+
const runsIn: Record<OriginId, string> = {
|
|
129
|
+
incluster: `a throwaway Pod in ${trace.subject.namespace || 'the cluster'}`,
|
|
130
|
+
'radar-incluster': 'Radar\u2019s own Pod',
|
|
131
|
+
apiserver: 'the kube-apiserver process',
|
|
132
|
+
local: 'your workstation (outside the cluster)',
|
|
133
|
+
caller: 'the application workload’s own Pod',
|
|
134
|
+
external: 'a client on the public internet',
|
|
135
|
+
}
|
|
136
|
+
return [
|
|
137
|
+
{ k: 'TESTED FROM', v: o.name },
|
|
138
|
+
{ k: 'RUNS IN', v: runsIn[o.id] },
|
|
139
|
+
// The laptop row's content is where the dial came FROM ("you dialled a
|
|
140
|
+
// public address…"), not who dialled - identity and network position are
|
|
141
|
+
// different facts and must not share a label.
|
|
142
|
+
{ k: o.id === 'local' ? 'NETWORK POSITION' : 'IDENTITY', v: o.identity },
|
|
143
|
+
{ k: 'MECHANISM', v: o.mech },
|
|
144
|
+
// A status code cannot be read without knowing what was asked for - "404"
|
|
145
|
+
// means nothing until you know the request was GET /.
|
|
146
|
+
...(request ? [{ k: 'REQUEST', v: request }] : []),
|
|
147
|
+
]
|
|
148
|
+
}
|
|
149
|
+
|
|
150
|
+
/**
|
|
151
|
+
* The next-step prompt, driven by the strongest gap Radar can CLOSE. The
|
|
152
|
+
* unreachable ceiling is stated underneath as a caveat instead of being offered
|
|
153
|
+
* as an action - a button that can never be pressed is not a next step.
|
|
154
|
+
*/
|
|
155
|
+
function gapNext(
|
|
156
|
+
origins: Origin[],
|
|
157
|
+
current: Origin,
|
|
158
|
+
namespace?: string,
|
|
159
|
+
multiPath?: boolean,
|
|
160
|
+
inClusterRunnable = true,
|
|
161
|
+
canRun = true,
|
|
162
|
+
): Sidebar['path']['next'] {
|
|
163
|
+
// The server can only test a route that carries a concrete in-cluster request.
|
|
164
|
+
// When none does, the run fails with "not supported for this subject" - so the
|
|
165
|
+
// panel must say that HERE, instead of recommending it as the strongest
|
|
166
|
+
// evidence available and letting the operator find out by clicking.
|
|
167
|
+
const notRunnable = 'No path on this resource has a request Radar can send from inside the cluster, so this test would have nothing to run.'
|
|
168
|
+
// No button here. The in-cluster run is offered ON the vantage capsule in the
|
|
169
|
+
// graph, which is the thing that would produce the missing evidence - and a
|
|
170
|
+
// third copy of the same control (header, panel, capsule) was one too many.
|
|
171
|
+
// The panel keeps the REASONING, which is what it is good at.
|
|
172
|
+
const inClusterCTA = (): InspectorCTA[] => {
|
|
173
|
+
// No handler wired (a library consumer that omits it) means the click would
|
|
174
|
+
// do nothing at all - offer nothing rather than a button that lies.
|
|
175
|
+
if (!canRun) return []
|
|
176
|
+
return inClusterRunnable
|
|
177
|
+
? [{ text: '⚗ Run the in-cluster test', action: 'run-in-cluster', primary: true }]
|
|
178
|
+
: [{ text: '⚗ Run the in-cluster test', action: 'run-in-cluster', disabledReason: notRunnable }]
|
|
179
|
+
}
|
|
180
|
+
const actionable = actionableGap(origins)
|
|
181
|
+
const ceiling = strongestGap(origins)
|
|
182
|
+
const ceilingNote = ceiling?.unsupported ? `Even then, ${ceiling.name.toLowerCase()} stays untested — ${ceiling.unavailable}` : undefined
|
|
183
|
+
const denied = origins.find((o) => o.mark === 'denied')
|
|
184
|
+
// A run is never scoped to the selected path - the server tests every declared
|
|
185
|
+
// path in one pass. Saying so here stops the picker above from reading as a
|
|
186
|
+
// filter on what the button will do.
|
|
187
|
+
const allPaths = multiPath ? ' The test covers every path on this resource, not only this one.' : ''
|
|
188
|
+
|
|
189
|
+
// Mid-run, a running origin is not a "gap" any more, so the no-gap branch
|
|
190
|
+
// below would claim Radar "already has the strongest evidence" BEFORE the
|
|
191
|
+
// result exists. Being in flight is its own state, and it is the whole truth
|
|
192
|
+
// about this panel right now.
|
|
193
|
+
const runningOrigin = origins.find((o) => o.mark === 'running')
|
|
194
|
+
if (runningOrigin) {
|
|
195
|
+
return {
|
|
196
|
+
header: 'TEST RUNNING',
|
|
197
|
+
body: `${runningOrigin.name} is testing now — the result lands here when it finishes.`,
|
|
198
|
+
ctas: [],
|
|
199
|
+
}
|
|
200
|
+
}
|
|
201
|
+
|
|
202
|
+
if (!actionable && denied) {
|
|
203
|
+
const ns = namespace || '<namespace>'
|
|
204
|
+
return {
|
|
205
|
+
header: 'ASK FOR THIS PERMISSION',
|
|
206
|
+
body: `Running an in-cluster test needs \`create\` on \`jobs\` in ${ns}. Grant it, or run the check from a workload you already control.`,
|
|
207
|
+
blocked: denied.unavailable,
|
|
208
|
+
ctas: [{ text: 'Copy the permission check', action: 'copy-command', command: `kubectl auth can-i create jobs -n ${ns}` }],
|
|
209
|
+
}
|
|
210
|
+
}
|
|
211
|
+
if (actionable && actionable.id === current.id) {
|
|
212
|
+
// You are looking at the vantage that is itself the gap. The section body
|
|
213
|
+
// already said nothing ran from here, and the caveats section already
|
|
214
|
+
// carries the ceiling - so this is the ACTION and nothing else.
|
|
215
|
+
return {
|
|
216
|
+
header: '',
|
|
217
|
+
body: inClusterRunnable ? '' : `${notRunnable} Every declared path here was skipped before a request could be formed.`,
|
|
218
|
+
ctas: actionable.id === 'incluster' ? inClusterCTA() : [{ text: '⟳ Re-run', action: 'run-probes' }],
|
|
219
|
+
}
|
|
220
|
+
}
|
|
221
|
+
if (!actionable) {
|
|
222
|
+
// Nothing to offer: say nothing. This was a resource-level statement living
|
|
223
|
+
// in the selection-scoped panel, identical on every vantage - the scope
|
|
224
|
+
// mixing this redesign exists to remove - and the ceiling it gestured at is
|
|
225
|
+
// already stated, with specifics, in WHAT THIS DOESN'T PROVE and in the
|
|
226
|
+
// footer's coverage ledger. An empty section is the honest render of
|
|
227
|
+
// "there is no next step".
|
|
228
|
+
return { header: '', body: '', ctas: [] }
|
|
229
|
+
}
|
|
230
|
+
return {
|
|
231
|
+
header: 'RUN THIS NEXT',
|
|
232
|
+
body:
|
|
233
|
+
actionable.id === 'incluster' && !inClusterRunnable
|
|
234
|
+
? `${notRunnable} Every declared path here was skipped before a request could be formed.`
|
|
235
|
+
: `${actionable.name} has not been used for this path, and is the strongest evidence Radar can still collect.${allPaths}`,
|
|
236
|
+
blocked: ceilingNote,
|
|
237
|
+
ctas: actionable.id === 'incluster' ? inClusterCTA() : [{ text: '⟳ Re-run', action: 'run-probes' }],
|
|
238
|
+
}
|
|
239
|
+
}
|
|
240
|
+
|
|
241
|
+
interface Ctx {
|
|
242
|
+
trace: Trace
|
|
243
|
+
route?: RouteResult
|
|
244
|
+
origin: Origin
|
|
245
|
+
origins: Origin[]
|
|
246
|
+
nodes: GraphNode[]
|
|
247
|
+
/** Where the path first broke, from the graph model. */
|
|
248
|
+
breakNodeId?: string
|
|
249
|
+
/** A Service-routing boundary: the node whose EXIT the request never got
|
|
250
|
+
* past. The break renders on that hop with exit-phrased copy - no node is
|
|
251
|
+
* blamed. */
|
|
252
|
+
breakAtExitOf?: string
|
|
253
|
+
/** Inline nodes that are NOT network hops (the workload). Never given
|
|
254
|
+
* journey states - a workload is neither "reached" nor a stopping place. */
|
|
255
|
+
nonNetworkNodeIds?: string[]
|
|
256
|
+
/** Configured-but-bypassed entries - context, never the journey. */
|
|
257
|
+
contextNodeIds?: string[]
|
|
258
|
+
/** Non-network nodes spliced into the display after a journey hop. */
|
|
259
|
+
interleave?: { id: string; afterId: string }[]
|
|
260
|
+
/** >1 when the journey's entries are parallel members of one stage. */
|
|
261
|
+
entryParallelCount?: number
|
|
262
|
+
/** The journey's entry-hop node ids, from the graph model. */
|
|
263
|
+
journeyEntryNodeIds?: string[]
|
|
264
|
+
/** The selected route's chain in traversal order, from the graph model. */
|
|
265
|
+
pathNodeIds?: string[]
|
|
266
|
+
stale?: boolean
|
|
267
|
+
running?: boolean
|
|
268
|
+
/** More than one scenario is on screen, so scope has to be stated explicitly. */
|
|
269
|
+
multiPath?: boolean
|
|
270
|
+
/** The HTTP path the run requests, as chosen in "what to test". */
|
|
271
|
+
httpPath?: string
|
|
272
|
+
/** False when the host wired no in-cluster handler, or permission is denied:
|
|
273
|
+
* the run must not be offered at all then. */
|
|
274
|
+
canRunInCluster?: boolean
|
|
275
|
+
}
|
|
276
|
+
|
|
277
|
+
/**
|
|
278
|
+
* Whether an in-cluster run has anything to send.
|
|
279
|
+
*
|
|
280
|
+
* The server tests a route only when it carries a concrete InClusterRequest; a
|
|
281
|
+
* route whose probes were all skipped never becomes one, so a subject can offer
|
|
282
|
+
* paths and still have nothing runnable. Knowing this from the trace is what
|
|
283
|
+
* lets the panel say so BEFORE the operator spends a Job on it.
|
|
284
|
+
*/
|
|
285
|
+
|
|
286
|
+
|
|
287
|
+
/** Whether the diagnosis says nothing the headline has not already said. Both
|
|
288
|
+
* are generated, so they collide whenever the producer falls back to the same
|
|
289
|
+
* generic sentence - and a banner repeating the line above it reads as a second
|
|
290
|
+
* problem rather than the same one. */
|
|
291
|
+
function restatesTitle(summary: string | undefined, title: string): boolean {
|
|
292
|
+
const norm = (x: string) => x.trim().toLowerCase().replace(/[.!]$/, '')
|
|
293
|
+
return !!summary && !!title && norm(summary) === norm(title)
|
|
294
|
+
}
|
|
295
|
+
|
|
296
|
+
/** The request this route would send, in one line - "GET /healthz", or "TCP
|
|
297
|
+
* connect" where there is no application request to make. */
|
|
298
|
+
function requestLabel(route: RouteResult | undefined, httpPath?: string): string | undefined {
|
|
299
|
+
const r = route?.inClusterRequest
|
|
300
|
+
if (!r) return undefined
|
|
301
|
+
if (r.protocol === 'tcp') return 'TCP connect'
|
|
302
|
+
const path = httpPath || r.path || '/'
|
|
303
|
+
return `GET ${path}`
|
|
304
|
+
}
|
|
305
|
+
|
|
306
|
+
/**
|
|
307
|
+
* What an ANSWER means for this page's question.
|
|
308
|
+
*
|
|
309
|
+
* "reached" plus an HTTP status is the single most-misread result on the tab:
|
|
310
|
+
* a reader sees amber and 404 and cannot tell whether that is a problem. It is
|
|
311
|
+
* not - for the question this page asks. The network path is proven the moment
|
|
312
|
+
* the app answers at all; the status code is a statement about the app's own
|
|
313
|
+
* routing, auth or health, which reachability does not judge. Say that, and say
|
|
314
|
+
* what would turn it into a verified pass.
|
|
315
|
+
*/
|
|
316
|
+
/**
|
|
317
|
+
* What an answer MEANS, and whether that answer is evidence the path works.
|
|
318
|
+
*
|
|
319
|
+
* `pathWorks` is the load-bearing half: an app answering 404 proves the journey
|
|
320
|
+
* reached it, while a gateway answering 502 proves the opposite. Both are
|
|
321
|
+
* "answers", so a caller that prefixes every meaning with "the path works"
|
|
322
|
+
* states the reverse of the truth on exactly the cases that matter most.
|
|
323
|
+
*/
|
|
324
|
+
function statusMeaning(evidence?: string, httpPath?: string, failedLayer?: string): { text: string; pathWorks: boolean } | undefined {
|
|
325
|
+
// A cert failure never gets an HTTP status - the handshake stops before a
|
|
326
|
+
// request is sent - so this must be decided before looking for a code.
|
|
327
|
+
if (failedLayer === 'tls') {
|
|
328
|
+
return { text: 'the TLS handshake did not verify - a certificate problem, not an application one.', pathWorks: false }
|
|
329
|
+
}
|
|
330
|
+
const m = /HTTP\s+(\d{3})/i.exec(evidence ?? '')
|
|
331
|
+
if (!m) return undefined
|
|
332
|
+
const code = Number(m[1])
|
|
333
|
+
const asked = httpPath && httpPath !== '/' ? httpPath : '/'
|
|
334
|
+
// 502/504 are NOT the app: a gateway answering that it could not reach its
|
|
335
|
+
// upstream. The producer says so with failedLayer 'upstream'; blaming app
|
|
336
|
+
// health here contradicted the chip beside it.
|
|
337
|
+
if (failedLayer === 'upstream' || code === 502 || code === 504) {
|
|
338
|
+
return {
|
|
339
|
+
text: `the front door answered, but only to say it could not reach the backend (HTTP ${code}) - the break is between the entry and the app, not inside the app.`,
|
|
340
|
+
pathWorks: false,
|
|
341
|
+
}
|
|
342
|
+
}
|
|
343
|
+
if (code >= 500) {
|
|
344
|
+
return { text: `the request reached the app and the app itself returned an error (HTTP ${code}) - application health, which this page does not judge.`, pathWorks: true }
|
|
345
|
+
}
|
|
346
|
+
if (code === 401 || code === 403 || code === 407) {
|
|
347
|
+
return { text: `the app answered by demanding credentials (HTTP ${code}) - it is enforcing auth, which is a different thing from reachability.`, pathWorks: true }
|
|
348
|
+
}
|
|
349
|
+
if (code >= 300 && code < 400) {
|
|
350
|
+
return { text: `the app answered with a redirect (HTTP ${code}), which Radar does not follow - so whatever it points at is untested from here.`, pathWorks: true }
|
|
351
|
+
}
|
|
352
|
+
if (code >= 400) {
|
|
353
|
+
return { text: `the app answered (HTTP ${code}) - that is it saying it serves no route for ${asked}, not a reachability problem. To verify a real route, re-run with a path your app serves.`, pathWorks: true }
|
|
354
|
+
}
|
|
355
|
+
return undefined
|
|
356
|
+
}
|
|
357
|
+
|
|
358
|
+
/** The persistent diagnosis: did traffic get through, from where, and what next. */
|
|
359
|
+
/** The suffix the localization lines carry; stripped when deduping so one
|
|
360
|
+
* observation cannot appear twice at two lengths. */
|
|
361
|
+
const LOCALIZED_SUFFIX_RE = / [-\u2014] checked directly, past the entry point$/
|
|
362
|
+
|
|
363
|
+
function pathSection(ctx: Ctx): Sidebar['path'] {
|
|
364
|
+
const { trace, route, origin, origins } = ctx
|
|
365
|
+
// The route outcome is merged across origins. Without this gate the panel
|
|
366
|
+
// rendered another vantage's success under the selected vantage's name - a
|
|
367
|
+
// permanently unavailable origin could read "a real request went through"
|
|
368
|
+
// while the graph beside it said "not routable". Same lie the graph already
|
|
369
|
+
// guards against, in the surface users actually read.
|
|
370
|
+
// This origin's OWN result when the producer sent one; the coarse
|
|
371
|
+
// "did this origin produce anything" gate only remains as the fallback.
|
|
372
|
+
const ev = originRouteEvidence(route, origin.id)
|
|
373
|
+
const asSeen = ev.kind === 'none' ? undefined : ev.result
|
|
374
|
+
// A config-derived break is true of every vantage and observed by none, so it
|
|
375
|
+
// reads as a configuration fact - never as this origin's failed dial.
|
|
376
|
+
const fromConfig = ev.kind === 'config'
|
|
377
|
+
// Which KIND of derived break. Declared-config is broken whatever the cluster
|
|
378
|
+
// is doing; cluster-state is true right now and changes when the workload
|
|
379
|
+
// does. Calling the second one a configuration failure sends the reader to
|
|
380
|
+
// edit YAML when the fix is to get Pods ready.
|
|
381
|
+
const basis = ev.kind === 'config' ? ev.result.basis : undefined
|
|
382
|
+
const hasEvidence = ev.kind === 'own' || (ev.kind === 'rollup' && originProducedEvidence(origin))
|
|
383
|
+
const mark: Mark = fromConfig
|
|
384
|
+
? 'config'
|
|
385
|
+
: hasEvidence
|
|
386
|
+
? asSeen
|
|
387
|
+
? routeMark(asSeen, { stale: ctx.stale, running: ctx.running })
|
|
388
|
+
: 'untested'
|
|
389
|
+
: origin.mark
|
|
390
|
+
|
|
391
|
+
const notProve: string[] = []
|
|
392
|
+
if (origin.kind === 'synthetic') notProve.push(SYNTHETIC_IDENTITY)
|
|
393
|
+
if (origin.kind === 'relayed') notProve.push(NOT_DATAPLANE)
|
|
394
|
+
const hasFrontDoor = (trace.upstreams ?? []).length > 0
|
|
395
|
+
const external = origins.find((o) => o.id === 'external')
|
|
396
|
+
if (hasFrontDoor && external?.unsupported && origin.id !== 'external') {
|
|
397
|
+
// "No request has come in from outside" is only true until Radar's own
|
|
398
|
+
// machine dials the public entry and gets an answer - that request DID come
|
|
399
|
+
// from outside. What stays unproven then is narrower: a real user's request.
|
|
400
|
+
const outsideDialAnswered = (trace.upstreams ?? []).some((h) => {
|
|
401
|
+
const m = hopEvidenceFor(h, { id: 'local' }, trace.runVantage)?.mark
|
|
402
|
+
return m === 'proved' || m === 'answered'
|
|
403
|
+
})
|
|
404
|
+
notProve.push(
|
|
405
|
+
outsideDialAnswered
|
|
406
|
+
? 'That real users can reach it — Radar dialled the public entry from outside and got an answer, but no real user request has been observed.'
|
|
407
|
+
: 'That people on the internet can reach it — no request has come in from outside.',
|
|
408
|
+
)
|
|
409
|
+
}
|
|
410
|
+
|
|
411
|
+
const evidence: { mark: Mark; text: string }[] = []
|
|
412
|
+
const seen = new Map<string, number>()
|
|
413
|
+
// Two lines can be the SAME observation worded at different lengths: a route's
|
|
414
|
+
// rollup evidence and the localization fact from that same dial differ only by
|
|
415
|
+
// the " - checked directly..." suffix, so exact-string dedupe let both through
|
|
416
|
+
// and the reader saw "HTTP 404 - reached" twice. Key on the observation and
|
|
417
|
+
// keep the more specific wording.
|
|
418
|
+
const add = (m: Mark, text: string) => {
|
|
419
|
+
const t = text.trim()
|
|
420
|
+
const key = t.toLowerCase().replace(LOCALIZED_SUFFIX_RE, '').trim()
|
|
421
|
+
if (!key) return
|
|
422
|
+
const at = seen.get(key)
|
|
423
|
+
if (at !== undefined) {
|
|
424
|
+
if (t.length > evidence[at].text.length) evidence[at] = { mark: evidence[at].mark, text: t }
|
|
425
|
+
return
|
|
426
|
+
}
|
|
427
|
+
seen.set(key, evidence.length)
|
|
428
|
+
evidence.push({ mark: m, text: t })
|
|
429
|
+
}
|
|
430
|
+
if ((hasEvidence || fromConfig) && asSeen?.evidence) add(mark, asSeen.evidence)
|
|
431
|
+
// A result with no evidence STRING is still a result: show what the mark
|
|
432
|
+
// means rather than leaving the section empty, which would read as "nothing
|
|
433
|
+
// ran" for a vantage that did.
|
|
434
|
+
else if (hasEvidence && asSeen) add(mark, markHelp(mark))
|
|
435
|
+
// What this vantage saw at each HOP. The route is built from the backend's
|
|
436
|
+
// probes, so a laptop that dialled the front door and got an answer had all
|
|
437
|
+
// of it discarded and read as "no test has been run from here" - beside a
|
|
438
|
+
// graph drawing that very dial.
|
|
439
|
+
const hopSeen = ctx.nodes
|
|
440
|
+
.filter((n) => !n.isOrigin && n.hop)
|
|
441
|
+
.map((n) => ({ node: n, ev: hopEvidenceFor(n.hop, origin, trace.runVantage, { stale: ctx.stale, running: ctx.running }) }))
|
|
442
|
+
.filter((x): x is { node: GraphNode; ev: NonNullable<ReturnType<typeof hopEvidenceFor>> } => !!x.ev)
|
|
443
|
+
if (!hasEvidence && !fromConfig) {
|
|
444
|
+
for (const { node, ev: e } of hopSeen) add(e.mark, `${node.kind.toLowerCase()} ${node.name} — ${e.title || e.label}`)
|
|
445
|
+
}
|
|
446
|
+
// The one boundary two observations can establish. Stated as the reasoning
|
|
447
|
+
// that produced it, so it reads as evidence rather than as a verdict.
|
|
448
|
+
if (ev.kind === 'own' && routeForOrigin(route, origin.id)?.failedBoundary === 'service-routing') {
|
|
449
|
+
add('failed', 'the Pods answered directly, but the Service did not — so the Service’s own routing is what breaks')
|
|
450
|
+
}
|
|
451
|
+
// Localization facts are behind-the-gate (apiserver / direct-pod) evidence.
|
|
452
|
+
// They belong to the relayed origin, not to whichever origin is selected -
|
|
453
|
+
// listing them under the in-cluster probe credited it with observations it
|
|
454
|
+
// never made.
|
|
455
|
+
for (const f of origin.id === 'apiserver' && hasEvidence ? route?.localization ?? [] : []) {
|
|
456
|
+
const layer = f.layer.toUpperCase()
|
|
457
|
+
const detail = f.detail?.trim()
|
|
458
|
+
const body = !detail ? layer : detail.toUpperCase().startsWith(layer) ? detail : `${layer} · ${detail}`
|
|
459
|
+
add(f.ok ? 'proxied' : 'failed', `${body} — checked directly, past the entry point`)
|
|
460
|
+
}
|
|
461
|
+
if (evidence.length === 0) {
|
|
462
|
+
// A route the run SKIPPED still carries why. The reason is looked up from
|
|
463
|
+
// the skip rows by the exact identity that produced it, so it appears only
|
|
464
|
+
// under the vantage whose dial was skipped - never charging the in-cluster
|
|
465
|
+
// probe with the proxy's limits, or the laptop with the proxy's timeouts.
|
|
466
|
+
const skipReason = ev.kind === 'rollup' && asSeen?.outcome === 'not-tested' ? originSkipReason(trace, origin.id, route) : undefined
|
|
467
|
+
// An in-cluster run that was ATTEMPTED and never started is not "no test
|
|
468
|
+
// has been run" - the capsule preserves the attempt and this row must not
|
|
469
|
+
// erase it. The error itself is the observation.
|
|
470
|
+
const attemptError = origin.id === 'incluster' && origin.mark === 'blocked' ? origin.unavailable : undefined
|
|
471
|
+
// A demoted run produced a real observation; "no test has been run from
|
|
472
|
+
// here" erased it.
|
|
473
|
+
const informational = origin.mark === 'inconclusive' ? originInformationalReason(trace, origin.id) : undefined
|
|
474
|
+
add(
|
|
475
|
+
mark,
|
|
476
|
+
// The attempt error outranks skip rows: a prior Job's leftover skips
|
|
477
|
+
// would otherwise paint a fresh failed run (image pull, quota) with last
|
|
478
|
+
// run's reason, one pane from a capsule saying "test couldn't run".
|
|
479
|
+
informational ||
|
|
480
|
+
attemptError ||
|
|
481
|
+
skipReason ||
|
|
482
|
+
(origin.unsupported
|
|
483
|
+
? 'Radar cannot test from here, so nothing has been learned this way'
|
|
484
|
+
: origin.mark === 'denied'
|
|
485
|
+
? 'not permitted to run this test'
|
|
486
|
+
: // A vantage that CANNOT be used states why (nothing dialable from
|
|
487
|
+
// the laptop, a skipped mechanism) - "no test has been run" reads
|
|
488
|
+
// as a test someone forgot, which is a different claim.
|
|
489
|
+
origin.unavailable || ''),
|
|
490
|
+
)
|
|
491
|
+
}
|
|
492
|
+
|
|
493
|
+
const failed = mark === 'failed'
|
|
494
|
+
// Only when it is about THIS path. A trace carries one diagnosis but many
|
|
495
|
+
// routes, so an unattributed or sibling-owned cause must not be rendered as
|
|
496
|
+
// the selected path's - the route's own evidence is what remains true.
|
|
497
|
+
const rawDiagnosis = trace.diagnosis
|
|
498
|
+
const diagnosis = rawDiagnosis && (!rawDiagnosis.route || rawDiagnosis.route === route?.route) ? rawDiagnosis : undefined
|
|
499
|
+
const reachedSomething = !hasEvidence && !fromConfig && hopSeen.some((x) => x.ev.mark !== 'failed')
|
|
500
|
+
const answer = statusMeaning(asSeen?.evidence, ctx.httpPath, asSeen?.failedLayer)
|
|
501
|
+
const body = basis === 'cluster-state'
|
|
502
|
+
? 'Nothing is ready to serve this path right now, so it cannot work from any vantage. No request was sent to establish that — it is read off the current state of the cluster, and it changes when the workload does.'
|
|
503
|
+
: fromConfig
|
|
504
|
+
? 'The configuration itself is broken, so this path cannot work from any vantage. No request was sent to establish that — it is read off what is declared.'
|
|
505
|
+
: reachedSomething
|
|
506
|
+
? 'This vantage did reach part of the path — see what it saw below. It has no result for this route as a whole, so the end-to-end journey from here is still unproven.'
|
|
507
|
+
: origin.mark === 'inconclusive' || mark === 'inconclusive'
|
|
508
|
+
// A demoted run RAN and answered - it is only kept out of the verdict. The
|
|
509
|
+
// "nothing has been tested" branch below fired first and denied it happened.
|
|
510
|
+
// Both the origin's mark and the route's own can carry the demotion, and the
|
|
511
|
+
// sentence is the same either way; the reason comes from the informational
|
|
512
|
+
// skip that recorded it, never from the route-scoped skip lookup.
|
|
513
|
+
? `The probe ran and got an answer, but it is kept as evidence rather than a verdict${
|
|
514
|
+
originInformationalReason(trace, origin.id) ? `: ${originInformationalReason(trace, origin.id)}` : ''
|
|
515
|
+
}. A throwaway identity cannot stand in for your application, so what it saw informs but never decides.`
|
|
516
|
+
: !hasEvidence
|
|
517
|
+
? (origin.unavailable || 'Nothing has been tested from here, so this says nothing about whether traffic gets through.') +
|
|
518
|
+
// When NOTHING was tested anywhere, the health dots are the only colour
|
|
519
|
+
// on the board and read as a passed test - the one state where the
|
|
520
|
+
// dot/line split needs saying out loud, not just in the caption.
|
|
521
|
+
((trace.coverage?.tested ?? 0) === 0 ? ' The dots on the graph show each resource’s own reported health — cluster state, not a test result.' : '')
|
|
522
|
+
: failed
|
|
523
|
+
? 'This is the first confirmed failure. Everything after it was never tried, so there is nothing to report past this point.'
|
|
524
|
+
: mark === 'proved'
|
|
525
|
+
? 'A real request went through and the target answered.'
|
|
526
|
+
: mark === 'proxied'
|
|
527
|
+
? [
|
|
528
|
+
'Kubernetes relayed a request and the target answered — which shows something is serving, not that the normal path works.',
|
|
529
|
+
// The relay caveat alone leaves "404" unreadable: the reader still
|
|
530
|
+
// cannot tell whether the answer itself was a problem.
|
|
531
|
+
answer && `As for the answer itself: ${answer.text}`,
|
|
532
|
+
]
|
|
533
|
+
.filter(Boolean)
|
|
534
|
+
.join(' ')
|
|
535
|
+
: mark === 'untested'
|
|
536
|
+
? 'Nothing has been tried from here yet. Configuration may look right, but that is intent, not proof.'
|
|
537
|
+
: mark === 'stale'
|
|
538
|
+
? 'This result predates a change to the cluster, so it is set aside rather than trusted.'
|
|
539
|
+
: mark === 'excluded'
|
|
540
|
+
// Benign by design (deliberately scaled to zero, a not-eligible
|
|
541
|
+
// endpoint). It fell through to the generic "answered" sentence,
|
|
542
|
+
// which described a request that was never sent.
|
|
543
|
+
? `${asSeen?.evidence || 'Nothing is behind this path right now'} — that is deliberate, not a failure: nothing was sent, because there is nothing to reach.`
|
|
544
|
+
: mark === 'running'
|
|
545
|
+
? 'A test is running. Earlier results stay until new ones replace them.'
|
|
546
|
+
: // A proxy-only failure wears the same amber mark as a real answer,
|
|
547
|
+
// but nothing answered - saying "the target answered" here sent
|
|
548
|
+
// the reader to debug an application response that never existed.
|
|
549
|
+
asSeen?.outcome === 'unreachable' && asSeen?.confidence === 'indirect'
|
|
550
|
+
? 'The relayed dial failed. The proxy bypasses the real path, so this does not condemn it — but nothing answered, and the real path is still untested.'
|
|
551
|
+
: // Only an answer that proves the journey completed earns "the
|
|
552
|
+
// path works" - a 502 or a failed handshake is an answer that
|
|
553
|
+
// says the opposite.
|
|
554
|
+
(answer ? (answer.pathWorks ? `The path works: ${answer.text}` : `The path did not work: ${answer.text}`) : undefined) ??
|
|
555
|
+
// A transport-only reach: nothing was asked of the application.
|
|
556
|
+
(asSeen?.outcome === 'reached'
|
|
557
|
+
? 'The port accepted a connection, but nothing was asked of the application - the transport works and the application protocol is unverified.'
|
|
558
|
+
: 'The target answered, but not with what was asked for.')
|
|
559
|
+
|
|
560
|
+
return {
|
|
561
|
+
chipTone: asSeen ? routeTone(asSeen, { stale: ctx.stale, running: ctx.running }) : 'unknown',
|
|
562
|
+
chipText: asSeen ? routeChip(asSeen, { stale: ctx.stale, running: ctx.running }) : 'not tested',
|
|
563
|
+
title: `${origin.name} → ${route?.target || trace.subject.name}`,
|
|
564
|
+
request: route ? `${route.route}${ctx.httpPath && ctx.httpPath !== '/' ? ` · HTTP path ${ctx.httpPath}` : ''}` : undefined,
|
|
565
|
+
body,
|
|
566
|
+
scope: [...originScope(origin, trace, requestLabel(route, ctx.httpPath)), ...(route ? [{ k: 'PATH', v: route.route }] : [])],
|
|
567
|
+
evidence,
|
|
568
|
+
notProve,
|
|
569
|
+
next:
|
|
570
|
+
failed && diagnosis
|
|
571
|
+
? {
|
|
572
|
+
header: 'LIKELY CAUSE',
|
|
573
|
+
body: diagnosis.summary + (diagnosis.nextAction ? ` ${diagnosis.nextAction}` : ''),
|
|
574
|
+
ctas: [
|
|
575
|
+
...(diagnosis.culpritResource ? [{ text: 'Open the culprit', primary: true, action: 'open-resource' as InspectorAction, ref: diagnosis.culpritResource }] : []),
|
|
576
|
+
...(diagnosis.command ? [{ text: 'Copy the command', action: 'copy-command' as InspectorAction, command: diagnosis.command }] : []),
|
|
577
|
+
],
|
|
578
|
+
}
|
|
579
|
+
: gapNext(origins, origin, trace.subject.namespace, ctx.multiPath, traceInClusterRunnable(trace), ctx.canRunInCluster !== false),
|
|
580
|
+
}
|
|
581
|
+
}
|
|
582
|
+
|
|
583
|
+
/** The additive detail for a selected node. Never replaces the diagnosis. */
|
|
584
|
+
function resourceSection(node: GraphNode): HopDetail {
|
|
585
|
+
const hop = node.hop
|
|
586
|
+
const findings = hop?.findings ?? []
|
|
587
|
+
|
|
588
|
+
if (node.podRows) {
|
|
589
|
+
const roster = hop?.config?.pods ?? []
|
|
590
|
+
const total = hop?.config?.podTotal ?? roster.length
|
|
591
|
+
const ready = typeof hop?.meta?.ready === 'number' ? (hop.meta.ready as number) : roster.filter((p) => p.ready).length
|
|
592
|
+
const selected = typeof hop?.meta?.selected === 'number' ? (hop.meta.selected as number) : total
|
|
593
|
+
const publishNotReady = !!hop?.meta?.publishNotReadyAddresses
|
|
594
|
+
const notReady = publishNotReady ? [] : roster.filter((p) => !p.ready)
|
|
595
|
+
const omitted = total - roster.length
|
|
596
|
+
const notProve: string[] = []
|
|
597
|
+
if (omitted > 0) notProve.push(`The ${omitted} Pods that were not tested. Untested is not proven.`)
|
|
598
|
+
if (notReady.length > 0) notProve.push(`The ${notReady.length} not-ready Pod${notReady.length > 1 ? 's' : ''} — nothing was sent to them, so nothing was learned.`)
|
|
599
|
+
return {
|
|
600
|
+
kind: 'PODS',
|
|
601
|
+
name: `${ready} of ${selected} eligible`,
|
|
602
|
+
chipTone: node.tone,
|
|
603
|
+
chipText: 'backends',
|
|
604
|
+
body: publishNotReady
|
|
605
|
+
? 'The Pods behind this Service. This Service is set to send traffic to Pods even before they report ready.'
|
|
606
|
+
: 'The Pods behind this Service. Kubernetes only sends traffic to the ones that report ready.',
|
|
607
|
+
facts: [
|
|
608
|
+
{ k: 'MATCHING PODS', v: `${selected}` },
|
|
609
|
+
// Derived from readiness, NOT from observed delivery - "taking traffic"
|
|
610
|
+
// claimed evidence we do not have.
|
|
611
|
+
{ k: 'ELIGIBLE', v: `${ready}` },
|
|
612
|
+
{
|
|
613
|
+
k: 'SITTING OUT',
|
|
614
|
+
v: publishNotReady ? 'none — not-ready Pods get traffic too' : notReady.length > 0 ? `${notReady.length} not ready` : 'none',
|
|
615
|
+
},
|
|
616
|
+
],
|
|
617
|
+
rows: node.podRows,
|
|
618
|
+
moreRows: node.moreRows,
|
|
619
|
+
anomalies: node.anomalies,
|
|
620
|
+
notProve,
|
|
621
|
+
}
|
|
622
|
+
}
|
|
623
|
+
|
|
624
|
+
const c = hop?.config
|
|
625
|
+
const facts: { k: string; v: string }[] = []
|
|
626
|
+
if (c?.clusterIP) facts.push({ k: 'CLUSTER IP', v: c.clusterIP })
|
|
627
|
+
if (c?.serviceType) facts.push({ k: 'TYPE', v: c.serviceType })
|
|
628
|
+
if (c?.ports?.length) facts.push({ k: 'PORTS', v: c.ports.map((x) => `${x.port}→${x.targetPort ?? x.port}`).join(', ') })
|
|
629
|
+
if (c?.addresses?.length) facts.push({ k: 'ADDRESS', v: c.addresses.join(', ') })
|
|
630
|
+
if (c?.hostnames?.length) facts.push({ k: 'HOSTS', v: c.hostnames.join(', ') })
|
|
631
|
+
if (c?.selector) facts.push({ k: 'SELECTOR', v: Object.entries(c.selector).map(([k, v]) => `${k}=${v}`).join(', ') })
|
|
632
|
+
if (node.ref?.namespace) facts.push({ k: 'NAMESPACE', v: node.ref.namespace })
|
|
633
|
+
|
|
634
|
+
return {
|
|
635
|
+
kind: node.kind,
|
|
636
|
+
name: node.name,
|
|
637
|
+
chipTone: node.tone,
|
|
638
|
+
chipText: node.dim ? 'not on this path' : '',
|
|
639
|
+
body: node.dim
|
|
640
|
+
? 'This entry point is attached to the resource but does not serve the host being tested.'
|
|
641
|
+
: findings.length > 0
|
|
642
|
+
? findings[0].cause || findings[0].message
|
|
643
|
+
: 'Configuration and health of this resource.',
|
|
644
|
+
facts,
|
|
645
|
+
notProve: [],
|
|
646
|
+
openRef: node.ref,
|
|
647
|
+
}
|
|
648
|
+
}
|
|
649
|
+
|
|
650
|
+
/**
|
|
651
|
+
* Builds the sidebar. The diagnosis is always computed; a node selection only
|
|
652
|
+
* appends to it.
|
|
653
|
+
*/
|
|
654
|
+
export function buildSidebar(sel: Selection, ctx: Ctx): Sidebar {
|
|
655
|
+
const path = pathSection(ctx)
|
|
656
|
+
// The selected route's own chain, from the graph's traversal - NOT every node
|
|
657
|
+
// sorted by position. Sorting served siblings up as if they were sequential
|
|
658
|
+
// hops and included branches this route never touches.
|
|
659
|
+
const byId = new Map(ctx.nodes.map((n) => [n.id, n]))
|
|
660
|
+
const journey = (ctx.pathNodeIds ?? [])
|
|
661
|
+
.map((id) => byId.get(id))
|
|
662
|
+
.filter((n): n is GraphNode => !!n && !n.isOrigin)
|
|
663
|
+
// Display order interleaves non-network nodes (the workload) after their
|
|
664
|
+
// parent - present to read, absent from the journey's state machine.
|
|
665
|
+
const chain: GraphNode[] = []
|
|
666
|
+
for (const n of journey) {
|
|
667
|
+
chain.push(n)
|
|
668
|
+
for (const iv of ctx.interleave ?? []) {
|
|
669
|
+
if (iv.afterId === n.id) {
|
|
670
|
+
const wl = byId.get(iv.id)
|
|
671
|
+
if (wl) chain.push(wl)
|
|
672
|
+
}
|
|
673
|
+
}
|
|
674
|
+
}
|
|
675
|
+
const entryIds = new Set(ctx.journeyEntryNodeIds ?? [])
|
|
676
|
+
const nonNetwork = new Set(ctx.nonNetworkNodeIds ?? [])
|
|
677
|
+
// A boundary break anchors to the EXIT of its source hop (the Service);
|
|
678
|
+
// breakNodeId anchors to a landing node. Either way one index carries 'break'.
|
|
679
|
+
const anchorId = ctx.breakNodeId ?? ctx.breakAtExitOf
|
|
680
|
+
const breakIdx = anchorId ? chain.findIndex((n) => n.id === anchorId) : -1
|
|
681
|
+
// A break at ONE of N parallel entries orders nothing: a sibling entry is
|
|
682
|
+
// not "before" it, and the hops behind the stage were not necessarily
|
|
683
|
+
// stopped - the request may have gone through a sibling. Serial
|
|
684
|
+
// before/after semantics apply only when the break is outside the stage.
|
|
685
|
+
const breakInParallelStage =
|
|
686
|
+
breakIdx >= 0 && (ctx.entryParallelCount ?? 0) > 1 && entryIds.has(chain[breakIdx]?.id ?? '')
|
|
687
|
+
const stateOf = (i: number, id: string): HopState => {
|
|
688
|
+
// The workload is not a network hop: it is never "reached", never a place
|
|
689
|
+
// a request stops - whatever happens around it.
|
|
690
|
+
if (nonNetwork.has(id)) return 'plain'
|
|
691
|
+
if (breakIdx < 0) return 'plain'
|
|
692
|
+
if (breakInParallelStage) return i === breakIdx ? 'break' : 'plain'
|
|
693
|
+
if (i < breakIdx) return 'before'
|
|
694
|
+
return i === breakIdx ? 'break' : 'after'
|
|
695
|
+
}
|
|
696
|
+
// What to open when the reader has not chosen: the break if there is one,
|
|
697
|
+
// else the destination - the point of the path. Never everything at once,
|
|
698
|
+
// which is how a report becomes a wall.
|
|
699
|
+
const defaultOpen = breakIdx >= 0 ? breakIdx : chain.length - 1
|
|
700
|
+
return {
|
|
701
|
+
path,
|
|
702
|
+
hops: chain.map((n, i) => {
|
|
703
|
+
const section = resourceSection(n)
|
|
704
|
+
return {
|
|
705
|
+
...section,
|
|
706
|
+
id: n.id,
|
|
707
|
+
state: stateOf(i, n.id),
|
|
708
|
+
// Exit-phrased for a boundary break: the SERVICE hop carries it, and
|
|
709
|
+
// "stopped here" would blame the hop itself for its exit's failure.
|
|
710
|
+
chipText:
|
|
711
|
+
i === breakIdx && ctx.breakAtExitOf === n.id
|
|
712
|
+
? 'routing to its Pods breaks just past here'
|
|
713
|
+
: section.chipText,
|
|
714
|
+
parallelCount: entryIds.has(n.id) && (ctx.entryParallelCount ?? 0) > 1 ? ctx.entryParallelCount : undefined,
|
|
715
|
+
expanded: sel ? n.id === sel : i === defaultOpen,
|
|
716
|
+
}
|
|
717
|
+
}),
|
|
718
|
+
context: contextGroup(ctx, byId, sel),
|
|
719
|
+
}
|
|
720
|
+
}
|
|
721
|
+
|
|
722
|
+
/** The headline verdict band. Derived from the selected scenario, never from an
|
|
723
|
+
* aggregate that could hide a failing route behind passing siblings. */
|
|
724
|
+
const VERDICT_TONE: Record<string, SevTone> = { healthy: 'healthy', degraded: 'degraded', broken: 'unhealthy', unknown: 'unknown' }
|
|
725
|
+
|
|
726
|
+
export function buildVerdict(
|
|
727
|
+
trace: Trace,
|
|
728
|
+
route: RouteResult | undefined,
|
|
729
|
+
opts: { stale?: boolean; running?: boolean; pathLabel?: string; originId?: string; originName?: string } = {},
|
|
730
|
+
): {
|
|
731
|
+
tone: SevTone
|
|
732
|
+
chipText: string
|
|
733
|
+
/** Set when the resource has more than one path, so the reader can tell the
|
|
734
|
+
* route-scoped badge apart from the resource-wide headline beside it. */
|
|
735
|
+
chipScope?: string
|
|
736
|
+
scopeLabel?: string
|
|
737
|
+
/** The selected vantage's name - the viewing strip renders it. */
|
|
738
|
+
originName?: string
|
|
739
|
+
title: string
|
|
740
|
+
problem?: string
|
|
741
|
+
body: string
|
|
742
|
+
} {
|
|
743
|
+
// With no route there is nothing to derive a tone from, but the backend has
|
|
744
|
+
// still reached a verdict (e.g. a config fault found without probing). Falling
|
|
745
|
+
// through to 'unknown' showed a grey dot on a resource the tracer called
|
|
746
|
+
// degraded.
|
|
747
|
+
// The band sits directly above the inspector, which reads the SELECTED
|
|
748
|
+
// origin's own result. Leaving the band on the merged rollup let the two
|
|
749
|
+
// contradict each other in adjacent panes - "could not get through" over
|
|
750
|
+
// "got through" - on exactly the disagreeing traces per-vantage evidence
|
|
751
|
+
// exists to represent.
|
|
752
|
+
const seen = opts.originId ? routeAsSeenFrom(route, opts.originId) : route
|
|
753
|
+
// A route exists but this origin has no row: the badge says "not tested",
|
|
754
|
+
// and it must wear the NEUTRAL tone - painting it with the resource-wide
|
|
755
|
+
// verdict colour (amber on a degraded trace) dressed a vantage-scoped
|
|
756
|
+
// "not tested" as a warning about that vantage. The verdict fallback stays
|
|
757
|
+
// for traces with no routes at all (a config fault found without probing).
|
|
758
|
+
const tone: SevTone = opts.running ? 'info' : seen ? routeTone(seen, opts) : route ? 'unknown' : VERDICT_TONE[trace.verdict] ?? 'unknown'
|
|
759
|
+
return {
|
|
760
|
+
tone,
|
|
761
|
+
chipText: opts.running ? 'testing' : seen ? routeChip(seen, opts) : 'not tested',
|
|
762
|
+
// Title, problem and coverage describe the WHOLE resource; tone and chip
|
|
763
|
+
// follow the selected path. With several paths on screen that difference is
|
|
764
|
+
// invisible unless each side says which scope it speaks for.
|
|
765
|
+
// The badge now follows the selected path AND vantage while the headline,
|
|
766
|
+
// problem and coverage stay resource-wide. Both scopes are legitimate; what
|
|
767
|
+
// is not legitimate is leaving the reader to guess which is which, so the
|
|
768
|
+
// badge names its own scope in full.
|
|
769
|
+
// Prefixed here rather than at the render site: with no pathLabel the old
|
|
770
|
+
// template produced the visible "for from Radar on your machine".
|
|
771
|
+
chipScope: [opts.pathLabel ? `for ${opts.pathLabel}` : '', opts.originName ? `from ${opts.originName}` : '']
|
|
772
|
+
.filter(Boolean)
|
|
773
|
+
.join(' · ') || undefined,
|
|
774
|
+
scopeLabel: opts.pathLabel || opts.originName ? 'THIS RESOURCE' : undefined,
|
|
775
|
+
originName: opts.originName,
|
|
776
|
+
// A stale screen previously led with the old headline ("Reachable...") and
|
|
777
|
+
// then said underneath that the result was excluded. That is a contradiction,
|
|
778
|
+
// not an exclusion.
|
|
779
|
+
title: opts.stale
|
|
780
|
+
? 'This result is out of date — re-test'
|
|
781
|
+
: trace.headline || route?.route || `Reachability · ${trace.subject.name}`,
|
|
782
|
+
// The diagnosis is a named fault with a culprit and a next action - it
|
|
783
|
+
// answers "why not", where the headline only says how much was tested. It
|
|
784
|
+
// is called out rather than rendered as body prose, which made the more
|
|
785
|
+
// important fact read as an explanation of the less important one.
|
|
786
|
+
// A diagnosis that only restates the headline is not a second fact. The
|
|
787
|
+
// banner rendered "couldn't actively test any route from here" directly
|
|
788
|
+
// under a title saying exactly that.
|
|
789
|
+
problem: restatesTitle(
|
|
790
|
+
trace.diagnosis?.summary,
|
|
791
|
+
opts.stale ? '' : trace.headline || route?.route || `Reachability · ${trace.subject.name}`,
|
|
792
|
+
)
|
|
793
|
+
? undefined
|
|
794
|
+
: trace.diagnosis?.summary,
|
|
795
|
+
body: trace.diagnosis ? '' : trace.reason || '',
|
|
796
|
+
}
|
|
797
|
+
}
|