sparkforensics-mcp 0.3.0 → 0.5.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/bin/sparkforensics-mcp.mjs +41 -10
- package/package.json +1 -1
- package/vendor-core/allocation.js +106 -0
- package/vendor-core/analyzer.js +168 -60
- package/vendor-core/check-coverage.js +88 -0
- package/vendor-core/cli/budgets.js +54 -27
- package/vendor-core/cli/collect-run.js +84 -32
- package/vendor-core/cli/regression-budgets.js +83 -0
- package/vendor-core/cli/threshold-config.js +28 -0
- package/vendor-core/comparison-verdict.js +177 -0
- package/vendor-core/core-source-hash.txt +1 -0
- package/vendor-core/core-usage-locality.js +56 -2
- package/vendor-core/detector-docs.js +58 -0
- package/vendor-core/detectors.js +1094 -500
- package/vendor-core/docs-config.js +0 -36
- package/vendor-core/docs-content/chapters/nav-index.json +31 -0
- package/vendor-core/docs-content/detection/cache.md +3 -2
- package/vendor-core/docs-content/detection/cfg.md +9 -8
- package/vendor-core/docs-content/detection/chrn.md +1 -2
- package/vendor-core/docs-content/detection/cold.md +4 -2
- package/vendor-core/docs-content/detection/cstor.md +9 -0
- package/vendor-core/docs-content/detection/fail.md +3 -2
- package/vendor-core/docs-content/detection/gc.md +3 -2
- package/vendor-core/docs-content/detection/host.md +2 -1
- package/vendor-core/docs-content/detection/local.md +1 -1
- package/vendor-core/docs-content/detection/mem.md +5 -2
- package/vendor-core/docs-content/detection/plan.md +2 -1
- package/vendor-core/docs-content/detection/sfail.md +2 -1
- package/vendor-core/docs-content/detection/shape.md +5 -4
- package/vendor-core/docs-content/detection/skew.md +3 -1
- package/vendor-core/docs-content/detection/slow.md +2 -2
- package/vendor-core/docs-content/detection/spec.md +2 -3
- package/vendor-core/docs-content/detection/spill.md +1 -1
- package/vendor-core/docs-site-config.js +3 -0
- package/vendor-core/effective-conf.js +107 -0
- package/vendor-core/efficiency-model.js +8 -6
- package/vendor-core/event-handlers.js +321 -44
- package/vendor-core/event-schemas.js +23 -0
- package/vendor-core/evidence-report.js +432 -115
- package/vendor-core/export-data.js +79 -6
- package/vendor-core/finding-action-label.js +9 -88
- package/vendor-core/finding-filter-predicate.js +9 -0
- package/vendor-core/finding-generic-recommendation.js +26 -105
- package/vendor-core/finding-names.js +28 -45
- package/vendor-core/finding-presentation.js +368 -0
- package/vendor-core/finding-tag-help.js +110 -0
- package/vendor-core/finding-types.js +373 -0
- package/vendor-core/findings-of-type.js +11 -0
- package/vendor-core/format-utils.js +96 -30
- package/vendor-core/html-export.js +51 -0
- package/vendor-core/impact-band.js +21 -8
- package/vendor-core/impact-estimator.js +25 -520
- package/vendor-core/impact-format.js +115 -0
- package/vendor-core/impact-model.js +197 -0
- package/vendor-core/ingest.js +6 -2
- package/vendor-core/intervals.js +13 -0
- package/vendor-core/list-runs.js +7 -5
- package/vendor-core/load-vendored.js +70 -5
- package/vendor-core/mcp-server-factory.js +14 -10
- package/vendor-core/mcp-tools.js +105 -45
- package/vendor-core/model-assembler.js +35 -1
- package/vendor-core/occupancy.js +1 -1
- package/vendor-core/parser-worker.js +2 -2
- package/vendor-core/plan-graph-model.js +3 -2
- package/vendor-core/plan-node-detail.js +1 -1
- package/vendor-core/proxy.js +3 -1
- package/vendor-core/python-stage.js +25 -0
- package/vendor-core/recommendation-rollup.js +70 -3
- package/vendor-core/redact.js +96 -37
- package/vendor-core/remediation.js +20 -0
- package/vendor-core/run-comparison.js +73 -29
- package/vendor-core/run-interpretation.js +291 -0
- package/vendor-core/run-metrics.js +198 -0
- package/vendor-core/run-outcome.js +74 -0
- package/vendor-core/run-payload.js +17 -0
- package/vendor-core/run-shape.js +40 -0
- package/vendor-core/run-totals.js +24 -0
- package/vendor-core/run-verdict.js +352 -0
- package/vendor-core/scaling-sim.js +4 -5
- package/vendor-core/scorecard-estimates.js +63 -0
- package/vendor-core/session-snapshot.js +7 -0
- package/vendor-core/shs-schemas.js +2 -2
- package/vendor-core/spark-memory.js +17 -0
- package/vendor-core/sql-stages.js +11 -0
- package/vendor-core/stage-plan-nodes.js +18 -0
- package/vendor-core/stage-quantiles.js +6 -0
- package/vendor-core/threshold-overrides.js +160 -0
- package/vendor-core/threshold-summary.js +11 -33
- package/vendor-core/types.js +54 -42
- package/vendor-core/wall-clock.js +1 -12
- package/vendor-core/wasted-core-hours.js +12 -9
- package/vendor-core/write-targets.js +312 -0
|
@@ -0,0 +1,368 @@
|
|
|
1
|
+
// How each finding type is presented: the one place a finding type's name, board tag, action
|
|
2
|
+
// label, clean-check threshold summary and generic recommendation are registered. FINDING_NAMES,
|
|
3
|
+
// TYPE_TAG_MAP, getThresholdSummary, findingActionLabel and coreFindingGenericRecommendation all
|
|
4
|
+
// read this table.
|
|
5
|
+
//
|
|
6
|
+
// It sits beside DETECTORS rather than on its entries because the HTML export renders names, tags
|
|
7
|
+
// and labels but may not reach detectors.ts (scripts/export-analysis-guard.mjs). The import from
|
|
8
|
+
// detectors.ts is type-only, so it is erased at build time. A detector's scope, order and emits
|
|
9
|
+
// list stay on its DETECTORS entry; renderers get them through detectorInfoByType().
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
/** A share threshold as captions and caveats state it: 0.005 -> "0.5%", never float noise like 7.000000000000001%. */
|
|
31
|
+
export const shareLabel = (share ) => `${Math.round(share * 1e6) / 1e4}%`;
|
|
32
|
+
|
|
33
|
+
// Whether the detector found `key` already logged on for this run. Its switchFix (detectors.ts)
|
|
34
|
+
// then worded the row's own text for that case and left the property out of the remediation, so
|
|
35
|
+
// a generic line reads the same decision and never recommends a switch the row says is on.
|
|
36
|
+
// A finding with no remediation (older or hand-built data) keeps the property wording.
|
|
37
|
+
function switchAlreadyOn(finding , key ) {
|
|
38
|
+
return finding.remediation != null && !finding.remediation.some((r) => r.key === key);
|
|
39
|
+
}
|
|
40
|
+
|
|
41
|
+
const SKEW_JOIN_KEY = 'spark.sql.adaptive.skewJoin.enabled';
|
|
42
|
+
const SKEW_JOIN_ALREADY_ON = 'AQE skew-join handling is already on, so salt the key or repartition on a better key.';
|
|
43
|
+
const SKEW_JOIN_AQE_OFF = 'AQE is off, so enable it (spark.sql.adaptive.enabled) for skew-join handling to apply; otherwise salt the key or repartition on a better key.';
|
|
44
|
+
|
|
45
|
+
// The skew-join generic line, worded per the row's remediation: AQE logged off, switch already on, or neither.
|
|
46
|
+
function skewJoinGeneric(f , unset ) {
|
|
47
|
+
if (f.remediation?.some((r) => r.key === 'spark.sql.adaptive.enabled')) return SKEW_JOIN_AQE_OFF;
|
|
48
|
+
return switchAlreadyOn(f, SKEW_JOIN_KEY) ? SKEW_JOIN_ALREADY_ON : unset;
|
|
49
|
+
}
|
|
50
|
+
const DYNAMIC_ALLOCATION_KEY = 'spark.dynamicAllocation.enabled';
|
|
51
|
+
|
|
52
|
+
// The four configAudit DETECTORS entries share this row, one per audited property.
|
|
53
|
+
const CONFIG_AUDIT_PRESENTATION = {
|
|
54
|
+
name: 'config audit',
|
|
55
|
+
tag: 'CFG',
|
|
56
|
+
thresholdSummary: () => 'a Spark conf value outside the recommended range',
|
|
57
|
+
actionLabel(f) {
|
|
58
|
+
switch (f.property) {
|
|
59
|
+
case 'spark.shuffle.service.enabled': return 'Enable shuffle service';
|
|
60
|
+
case 'spark.dynamicAllocation.minExecutors': return 'Fix autoscaling bounds';
|
|
61
|
+
case 'spark.dynamicAllocation.maxExecutors': return 'Set max executors';
|
|
62
|
+
case 'spark.serializer': return 'Switch to Kryo';
|
|
63
|
+
case 'spark.executor.memoryOverhead': return 'Raise memory overhead';
|
|
64
|
+
}
|
|
65
|
+
return undefined;
|
|
66
|
+
},
|
|
67
|
+
genericRecommendation(f) {
|
|
68
|
+
switch (f.property) {
|
|
69
|
+
case 'spark.shuffle.service.enabled': return 'Set spark.shuffle.service.enabled=true so shuffle data survives executor removal.';
|
|
70
|
+
case 'spark.dynamicAllocation.minExecutors': return 'Set the minimum executor bound at or below the maximum.';
|
|
71
|
+
case 'spark.dynamicAllocation.maxExecutors': return 'Set spark.dynamicAllocation.maxExecutors to cap cluster growth.';
|
|
72
|
+
case 'spark.serializer': return 'Consider spark.serializer=org.apache.spark.serializer.KryoSerializer for faster, smaller buffers.';
|
|
73
|
+
case 'spark.executor.memoryOverhead': return 'Raise executor memoryOverhead above Spark\'s default floor to avoid off-heap OOM-kills.';
|
|
74
|
+
}
|
|
75
|
+
return undefined;
|
|
76
|
+
},
|
|
77
|
+
};
|
|
78
|
+
|
|
79
|
+
/** One row per emitted finding type: the mapped type makes a missing or stray row a compile error. */
|
|
80
|
+
export const FINDING_PRESENTATION = {
|
|
81
|
+
incompleteRun: {
|
|
82
|
+
name: 'incomplete run',
|
|
83
|
+
tag: 'INCMP',
|
|
84
|
+
thresholdSummary: () => 'an event log missing its terminal ApplicationEnd event',
|
|
85
|
+
actionLabel: () => undefined,
|
|
86
|
+
genericRecommendation: () => undefined,
|
|
87
|
+
},
|
|
88
|
+
|
|
89
|
+
skew: {
|
|
90
|
+
name: 'task skew',
|
|
91
|
+
tag: 'SKEW',
|
|
92
|
+
thresholdSummary: (t) => `P95 task time over ${t.ratioWarn}× the median (the longest task on stages under ${t.minTasksForP95} tasks)`,
|
|
93
|
+
actionLabel: () => 'Fix task skew',
|
|
94
|
+
genericRecommendation: (f) => skewJoinGeneric(f,
|
|
95
|
+
'For join-driven skew, enable AQE skew-join handling (spark.sql.adaptive.skewJoin.enabled); otherwise salt the key or repartition on a better key.'),
|
|
96
|
+
},
|
|
97
|
+
stageShape: {
|
|
98
|
+
name: 'stage shape',
|
|
99
|
+
tag: 'SHAPE',
|
|
100
|
+
thresholdSummary: (t) => `under ${t.pRatioMax} tasks per core, output over ${t.oiRatioMax}× input, or one task spanning over ${Math.round(t.stageShareMin * 100)}% of the stage's wall-clock at over ${t.skewWarn}× the median task`,
|
|
101
|
+
actionLabel(f) {
|
|
102
|
+
switch (f.rule) {
|
|
103
|
+
case 'lowParallelism': return 'Increase parallelism';
|
|
104
|
+
case 'dataExplosion': return 'Check for exploding join';
|
|
105
|
+
case 'taskStageSkew': return 'Fix straggler task';
|
|
106
|
+
}
|
|
107
|
+
return undefined;
|
|
108
|
+
},
|
|
109
|
+
genericRecommendation(f) {
|
|
110
|
+
switch (f.rule) {
|
|
111
|
+
case 'lowParallelism': return 'Too few tasks run relative to the cores available, leaving cluster capacity idle: repartition to use more of it.';
|
|
112
|
+
case 'dataExplosion': return 'Output volume far exceeds input volume: check for an exploding join or a cross product.';
|
|
113
|
+
case 'taskStageSkew': return 'A single straggler task gates the whole stage\'s wall-clock duration.';
|
|
114
|
+
}
|
|
115
|
+
return undefined;
|
|
116
|
+
},
|
|
117
|
+
},
|
|
118
|
+
tinyTask: {
|
|
119
|
+
name: 'tiny tasks',
|
|
120
|
+
tag: 'TINY',
|
|
121
|
+
thresholdSummary: (t) => `${t.minTasks}+ tasks with a median of ${t.maxP50}ms or less and a P95 of ${t.maxP95}ms or less`,
|
|
122
|
+
actionLabel: () => 'Coalesce small tasks',
|
|
123
|
+
// The shuffle-vs-no-shuffle fix isn't a Finding field, so one sentence covers both.
|
|
124
|
+
genericRecommendation: () => 'Scheduler overhead may dominate: lower spark.sql.shuffle.partitions, or coalesce down to fewer, larger tasks.',
|
|
125
|
+
},
|
|
126
|
+
|
|
127
|
+
shuffle: {
|
|
128
|
+
name: 'shuffle I/O',
|
|
129
|
+
tag: 'SHFL',
|
|
130
|
+
thresholdSummary: (t) => `stage shuffle read above ${t.minBytes / 1048576} MiB`,
|
|
131
|
+
actionLabel: () => 'Reduce shuffle size',
|
|
132
|
+
genericRecommendation: () => 'Raise spark.sql.shuffle.partitions, or use a broadcast join for the smaller side.',
|
|
133
|
+
},
|
|
134
|
+
partitionSizing: {
|
|
135
|
+
name: 'partition sizing',
|
|
136
|
+
tag: 'PART',
|
|
137
|
+
thresholdSummary: (t) => `a shuffle partition over ${t.skewRatio}× the median or over ${t.maxPartBytes / 1073741824} GiB, or ${t.lowParTotalBytes / 1073741824} GiB of shuffle on ${t.lowParMaxTasks} tasks or fewer`,
|
|
138
|
+
actionLabel(f) {
|
|
139
|
+
switch (f.rule) {
|
|
140
|
+
case 'shufflePartitionSkew': return 'Fix skewed partition';
|
|
141
|
+
case 'lowShuffleParallelism': return 'Add shuffle partitions';
|
|
142
|
+
case 'maxPartitionTooBig': return 'Repartition oversized data';
|
|
143
|
+
}
|
|
144
|
+
return undefined;
|
|
145
|
+
},
|
|
146
|
+
genericRecommendation(f) {
|
|
147
|
+
switch (f.rule) {
|
|
148
|
+
case 'shufflePartitionSkew': return skewJoinGeneric(f, 'For join skew, enable AQE skew-join handling (spark.sql.adaptive.skewJoin.enabled); otherwise salt the key or repartition on a better key.');
|
|
149
|
+
case 'lowShuffleParallelism': return switchAlreadyOn(f, 'spark.sql.shuffle.partitions')
|
|
150
|
+
? "spark.sql.shuffle.partitions is already high enough, so raise this stage's own partition count (its repartition(n) or RDD parallelism) so each partition is smaller."
|
|
151
|
+
: 'Raise spark.sql.shuffle.partitions so each partition is smaller.';
|
|
152
|
+
case 'maxPartitionTooBig': return 'Repartition to break up the oversized partition before this stage.';
|
|
153
|
+
}
|
|
154
|
+
return undefined;
|
|
155
|
+
},
|
|
156
|
+
},
|
|
157
|
+
|
|
158
|
+
spill: {
|
|
159
|
+
name: 'spill',
|
|
160
|
+
tag: 'SPILL',
|
|
161
|
+
thresholdSummary: (t) => `single-task disk spill above ${t.singleTaskDiskGiB} GiB`,
|
|
162
|
+
actionLabel: () => 'Reduce spill',
|
|
163
|
+
// The skew/volume classification isn't a Finding field, so one sentence covers both.
|
|
164
|
+
genericRecommendation: () => 'If the spill is skew-driven, fix task skew first: adding memory will not help. Otherwise raise spark.sql.shuffle.partitions or increase executor memory.',
|
|
165
|
+
},
|
|
166
|
+
|
|
167
|
+
gc: {
|
|
168
|
+
name: 'GC pressure',
|
|
169
|
+
tag: 'GC',
|
|
170
|
+
thresholdSummary: (t) => `GC above ${t.warnPct100}% (or below ${t.lowInfoPct100}%) of executor run time`,
|
|
171
|
+
actionLabel: (f) => (f.direction === 'low' ? 'Right-size executor memory' : 'Reduce GC pressure'),
|
|
172
|
+
genericRecommendation: (f) => (f.direction === 'low'
|
|
173
|
+
? 'Memory may be over-provisioned here: consider reducing spark.executor.memory for cost savings.'
|
|
174
|
+
: 'Reduce object creation, use primitive types, avoid UDFs, or increase executor memory to cut GC time.'),
|
|
175
|
+
},
|
|
176
|
+
|
|
177
|
+
stageFailed: {
|
|
178
|
+
name: 'failed stage',
|
|
179
|
+
tag: 'SFAIL',
|
|
180
|
+
thresholdSummary: () => 'a stage that failed outright',
|
|
181
|
+
actionLabel: () => 'Inspect stage failure',
|
|
182
|
+
genericRecommendation: () => 'Inspect the driver log for the failure reason and the job that triggered it.',
|
|
183
|
+
},
|
|
184
|
+
failures: {
|
|
185
|
+
name: 'failed tasks',
|
|
186
|
+
tag: 'FAIL',
|
|
187
|
+
thresholdSummary: (t) => `over ${shareLabel(t.warnRate)} of a stage's tasks failing`,
|
|
188
|
+
actionLabel: () => 'Investigate task failures',
|
|
189
|
+
genericRecommendation: () => 'Investigate driver logs for executor instability or data-driven errors.',
|
|
190
|
+
},
|
|
191
|
+
retryWaste: {
|
|
192
|
+
name: 'retry waste',
|
|
193
|
+
tag: 'RETRY',
|
|
194
|
+
thresholdSummary: (t) => `${t.minWasted}+ retried attempts wasting at least ${t.minWasteMs / 1000}s`,
|
|
195
|
+
actionLabel: () => 'Investigate retry cause',
|
|
196
|
+
genericRecommendation: () => 'Investigate executor loss or fetch failures behind the retried attempts.',
|
|
197
|
+
},
|
|
198
|
+
|
|
199
|
+
slowHost: {
|
|
200
|
+
name: 'slow executor host',
|
|
201
|
+
tag: 'HOST',
|
|
202
|
+
thresholdSummary: (t) => `a host ${t.ratioWarn}× slower than its peers by mean task time (per-executor figures from ${t.ratioTiers[0]}×)`,
|
|
203
|
+
actionLabel(f) {
|
|
204
|
+
if (f.variant === 'durationShare') return 'Fix data locality';
|
|
205
|
+
if (f.variant === 'multiDim') return 'Investigate degraded executor';
|
|
206
|
+
return 'Check slow host';
|
|
207
|
+
},
|
|
208
|
+
genericRecommendation(f) {
|
|
209
|
+
if (f.variant === 'durationShare') return 'Check for data locality or partition assignment skewing work onto one node.';
|
|
210
|
+
if (f.variant === 'multiDim') return 'Investigate uneven partition assignment or a degraded executor.';
|
|
211
|
+
const check = 'Check what this host was running: it may just hold data locality for its tasks or carry one heavy stage, rather than a hardware fault.';
|
|
212
|
+
return switchAlreadyOn(f, 'spark.speculation')
|
|
213
|
+
? `${check} Speculation is already on, so a lagging task there is already relaunched.`
|
|
214
|
+
: `${check} Enable spark.speculation to relaunch a lagging task automatically.`;
|
|
215
|
+
},
|
|
216
|
+
},
|
|
217
|
+
stageSlowness: {
|
|
218
|
+
name: 'slow stage',
|
|
219
|
+
tag: 'SLOW',
|
|
220
|
+
thresholdSummary: () => 'a stage running far longer than its peers, not attributable to a single slow host',
|
|
221
|
+
actionLabel: () => 'Profile slow stage',
|
|
222
|
+
genericRecommendation: () => 'Often a partition-count problem: raise parallelism via spark.sql.shuffle.partitions or spark.default.parallelism, or check for a large per-task data volume driving heavy shuffle and spill.',
|
|
223
|
+
},
|
|
224
|
+
straggler: {
|
|
225
|
+
name: 'straggling task',
|
|
226
|
+
tag: 'STRAG',
|
|
227
|
+
thresholdSummary: () => 'one or more tasks finishing far after the rest of their stage',
|
|
228
|
+
actionLabel: () => 'Fix stragglers',
|
|
229
|
+
genericRecommendation: () => 'Rule out a GC pause or a slow shuffle fetch before assuming a hardware issue. If a skewed key is the real cause, that is a candidate for AQE\'s skew-join handling.',
|
|
230
|
+
},
|
|
231
|
+
speculationWaste: {
|
|
232
|
+
name: 'speculation waste',
|
|
233
|
+
tag: 'SPEC',
|
|
234
|
+
thresholdSummary: (t) => `${t.minWasted}+ discarded speculative attempts wasting at least ${t.minWasteMs / 1000}s`,
|
|
235
|
+
actionLabel: () => 'Tune speculation settings',
|
|
236
|
+
genericRecommendation: () => 'If task durations are naturally variable rather than genuine stragglers, consider tuning spark.speculation.multiplier/quantile.',
|
|
237
|
+
},
|
|
238
|
+
coldStart: {
|
|
239
|
+
name: 'cold start',
|
|
240
|
+
tag: 'COLD',
|
|
241
|
+
thresholdSummary: (t) => `the first stage waiting over ${t.gapSeconds}s for an executor`,
|
|
242
|
+
actionLabel: () => 'Pre-warm cluster',
|
|
243
|
+
genericRecommendation: () => 'Keep a warm pool of idle executors, or if using dynamic allocation, raise the minimum/initial executor count so it does not scale up from zero.',
|
|
244
|
+
},
|
|
245
|
+
|
|
246
|
+
memoryUtilization: {
|
|
247
|
+
name: 'memory utilization',
|
|
248
|
+
tag: 'MEM',
|
|
249
|
+
thresholdSummary: () => 'executor heap usage outside the configured band',
|
|
250
|
+
actionLabel(f) {
|
|
251
|
+
switch (f.variant) {
|
|
252
|
+
case 'idleCores': return 'Reduce idle cores';
|
|
253
|
+
case 'wasteModel': return 'Right-size executor memory';
|
|
254
|
+
case 'memoryBand':
|
|
255
|
+
if (f.dataUnavailable) return 'Enable memory metrics';
|
|
256
|
+
return f.rule === 'heapNearCapacity' ? 'Increase executor memory' : 'Reduce executor memory';
|
|
257
|
+
}
|
|
258
|
+
return undefined;
|
|
259
|
+
},
|
|
260
|
+
genericRecommendation(f) {
|
|
261
|
+
switch (f.variant) {
|
|
262
|
+
case 'idleCores': return switchAlreadyOn(f, DYNAMIC_ALLOCATION_KEY)
|
|
263
|
+
? 'Dynamic allocation is already on, so reduce cluster size.'
|
|
264
|
+
: 'Reduce cluster size or enable dynamic allocation.';
|
|
265
|
+
case 'wasteModel': return 'Review spark.executor.memory and executor count.';
|
|
266
|
+
case 'memoryBand':
|
|
267
|
+
if (f.dataUnavailable) return undefined;
|
|
268
|
+
return f.rule === 'heapNearCapacity'
|
|
269
|
+
? 'Memory may be too small: raise spark.executor.memory to avoid OOM/spill.'
|
|
270
|
+
: 'Memory may be over-provisioned: consider reducing spark.executor.memory for cost savings.';
|
|
271
|
+
}
|
|
272
|
+
return undefined;
|
|
273
|
+
},
|
|
274
|
+
},
|
|
275
|
+
utilization: {
|
|
276
|
+
name: 'executor utilization',
|
|
277
|
+
tag: 'UTIL',
|
|
278
|
+
thresholdSummary: (t) => `average executor utilization below ${shareLabel(t.minUtil)}`,
|
|
279
|
+
actionLabel: () => 'Reduce cluster size',
|
|
280
|
+
genericRecommendation: (f) => (switchAlreadyOn(f, DYNAMIC_ALLOCATION_KEY)
|
|
281
|
+
? 'Dynamic allocation is already on, so consider reducing cluster size.'
|
|
282
|
+
: 'Consider reducing cluster size or enabling dynamic allocation.'),
|
|
283
|
+
},
|
|
284
|
+
coreLocality: {
|
|
285
|
+
name: 'core locality',
|
|
286
|
+
tag: 'LOCAL',
|
|
287
|
+
thresholdSummary: () => 'task placement missing data-local core assignment',
|
|
288
|
+
actionLabel: () => 'Fix data locality',
|
|
289
|
+
genericRecommendation: () => 'Check executor/data colocation.',
|
|
290
|
+
},
|
|
291
|
+
cachingOpportunity: {
|
|
292
|
+
name: 'caching opportunity',
|
|
293
|
+
tag: 'CACHE',
|
|
294
|
+
thresholdSummary: () => 'a dataset re-read from source multiple times with no cache/persist',
|
|
295
|
+
actionLabel: (f) => (f.variant === 'composite' ? 'Cache repeated result' : 'Cache shared table'),
|
|
296
|
+
genericRecommendation: (f) => (f.variant === 'composite'
|
|
297
|
+
? 'Cache or persist the repeated join/union result so it is computed once instead of recomputed per query.'
|
|
298
|
+
: 'Cache the shared DataFrame, or broadcast it if it is a small join lookup.'),
|
|
299
|
+
},
|
|
300
|
+
cacheUtilization: {
|
|
301
|
+
name: 'cache utilization',
|
|
302
|
+
tag: 'CSTOR',
|
|
303
|
+
thresholdSummary: () => 'cached partitions evicted or spilled to disk',
|
|
304
|
+
actionLabel: (f) => (f.dataUnavailable ? 'Enable block-update logging' : 'Increase cache memory'),
|
|
305
|
+
genericRecommendation(f) {
|
|
306
|
+
switch (f.variant) {
|
|
307
|
+
case 'partialCache': return 'Increase executor memory or reduce the cached dataset size so more of it stays cached.';
|
|
308
|
+
case 'diskSpillover': return 'Executor memory may be too small for this cached dataset: increase executor memory or reduce its size.';
|
|
309
|
+
}
|
|
310
|
+
return undefined;
|
|
311
|
+
},
|
|
312
|
+
},
|
|
313
|
+
jobFailureRate: {
|
|
314
|
+
name: 'job failure rate',
|
|
315
|
+
tag: 'JOBS',
|
|
316
|
+
thresholdSummary: (t) => `at least ${shareLabel(t.infoRate)} of jobs failing`,
|
|
317
|
+
actionLabel: () => 'Investigate failed jobs',
|
|
318
|
+
genericRecommendation: () => 'Inspect the driver log for the failed job(s) and the stage failures that triggered them.',
|
|
319
|
+
},
|
|
320
|
+
autoscalingChurn: {
|
|
321
|
+
name: 'autoscaling churn',
|
|
322
|
+
tag: 'CHRN',
|
|
323
|
+
thresholdSummary: (t) => `over ${shareLabel(t.warningPct)} of executors living under ${t.shortLivedMs / 60000} minutes`,
|
|
324
|
+
actionLabel: () => 'Reduce autoscaling churn',
|
|
325
|
+
genericRecommendation: () => 'This looks like wasteful re-provisioning rather than normal scale-down: consider raising spark.dynamicAllocation.executorIdleTimeout or widening the minExecutors/maxExecutors bounds to reduce flapping.',
|
|
326
|
+
},
|
|
327
|
+
|
|
328
|
+
configAudit: CONFIG_AUDIT_PRESENTATION,
|
|
329
|
+
|
|
330
|
+
duplicatePlanSubtree: {
|
|
331
|
+
name: 'duplicate plan subtree',
|
|
332
|
+
tag: 'PLAN',
|
|
333
|
+
thresholdSummary: () => 'the same physical plan subtree executed more than once',
|
|
334
|
+
actionLabel: () => 'Dedupe repeated subtree',
|
|
335
|
+
// isExchangeRoot isn't a Finding field, so one sentence covers both cases.
|
|
336
|
+
genericRecommendation: () => 'Check whether the repeated subtree could be computed once and reused, or cache/persist the shared computation.',
|
|
337
|
+
},
|
|
338
|
+
smallFiles: {
|
|
339
|
+
name: 'small files',
|
|
340
|
+
tag: 'PLAN',
|
|
341
|
+
thresholdSummary: (t) => `over ${t.minFiles} files averaging under ${t.maxAvgFileSizeMB} MiB`,
|
|
342
|
+
actionLabel: (f) => (f.direction === 'write' ? 'Coalesce output files' : 'Compact small files'),
|
|
343
|
+
genericRecommendation: (f) => (f.direction === 'write'
|
|
344
|
+
? 'Repartition or coalesce before writing to raise the average file size.'
|
|
345
|
+
: 'Compact the upstream output so fewer, larger files are produced.'),
|
|
346
|
+
},
|
|
347
|
+
underBroadcast: {
|
|
348
|
+
name: 'missed broadcast join',
|
|
349
|
+
tag: 'PLAN',
|
|
350
|
+
thresholdSummary: () => 'a join below the configured size floor that skipped broadcast',
|
|
351
|
+
actionLabel: () => 'Use broadcast join',
|
|
352
|
+
genericRecommendation: () => 'This could have been a broadcast join: consider a broadcast() hint or raising spark.sql.autoBroadcastJoinThreshold.',
|
|
353
|
+
},
|
|
354
|
+
overBroadcast: {
|
|
355
|
+
name: 'oversized broadcast join',
|
|
356
|
+
tag: 'PLAN',
|
|
357
|
+
thresholdSummary: (t) => `a broadcast over ${t.overBroadcastBytes / 1073741824} GiB`,
|
|
358
|
+
actionLabel: () => 'Fix oversized broadcast',
|
|
359
|
+
genericRecommendation: (f) => (switchAlreadyOn(f, 'spark.sql.autoBroadcastJoinThreshold')
|
|
360
|
+
? 'Automatic broadcast is already disabled, so remove the broadcast() hint that forced it.'
|
|
361
|
+
: 'Check for a misapplied broadcast hint or a misconfigured spark.sql.autoBroadcastJoinThreshold.'),
|
|
362
|
+
},
|
|
363
|
+
};
|
|
364
|
+
|
|
365
|
+
/** The presentation row for a free-form type string (report JSON, a filter, a test double). */
|
|
366
|
+
export function presentationOf(type ) {
|
|
367
|
+
return (FINDING_PRESENTATION )[type];
|
|
368
|
+
}
|
|
@@ -0,0 +1,110 @@
|
|
|
1
|
+
// Plain-language help per finding tag: the expansion the tag stands for and a one-line description
|
|
2
|
+
// of what it means. Shared by the dashboard (tag tooltips) and the
|
|
3
|
+
// comparison verdict, which names finding categories by their expansion on every path.
|
|
4
|
+
|
|
5
|
+
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
export const TAG_HELP = {
|
|
10
|
+
INCMP: {
|
|
11
|
+
expansion: 'Incomplete run',
|
|
12
|
+
description: 'This event log never recorded an application-end event, so other findings and metrics reflect only what was captured.',
|
|
13
|
+
},
|
|
14
|
+
SKEW: {
|
|
15
|
+
expansion: 'Task skew',
|
|
16
|
+
description: 'A small number of tasks take much longer than their peers.',
|
|
17
|
+
},
|
|
18
|
+
SHFL: {
|
|
19
|
+
expansion: 'Shuffle I/O',
|
|
20
|
+
description: 'Tasks are moving a large amount of intermediate data between stages.',
|
|
21
|
+
},
|
|
22
|
+
SPILL: {
|
|
23
|
+
expansion: 'Memory and disk spill',
|
|
24
|
+
description: 'Tasks are writing data out of memory, which slows execution.',
|
|
25
|
+
},
|
|
26
|
+
GC: {
|
|
27
|
+
expansion: 'Garbage collection pressure',
|
|
28
|
+
description: 'Tasks are spending an unusually large share of time reclaiming memory.',
|
|
29
|
+
},
|
|
30
|
+
COLD: {
|
|
31
|
+
expansion: 'Executor cold start',
|
|
32
|
+
description: 'New executors are taking time to become available for work.',
|
|
33
|
+
},
|
|
34
|
+
UTIL: {
|
|
35
|
+
expansion: 'Low utilization',
|
|
36
|
+
description: 'Allocated executors are idle for a large share of the application run.',
|
|
37
|
+
},
|
|
38
|
+
MEM: {
|
|
39
|
+
expansion: 'Memory utilization',
|
|
40
|
+
description: 'Executor memory or core capacity may be over- or under-provisioned.',
|
|
41
|
+
},
|
|
42
|
+
LOCAL: {
|
|
43
|
+
expansion: 'Core usage locality',
|
|
44
|
+
description: 'Tasks are running without process- or node-local data placement more than expected.',
|
|
45
|
+
},
|
|
46
|
+
HOST: {
|
|
47
|
+
expansion: 'Slow host',
|
|
48
|
+
description: 'One executor is substantially slower than its peers.',
|
|
49
|
+
},
|
|
50
|
+
FAIL: {
|
|
51
|
+
expansion: 'Failed tasks',
|
|
52
|
+
description: 'Tasks are failing often enough to affect the stage.',
|
|
53
|
+
},
|
|
54
|
+
STRAG: {
|
|
55
|
+
expansion: 'Straggler tasks',
|
|
56
|
+
description: 'A few tasks are much slower than the rest of their stage.',
|
|
57
|
+
},
|
|
58
|
+
SPEC: {
|
|
59
|
+
expansion: 'Speculation waste',
|
|
60
|
+
description: 'Speculative task attempts used a lot of executor time without confirming a genuine straggler.',
|
|
61
|
+
},
|
|
62
|
+
RETRY: {
|
|
63
|
+
expansion: 'Retry waste',
|
|
64
|
+
description: 'Repeated task attempts are consuming avoidable execution time.',
|
|
65
|
+
},
|
|
66
|
+
TINY: {
|
|
67
|
+
expansion: 'Tiny tasks',
|
|
68
|
+
description: 'Many very short tasks are adding scheduling overhead.',
|
|
69
|
+
},
|
|
70
|
+
SFAIL: {
|
|
71
|
+
expansion: 'Failed stage',
|
|
72
|
+
description: 'A stage attempt failed outright.',
|
|
73
|
+
},
|
|
74
|
+
PART: {
|
|
75
|
+
expansion: 'Partition sizing',
|
|
76
|
+
description: 'Shuffle partitions are too large, too uneven, or too few for the work.',
|
|
77
|
+
},
|
|
78
|
+
SLOW: {
|
|
79
|
+
expansion: 'Stage slowness',
|
|
80
|
+
description: 'A stage is slow overall without a more specific diagnosed cause.',
|
|
81
|
+
},
|
|
82
|
+
SHAPE: {
|
|
83
|
+
expansion: 'Stage shape',
|
|
84
|
+
description: 'The stage has an inefficient task count, output shape, or task-to-stage balance.',
|
|
85
|
+
},
|
|
86
|
+
CACHE: {
|
|
87
|
+
expansion: 'Caching opportunity',
|
|
88
|
+
description: 'A reusable dataset may benefit from being persisted between stages.',
|
|
89
|
+
},
|
|
90
|
+
CSTOR: {
|
|
91
|
+
expansion: 'Cache storage',
|
|
92
|
+
description: 'A persisted dataset is not fully cached in memory or is spilling to disk.',
|
|
93
|
+
},
|
|
94
|
+
CHRN: {
|
|
95
|
+
expansion: 'Autoscaling churn',
|
|
96
|
+
description: 'Executors are being stood up and torn down again before they can do useful work.',
|
|
97
|
+
},
|
|
98
|
+
JOBS: {
|
|
99
|
+
expansion: 'Job failure rate',
|
|
100
|
+
description: 'A large share of completed jobs did not succeed.',
|
|
101
|
+
},
|
|
102
|
+
CFG: {
|
|
103
|
+
expansion: 'Configuration audit',
|
|
104
|
+
description: 'Configuration settings may cause reliability or efficiency problems.',
|
|
105
|
+
},
|
|
106
|
+
PLAN: {
|
|
107
|
+
expansion: 'Plan advisor',
|
|
108
|
+
description: 'The SQL execution plan has a pattern worth reviewing.',
|
|
109
|
+
},
|
|
110
|
+
};
|