sparkforensics-mcp 0.2.4 → 0.4.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -0
- package/README.md +7 -1
- package/bin/sparkforensics-mcp.mjs +41 -10
- package/package.json +1 -1
- package/vendor-core/analyzer.js +156 -48
- package/vendor-core/check-coverage.js +88 -0
- package/vendor-core/cli/budgets.js +31 -18
- package/vendor-core/cli/collect-run.js +76 -31
- package/vendor-core/cli/native-zstd.js +2 -2
- package/vendor-core/cli/threshold-config.js +28 -0
- package/vendor-core/comparison-verdict.js +177 -0
- package/vendor-core/core-source-hash.txt +1 -0
- package/vendor-core/core-usage-locality.js +56 -2
- package/vendor-core/detector-docs.js +58 -0
- package/vendor-core/detectors.js +933 -459
- package/vendor-core/docs-config.js +0 -36
- package/vendor-core/docs-content/chapters/nav-index.json +31 -0
- package/vendor-core/docs-content/detection/cstor.md +9 -0
- package/vendor-core/docs-content/detection/fail.md +6 -2
- package/vendor-core/docs-site-config.js +3 -0
- package/vendor-core/event-handlers.js +191 -6
- package/vendor-core/event-schemas.js +29 -0
- package/vendor-core/evidence-report.js +440 -112
- package/vendor-core/export-data.js +79 -6
- package/vendor-core/finding-action-label.js +9 -88
- package/vendor-core/finding-filter-predicate.js +9 -0
- package/vendor-core/finding-generic-recommendation.js +6 -104
- package/vendor-core/finding-names.js +21 -45
- package/vendor-core/finding-presentation.js +333 -0
- package/vendor-core/finding-tag-help.js +110 -0
- package/vendor-core/finding-types.js +361 -0
- package/vendor-core/findings-of-type.js +11 -0
- package/vendor-core/format-utils.js +92 -27
- package/vendor-core/html-export.js +51 -0
- package/vendor-core/impact-band.js +21 -8
- package/vendor-core/impact-estimator.js +8 -521
- package/vendor-core/impact-format.js +114 -0
- package/vendor-core/impact-model.js +175 -0
- package/vendor-core/ingest.js +2 -0
- package/vendor-core/intervals.js +13 -0
- package/vendor-core/list-runs.js +2 -3
- package/vendor-core/load-vendored.js +70 -5
- package/vendor-core/mcp-server-factory.js +14 -10
- package/vendor-core/mcp-tools.js +105 -45
- package/vendor-core/model-assembler.js +12 -0
- package/vendor-core/occupancy.js +1 -1
- package/vendor-core/parser-worker.js +22 -5
- package/vendor-core/plan-graph-model.js +3 -2
- package/vendor-core/plan-node-detail.js +1 -1
- package/vendor-core/recommendation-rollup.js +63 -3
- package/vendor-core/redact.js +68 -28
- package/vendor-core/run-comparison.js +40 -7
- package/vendor-core/run-interpretation.js +290 -0
- package/vendor-core/run-outcome.js +74 -0
- package/vendor-core/run-payload.js +17 -0
- package/vendor-core/run-shape.js +40 -0
- package/vendor-core/run-verdict.js +353 -0
- package/vendor-core/scaling-sim.js +4 -5
- package/vendor-core/scorecard-estimates.js +62 -0
- package/vendor-core/shs-fetch.js +175 -65
- package/vendor-core/shs-load.js +1 -1
- package/vendor-core/sql-stages.js +11 -0
- package/vendor-core/stage-quantiles.js +14 -0
- package/vendor-core/task-failure.js +151 -0
- package/vendor-core/threshold-overrides.js +160 -0
- package/vendor-core/threshold-summary.js +11 -33
- package/vendor-core/types.js +6 -42
- package/vendor-core/vendor/fflate.js +1 -1
- package/vendor-core/wall-clock.js +1 -12
- package/vendor-core/wasted-core-hours.js +2 -2
- package/vendor-core/zip-archive.js +167 -0
|
@@ -0,0 +1,333 @@
|
|
|
1
|
+
// How each finding type is presented: the one place a finding type's name, board tag, action
|
|
2
|
+
// label, clean-check threshold summary and generic recommendation are registered. FINDING_NAMES,
|
|
3
|
+
// TYPE_TAG_MAP, getThresholdSummary, findingActionLabel and coreFindingGenericRecommendation all
|
|
4
|
+
// read this table.
|
|
5
|
+
//
|
|
6
|
+
// It sits beside DETECTORS rather than on its entries because the HTML export renders names, tags
|
|
7
|
+
// and labels but may not reach detectors.ts (scripts/export-analysis-guard.mjs). The import from
|
|
8
|
+
// detectors.ts is type-only, so it is erased at build time. A detector's scope, order and emits
|
|
9
|
+
// list stay on its DETECTORS entry; renderers get them through detectorInfoByType().
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
// The four configAudit DETECTORS entries share this row, one per audited property.
|
|
30
|
+
const CONFIG_AUDIT_PRESENTATION = {
|
|
31
|
+
name: 'config audit',
|
|
32
|
+
tag: 'CFG',
|
|
33
|
+
thresholdSummary: () => 'a Spark conf value outside the recommended range',
|
|
34
|
+
actionLabel(f) {
|
|
35
|
+
switch (f.property) {
|
|
36
|
+
case 'spark.shuffle.service.enabled': return 'Enable shuffle service';
|
|
37
|
+
case 'spark.dynamicAllocation.minExecutors': return 'Fix autoscaling bounds';
|
|
38
|
+
case 'spark.dynamicAllocation.maxExecutors': return 'Set max executors';
|
|
39
|
+
case 'spark.serializer': return 'Switch to Kryo';
|
|
40
|
+
case 'spark.executor.memoryOverhead': return 'Raise memory overhead';
|
|
41
|
+
}
|
|
42
|
+
return undefined;
|
|
43
|
+
},
|
|
44
|
+
genericRecommendation(f) {
|
|
45
|
+
switch (f.property) {
|
|
46
|
+
case 'spark.shuffle.service.enabled': return 'Set spark.shuffle.service.enabled=true so shuffle data survives executor removal.';
|
|
47
|
+
case 'spark.dynamicAllocation.minExecutors': return 'Set the minimum executor bound at or below the maximum.';
|
|
48
|
+
case 'spark.dynamicAllocation.maxExecutors': return 'Set spark.dynamicAllocation.maxExecutors to cap cluster growth.';
|
|
49
|
+
case 'spark.serializer': return 'Consider spark.serializer=org.apache.spark.serializer.KryoSerializer for faster, smaller buffers.';
|
|
50
|
+
case 'spark.executor.memoryOverhead': return 'Raise executor memoryOverhead above Spark\'s default floor to avoid off-heap OOM-kills.';
|
|
51
|
+
}
|
|
52
|
+
return undefined;
|
|
53
|
+
},
|
|
54
|
+
};
|
|
55
|
+
|
|
56
|
+
/** One row per emitted finding type: the mapped type makes a missing or stray row a compile error. */
|
|
57
|
+
export const FINDING_PRESENTATION = {
|
|
58
|
+
incompleteRun: {
|
|
59
|
+
name: 'incomplete run',
|
|
60
|
+
tag: 'INCMP',
|
|
61
|
+
thresholdSummary: () => 'an event log missing its terminal ApplicationEnd/job-completion event',
|
|
62
|
+
actionLabel: () => undefined,
|
|
63
|
+
genericRecommendation: () => undefined,
|
|
64
|
+
},
|
|
65
|
+
|
|
66
|
+
skew: {
|
|
67
|
+
name: 'task skew',
|
|
68
|
+
tag: 'SKEW',
|
|
69
|
+
thresholdSummary: () => 'task duration skew above the configured ratio',
|
|
70
|
+
actionLabel: () => 'Fix task skew',
|
|
71
|
+
genericRecommendation: () => 'For join-driven skew, enable AQE skew-join handling (spark.sql.adaptive.skewJoin.enabled); otherwise salt the key or repartition on a better key to reduce task skew.',
|
|
72
|
+
},
|
|
73
|
+
stageShape: {
|
|
74
|
+
name: 'stage shape',
|
|
75
|
+
tag: 'SHAPE',
|
|
76
|
+
thresholdSummary: () => 'low parallelism, data explosion, or task-count skew relative to core count',
|
|
77
|
+
actionLabel(f) {
|
|
78
|
+
switch (f.rule) {
|
|
79
|
+
case 'lowParallelism': return 'Increase parallelism';
|
|
80
|
+
case 'dataExplosion': return 'Check for exploding join';
|
|
81
|
+
case 'taskStageSkew': return 'Fix straggler task';
|
|
82
|
+
}
|
|
83
|
+
return undefined;
|
|
84
|
+
},
|
|
85
|
+
genericRecommendation(f) {
|
|
86
|
+
switch (f.rule) {
|
|
87
|
+
case 'lowParallelism': return 'Too few tasks run relative to the cores available, leaving cluster capacity idle: repartition to use more of it.';
|
|
88
|
+
case 'dataExplosion': return 'Output volume far exceeds input volume: check for an exploding join or a cross product.';
|
|
89
|
+
case 'taskStageSkew': return 'A single straggler task gates the whole stage\'s wall-clock duration.';
|
|
90
|
+
}
|
|
91
|
+
return undefined;
|
|
92
|
+
},
|
|
93
|
+
},
|
|
94
|
+
tinyTask: {
|
|
95
|
+
name: 'tiny tasks',
|
|
96
|
+
tag: 'TINY',
|
|
97
|
+
thresholdSummary: () => 'median task duration below the configured floor',
|
|
98
|
+
actionLabel: () => 'Coalesce small tasks',
|
|
99
|
+
// The shuffle-vs-no-shuffle fix isn't a Finding field, so one sentence covers both.
|
|
100
|
+
genericRecommendation: () => 'Scheduler overhead may dominate: lower spark.sql.shuffle.partitions, or coalesce down to fewer, larger tasks.',
|
|
101
|
+
},
|
|
102
|
+
|
|
103
|
+
shuffle: {
|
|
104
|
+
name: 'shuffle I/O',
|
|
105
|
+
tag: 'SHFL',
|
|
106
|
+
thresholdSummary: () => 'shuffle read above the configured minimum byte threshold',
|
|
107
|
+
actionLabel: () => 'Reduce shuffle size',
|
|
108
|
+
genericRecommendation: () => 'Consider increasing spark.sql.shuffle.partitions or adding a broadcast join to shrink the shuffle.',
|
|
109
|
+
},
|
|
110
|
+
partitionSizing: {
|
|
111
|
+
name: 'partition sizing',
|
|
112
|
+
tag: 'PART',
|
|
113
|
+
thresholdSummary: () => 'partition byte size outside the configured target range',
|
|
114
|
+
actionLabel(f) {
|
|
115
|
+
switch (f.rule) {
|
|
116
|
+
case 'shufflePartitionSkew': return 'Fix skewed partition';
|
|
117
|
+
case 'lowShuffleParallelism': return 'Add shuffle partitions';
|
|
118
|
+
case 'maxPartitionTooBig': return 'Repartition oversized data';
|
|
119
|
+
}
|
|
120
|
+
return undefined;
|
|
121
|
+
},
|
|
122
|
+
genericRecommendation(f) {
|
|
123
|
+
switch (f.rule) {
|
|
124
|
+
case 'shufflePartitionSkew': return 'For join skew, enable AQE skew-join handling (spark.sql.adaptive.skewJoin.enabled); otherwise salt the key or repartition on a better key.';
|
|
125
|
+
case 'lowShuffleParallelism': return 'Raise spark.sql.shuffle.partitions so each partition is smaller.';
|
|
126
|
+
case 'maxPartitionTooBig': return 'Repartition to break up the oversized partition before this stage.';
|
|
127
|
+
}
|
|
128
|
+
return undefined;
|
|
129
|
+
},
|
|
130
|
+
},
|
|
131
|
+
|
|
132
|
+
spill: {
|
|
133
|
+
name: 'spill',
|
|
134
|
+
tag: 'SPILL',
|
|
135
|
+
thresholdSummary: (t) => `single-task disk spill above ${t.singleTaskDiskGiB} GiB`,
|
|
136
|
+
actionLabel: () => 'Reduce spill',
|
|
137
|
+
// The skew/volume classification isn't a Finding field, so one sentence covers both.
|
|
138
|
+
genericRecommendation: () => 'If the spill is skew-driven, fix task skew first: adding memory will not help. Otherwise raise spark.sql.shuffle.partitions or increase executor memory.',
|
|
139
|
+
},
|
|
140
|
+
|
|
141
|
+
gc: {
|
|
142
|
+
name: 'GC pressure',
|
|
143
|
+
tag: 'GC',
|
|
144
|
+
thresholdSummary: () => 'JVM GC time share above the configured ratio',
|
|
145
|
+
actionLabel: (f) => (f.direction === 'low' ? 'Right-size executor memory' : 'Reduce GC pressure'),
|
|
146
|
+
genericRecommendation: (f) => (f.direction === 'low'
|
|
147
|
+
? 'Memory may be over-provisioned here: consider reducing spark.executor.memory for cost savings.'
|
|
148
|
+
: 'Reduce object creation, use primitive types, avoid UDFs, or increase executor memory to cut GC time.'),
|
|
149
|
+
},
|
|
150
|
+
|
|
151
|
+
stageFailed: {
|
|
152
|
+
name: 'failed stage',
|
|
153
|
+
tag: 'SFAIL',
|
|
154
|
+
thresholdSummary: () => 'a stage that failed outright',
|
|
155
|
+
actionLabel: () => 'Inspect stage failure',
|
|
156
|
+
genericRecommendation: () => 'Inspect the driver log for the failure reason and the job that triggered it.',
|
|
157
|
+
},
|
|
158
|
+
failures: {
|
|
159
|
+
name: 'failed tasks',
|
|
160
|
+
tag: 'FAIL',
|
|
161
|
+
thresholdSummary: () => 'task failures above the configured rate',
|
|
162
|
+
actionLabel: () => 'Investigate task failures',
|
|
163
|
+
genericRecommendation: () => 'Investigate driver logs for executor instability or data-driven errors.',
|
|
164
|
+
},
|
|
165
|
+
retryWaste: {
|
|
166
|
+
name: 'retry waste',
|
|
167
|
+
tag: 'RETRY',
|
|
168
|
+
thresholdSummary: () => 'retried task attempts consuming executor time',
|
|
169
|
+
actionLabel: () => 'Investigate retry cause',
|
|
170
|
+
genericRecommendation: () => 'Investigate executor loss or fetch failures behind the retried attempts.',
|
|
171
|
+
},
|
|
172
|
+
|
|
173
|
+
slowHost: {
|
|
174
|
+
name: 'slow executor host',
|
|
175
|
+
tag: 'HOST',
|
|
176
|
+
thresholdSummary: (t) => `a host running ${t.ratioWarn}x+ slower than its peers by mean task duration (per-executor byte/time dimensions use a separate, narrower ratio ladder starting at ${t.ratioTiers[0]}x; only those can reach critical on ratio alone)`,
|
|
177
|
+
actionLabel(f) {
|
|
178
|
+
if (f.variant === 'durationShare') return 'Fix data locality';
|
|
179
|
+
if (f.variant === 'multiDim') return 'Investigate degraded executor';
|
|
180
|
+
return 'Check slow host';
|
|
181
|
+
},
|
|
182
|
+
genericRecommendation(f) {
|
|
183
|
+
if (f.variant === 'durationShare') return 'Check for data locality or partition assignment skewing work onto one node.';
|
|
184
|
+
if (f.variant === 'multiDim') return 'Investigate uneven partition assignment or a degraded executor.';
|
|
185
|
+
return 'Check what this host was running: it may just hold data locality for its tasks or carry one heavy stage, rather than a hardware fault. Enable spark.speculation to relaunch a lagging task automatically.';
|
|
186
|
+
},
|
|
187
|
+
},
|
|
188
|
+
stageSlowness: {
|
|
189
|
+
name: 'slow stage',
|
|
190
|
+
tag: 'SLOW',
|
|
191
|
+
thresholdSummary: () => 'a stage running far longer than its peers, not attributable to a single slow host',
|
|
192
|
+
actionLabel: () => 'Profile slow stage',
|
|
193
|
+
genericRecommendation: () => 'Often a partition-count problem: raise parallelism via spark.sql.shuffle.partitions or spark.default.parallelism, or check for a large per-task data volume driving heavy shuffle and spill.',
|
|
194
|
+
},
|
|
195
|
+
straggler: {
|
|
196
|
+
name: 'straggling task',
|
|
197
|
+
tag: 'STRAG',
|
|
198
|
+
thresholdSummary: () => 'one or more tasks finishing far after the rest of their stage',
|
|
199
|
+
actionLabel: () => 'Fix stragglers',
|
|
200
|
+
genericRecommendation: () => 'Rule out a GC pause or a slow shuffle fetch before assuming a hardware issue. If a skewed key is the real cause, that is a candidate for AQE\'s skew-join handling.',
|
|
201
|
+
},
|
|
202
|
+
speculationWaste: {
|
|
203
|
+
name: 'speculation waste',
|
|
204
|
+
tag: 'SPEC',
|
|
205
|
+
thresholdSummary: () => 'speculative task attempts that completed after the original',
|
|
206
|
+
actionLabel: () => 'Tune speculation settings',
|
|
207
|
+
genericRecommendation: () => 'If task durations are naturally variable rather than genuine stragglers, consider tuning spark.speculation.multiplier/quantile.',
|
|
208
|
+
},
|
|
209
|
+
coldStart: {
|
|
210
|
+
name: 'cold start',
|
|
211
|
+
tag: 'COLD',
|
|
212
|
+
thresholdSummary: () => 'executor startup time above the configured floor',
|
|
213
|
+
actionLabel: () => 'Pre-warm cluster',
|
|
214
|
+
genericRecommendation: () => 'Keep a warm pool of idle executors, or if using dynamic allocation, raise the minimum/initial executor count so it does not scale up from zero.',
|
|
215
|
+
},
|
|
216
|
+
|
|
217
|
+
memoryUtilization: {
|
|
218
|
+
name: 'memory utilization',
|
|
219
|
+
tag: 'MEM',
|
|
220
|
+
thresholdSummary: () => 'executor heap usage outside the configured band',
|
|
221
|
+
actionLabel(f) {
|
|
222
|
+
switch (f.variant) {
|
|
223
|
+
case 'idleCores': return 'Reduce idle cores';
|
|
224
|
+
case 'wasteModel': return 'Right-size executor memory';
|
|
225
|
+
case 'memoryBand':
|
|
226
|
+
if (f.dataUnavailable) return 'Enable memory metrics';
|
|
227
|
+
return f.rule === 'heapNearCapacity' ? 'Increase executor memory' : 'Reduce executor memory';
|
|
228
|
+
}
|
|
229
|
+
return undefined;
|
|
230
|
+
},
|
|
231
|
+
genericRecommendation(f) {
|
|
232
|
+
switch (f.variant) {
|
|
233
|
+
case 'idleCores': return 'Reduce cluster size or enable dynamic allocation.';
|
|
234
|
+
case 'wasteModel': return 'Review spark.executor.memory and executor count.';
|
|
235
|
+
case 'memoryBand':
|
|
236
|
+
if (f.dataUnavailable) return undefined;
|
|
237
|
+
return f.rule === 'heapNearCapacity'
|
|
238
|
+
? 'Memory may be too small: raise spark.executor.memory to avoid OOM/spill.'
|
|
239
|
+
: 'Memory may be over-provisioned: consider reducing spark.executor.memory for cost savings.';
|
|
240
|
+
}
|
|
241
|
+
return undefined;
|
|
242
|
+
},
|
|
243
|
+
},
|
|
244
|
+
utilization: {
|
|
245
|
+
name: 'executor utilization',
|
|
246
|
+
tag: 'UTIL',
|
|
247
|
+
thresholdSummary: () => 'core occupancy below the configured floor across the run',
|
|
248
|
+
actionLabel: () => 'Reduce cluster size',
|
|
249
|
+
genericRecommendation: () => 'Consider reducing cluster size or enabling dynamic allocation.',
|
|
250
|
+
},
|
|
251
|
+
coreLocality: {
|
|
252
|
+
name: 'core locality',
|
|
253
|
+
tag: 'LOCAL',
|
|
254
|
+
thresholdSummary: () => 'task placement missing data-local core assignment',
|
|
255
|
+
actionLabel: () => 'Fix data locality',
|
|
256
|
+
genericRecommendation: () => 'Check spark.locality.wait settings and executor/data colocation.',
|
|
257
|
+
},
|
|
258
|
+
cachingOpportunity: {
|
|
259
|
+
name: 'caching opportunity',
|
|
260
|
+
tag: 'CACHE',
|
|
261
|
+
thresholdSummary: () => 'a dataset re-read from source multiple times with no cache/persist',
|
|
262
|
+
actionLabel: (f) => (f.variant === 'composite' ? 'Cache repeated result' : 'Cache shared table'),
|
|
263
|
+
genericRecommendation: (f) => (f.variant === 'composite'
|
|
264
|
+
? 'Cache or persist the repeated join/union result so it is computed once instead of recomputed per query.'
|
|
265
|
+
: 'Cache the shared DataFrame, or broadcast it if it is a small join lookup.'),
|
|
266
|
+
},
|
|
267
|
+
cacheUtilization: {
|
|
268
|
+
name: 'cache utilization',
|
|
269
|
+
tag: 'CSTOR',
|
|
270
|
+
thresholdSummary: () => 'cached partitions evicted or spilled to disk',
|
|
271
|
+
actionLabel: (f) => (f.dataUnavailable ? 'Enable block-update logging' : 'Increase cache memory'),
|
|
272
|
+
genericRecommendation(f) {
|
|
273
|
+
switch (f.variant) {
|
|
274
|
+
case 'partialCache': return 'Increase executor memory or reduce the cached dataset size so more of it stays cached.';
|
|
275
|
+
case 'diskSpillover': return 'Executor memory may be too small for this cached dataset: increase executor memory or reduce its size.';
|
|
276
|
+
}
|
|
277
|
+
return undefined;
|
|
278
|
+
},
|
|
279
|
+
},
|
|
280
|
+
jobFailureRate: {
|
|
281
|
+
name: 'job failure rate',
|
|
282
|
+
tag: 'JOBS',
|
|
283
|
+
thresholdSummary: () => 'job failure rate above the configured threshold',
|
|
284
|
+
actionLabel: () => 'Investigate failed jobs',
|
|
285
|
+
genericRecommendation: () => 'Inspect the driver log for the failed job(s) and the stage failures that triggered them.',
|
|
286
|
+
},
|
|
287
|
+
autoscalingChurn: {
|
|
288
|
+
name: 'autoscaling churn',
|
|
289
|
+
tag: 'CHRN',
|
|
290
|
+
thresholdSummary: () => 'executor add/remove churn above the configured rate',
|
|
291
|
+
actionLabel: () => 'Reduce autoscaling churn',
|
|
292
|
+
genericRecommendation: () => 'This looks like wasteful re-provisioning rather than normal scale-down: consider raising spark.dynamicAllocation.executorIdleTimeout or widening the minExecutors/maxExecutors bounds to reduce flapping.',
|
|
293
|
+
},
|
|
294
|
+
|
|
295
|
+
configAudit: CONFIG_AUDIT_PRESENTATION,
|
|
296
|
+
|
|
297
|
+
duplicatePlanSubtree: {
|
|
298
|
+
name: 'duplicate plan subtree',
|
|
299
|
+
tag: 'PLAN',
|
|
300
|
+
thresholdSummary: () => 'the same physical plan subtree executed more than once',
|
|
301
|
+
actionLabel: () => 'Dedupe repeated subtree',
|
|
302
|
+
// isExchangeRoot isn't a Finding field, so one sentence covers both cases.
|
|
303
|
+
genericRecommendation: () => 'Check whether the repeated subtree could be computed once and reused, or cache/persist the shared computation.',
|
|
304
|
+
},
|
|
305
|
+
smallFiles: {
|
|
306
|
+
name: 'small files',
|
|
307
|
+
tag: 'PLAN',
|
|
308
|
+
thresholdSummary: () => 'output files below the configured target size',
|
|
309
|
+
actionLabel: (f) => (f.direction === 'write' ? 'Coalesce output files' : 'Compact small files'),
|
|
310
|
+
genericRecommendation: (f) => (f.direction === 'write'
|
|
311
|
+
? 'Repartition or coalesce before writing to raise the average file size.'
|
|
312
|
+
: 'Compact the upstream output so fewer, larger files are produced.'),
|
|
313
|
+
},
|
|
314
|
+
underBroadcast: {
|
|
315
|
+
name: 'missed broadcast join',
|
|
316
|
+
tag: 'PLAN',
|
|
317
|
+
thresholdSummary: () => 'a join below the configured size floor that skipped broadcast',
|
|
318
|
+
actionLabel: () => 'Use broadcast join',
|
|
319
|
+
genericRecommendation: () => 'This could have been a broadcast join: consider a broadcast() hint or raising spark.sql.autoBroadcastJoinThreshold.',
|
|
320
|
+
},
|
|
321
|
+
overBroadcast: {
|
|
322
|
+
name: 'oversized broadcast join',
|
|
323
|
+
tag: 'PLAN',
|
|
324
|
+
thresholdSummary: () => 'a broadcast join above the configured size ceiling',
|
|
325
|
+
actionLabel: () => 'Fix oversized broadcast',
|
|
326
|
+
genericRecommendation: () => 'Check for a misapplied broadcast hint or a misconfigured spark.sql.autoBroadcastJoinThreshold.',
|
|
327
|
+
},
|
|
328
|
+
};
|
|
329
|
+
|
|
330
|
+
/** The presentation row for a free-form type string (report JSON, a filter, a test double). */
|
|
331
|
+
export function presentationOf(type ) {
|
|
332
|
+
return (FINDING_PRESENTATION )[type];
|
|
333
|
+
}
|
|
@@ -0,0 +1,110 @@
|
|
|
1
|
+
// Plain-language help per finding tag: the expansion the tag stands for and a one-line description
|
|
2
|
+
// of what it means. Shared by the dashboard (tag tooltips, verdict "What's happening") and the
|
|
3
|
+
// comparison verdict, which names finding categories by their expansion on every path.
|
|
4
|
+
|
|
5
|
+
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
export const TAG_HELP = {
|
|
10
|
+
INCMP: {
|
|
11
|
+
expansion: 'Incomplete run',
|
|
12
|
+
description: 'This event log never recorded an application-end event, so other findings and metrics reflect only what was captured.',
|
|
13
|
+
},
|
|
14
|
+
SKEW: {
|
|
15
|
+
expansion: 'Task skew',
|
|
16
|
+
description: 'A small number of tasks take much longer than their peers.',
|
|
17
|
+
},
|
|
18
|
+
SHFL: {
|
|
19
|
+
expansion: 'Shuffle I/O',
|
|
20
|
+
description: 'Tasks are moving a large amount of intermediate data between stages.',
|
|
21
|
+
},
|
|
22
|
+
SPILL: {
|
|
23
|
+
expansion: 'Memory and disk spill',
|
|
24
|
+
description: 'Tasks are writing data out of memory, which slows execution.',
|
|
25
|
+
},
|
|
26
|
+
GC: {
|
|
27
|
+
expansion: 'Garbage collection pressure',
|
|
28
|
+
description: 'Tasks are spending an unusually large share of time reclaiming memory.',
|
|
29
|
+
},
|
|
30
|
+
COLD: {
|
|
31
|
+
expansion: 'Executor cold start',
|
|
32
|
+
description: 'New executors are taking time to become available for work.',
|
|
33
|
+
},
|
|
34
|
+
UTIL: {
|
|
35
|
+
expansion: 'Low utilization',
|
|
36
|
+
description: 'Allocated executors are idle for a large share of the application run.',
|
|
37
|
+
},
|
|
38
|
+
MEM: {
|
|
39
|
+
expansion: 'Memory utilization',
|
|
40
|
+
description: 'Executor memory or core capacity may be over- or under-provisioned.',
|
|
41
|
+
},
|
|
42
|
+
LOCAL: {
|
|
43
|
+
expansion: 'Core usage locality',
|
|
44
|
+
description: 'Tasks are running without process- or node-local data placement more than expected.',
|
|
45
|
+
},
|
|
46
|
+
HOST: {
|
|
47
|
+
expansion: 'Slow host',
|
|
48
|
+
description: 'One executor is substantially slower than its peers.',
|
|
49
|
+
},
|
|
50
|
+
FAIL: {
|
|
51
|
+
expansion: 'Failed tasks',
|
|
52
|
+
description: 'Tasks are failing often enough to affect the stage.',
|
|
53
|
+
},
|
|
54
|
+
STRAG: {
|
|
55
|
+
expansion: 'Straggler tasks',
|
|
56
|
+
description: 'A few tasks are much slower than the rest of their stage.',
|
|
57
|
+
},
|
|
58
|
+
SPEC: {
|
|
59
|
+
expansion: 'Speculation waste',
|
|
60
|
+
description: 'Speculative task attempts used a lot of executor time without confirming a genuine straggler.',
|
|
61
|
+
},
|
|
62
|
+
RETRY: {
|
|
63
|
+
expansion: 'Retry waste',
|
|
64
|
+
description: 'Repeated task attempts are consuming avoidable execution time.',
|
|
65
|
+
},
|
|
66
|
+
TINY: {
|
|
67
|
+
expansion: 'Tiny tasks',
|
|
68
|
+
description: 'Many very short tasks are adding scheduling overhead.',
|
|
69
|
+
},
|
|
70
|
+
SFAIL: {
|
|
71
|
+
expansion: 'Failed stage',
|
|
72
|
+
description: 'A stage attempt failed outright.',
|
|
73
|
+
},
|
|
74
|
+
PART: {
|
|
75
|
+
expansion: 'Partition sizing',
|
|
76
|
+
description: 'Shuffle partitions are too large, too uneven, or too few for the work.',
|
|
77
|
+
},
|
|
78
|
+
SLOW: {
|
|
79
|
+
expansion: 'Stage slowness',
|
|
80
|
+
description: 'A stage is slow overall without a more specific diagnosed cause.',
|
|
81
|
+
},
|
|
82
|
+
SHAPE: {
|
|
83
|
+
expansion: 'Stage shape',
|
|
84
|
+
description: 'The stage has an inefficient task count, output shape, or task-to-stage balance.',
|
|
85
|
+
},
|
|
86
|
+
CACHE: {
|
|
87
|
+
expansion: 'Caching opportunity',
|
|
88
|
+
description: 'A reusable dataset may benefit from being persisted between stages.',
|
|
89
|
+
},
|
|
90
|
+
CSTOR: {
|
|
91
|
+
expansion: 'Cache storage',
|
|
92
|
+
description: 'A persisted dataset is not fully cached in memory or is spilling to disk.',
|
|
93
|
+
},
|
|
94
|
+
CHRN: {
|
|
95
|
+
expansion: 'Autoscaling churn',
|
|
96
|
+
description: 'Executors are being stood up and torn down again before they can do useful work.',
|
|
97
|
+
},
|
|
98
|
+
JOBS: {
|
|
99
|
+
expansion: 'Job failure rate',
|
|
100
|
+
description: 'A large share of completed jobs did not succeed.',
|
|
101
|
+
},
|
|
102
|
+
CFG: {
|
|
103
|
+
expansion: 'Configuration audit',
|
|
104
|
+
description: 'Configuration settings may cause reliability or efficiency problems.',
|
|
105
|
+
},
|
|
106
|
+
PLAN: {
|
|
107
|
+
expansion: 'Plan advisor',
|
|
108
|
+
description: 'The SQL execution plan has a pattern worth reviewing.',
|
|
109
|
+
},
|
|
110
|
+
};
|