sparkforensics-mcp 0.2.2 → 0.2.4
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/bin/sparkforensics-mcp.mjs +11 -5
- package/package.json +3 -3
- package/vendor-core/cli/collect-run.js +3 -2
- package/vendor-core/cli/native-zstd.js +351 -0
- package/vendor-core/detectors.js +243 -53
- package/vendor-core/docs-config.js +34 -8
- package/vendor-core/docs-content/chapters/03-memory-model.md +39 -0
- package/vendor-core/docs-content/chapters/11-cluster-config.md +40 -0
- package/vendor-core/docs-content/detection/gc.md +2 -0
- package/vendor-core/docs-content/detection/host.md +2 -1
- package/vendor-core/docs-content/detection/plan.md +3 -1
- package/vendor-core/docs-content/detection/shape.md +2 -1
- package/vendor-core/docs-content/detection/shfl.md +2 -1
- package/vendor-core/docs-content/detection/spill.md +1 -1
- package/vendor-core/docs-content/detection/strag.md +2 -1
- package/vendor-core/docs-content/detection/tiny.md +2 -1
- package/vendor-core/docs-content/tuning/failures.md +1 -1
- package/vendor-core/docs-content/tuning/gc.md +11 -4
- package/vendor-core/docs-content/tuning/shuffle.md +25 -5
- package/vendor-core/docs-content/tuning/skew.md +14 -6
- package/vendor-core/docs-content/tuning/small-files.md +12 -7
- package/vendor-core/docs-content/tuning/straggler.md +34 -0
- package/vendor-core/docs-content/tuning/tiny-tasks.md +1 -1
- package/vendor-core/docs-content/tuning/utilization.md +57 -7
- package/vendor-core/docs-content/upstream.json +4 -0
- package/vendor-core/event-handlers.js +321 -69
- package/vendor-core/event-schemas.js +8 -6
- package/vendor-core/evidence-report.js +3 -1
- package/vendor-core/impact-estimator.js +170 -34
- package/vendor-core/mcp-tools.js +20 -7
- package/vendor-core/occupancy.js +71 -2
- package/vendor-core/parser-worker.js +56 -24
- package/vendor-core/plan-summary.js +5 -1
- package/vendor-core/run-comparison.js +1 -15
- package/vendor-core/shs-fetch.js +18 -7
- package/vendor-core/shs-load.js +2 -1
- package/vendor-core/stage-quantiles.js +111 -3
- package/vendor-core/string-hash.js +15 -0
- package/vendor-core/types.js +14 -1
- package/vendor-core/vendor/fzstd.js +94 -18
- package/vendor-core/zstd-worker-client.js +180 -0
- package/vendor-core/zstd-worker.js +103 -0
|
@@ -4,15 +4,27 @@ import { detectorCatalog } from './detectors.js';
|
|
|
4
4
|
// serves the tuning reference, relative to the app's origin.
|
|
5
5
|
export const DOCS_BASE_DIR = 'docs/tuning-reference';
|
|
6
6
|
|
|
7
|
+
// Bottleneck sub-anchors that are sections of another entry's page, not pages of their own.
|
|
8
|
+
// Most live on a sibling bottleneck page; autoscaling-churn and cache-utilization live on the
|
|
9
|
+
// cluster-config and memory-model chapters. Keep in sync with spark-tuning-reference's anchor-map.
|
|
10
|
+
const SUB_ANCHOR_PAGES = {
|
|
11
|
+
'bottleneck-stage-shape': 'bottleneck-skew',
|
|
12
|
+
'bottleneck-stage-slowness': 'bottleneck-slow-host',
|
|
13
|
+
'bottleneck-partition-sizing': 'bottleneck-shuffle',
|
|
14
|
+
'bottleneck-speculation-waste': 'bottleneck-straggler',
|
|
15
|
+
'bottleneck-core-locality': 'bottleneck-utilization',
|
|
16
|
+
'bottleneck-caching-opportunity': 'bottleneck-utilization',
|
|
17
|
+
'bottleneck-autoscaling-churn': 'cluster-config',
|
|
18
|
+
'bottleneck-cache-utilization': 'memory-model',
|
|
19
|
+
};
|
|
20
|
+
|
|
7
21
|
// Some anchors are in-page fragments on another entry's page: config-audit sub-findings and
|
|
8
|
-
// metric-glossary entries live on the 'config'/'metrics' pages, and
|
|
9
|
-
//
|
|
22
|
+
// metric-glossary entries live on the 'config'/'metrics' pages, and SUB_ANCHOR_PAGES lists the
|
|
23
|
+
// bottleneck sub-anchors.
|
|
10
24
|
export function pageForAnchor(anchor ) {
|
|
11
25
|
if (anchor.startsWith('metric-')) return 'metrics';
|
|
12
26
|
if (anchor.startsWith('config-')) return 'config';
|
|
13
|
-
|
|
14
|
-
if (anchor === 'bottleneck-stage-slowness') return 'bottleneck-slow-host';
|
|
15
|
-
return anchor;
|
|
27
|
+
return SUB_ANCHOR_PAGES[anchor] ?? anchor;
|
|
16
28
|
}
|
|
17
29
|
|
|
18
30
|
// Build a docs URL for an anchor like '#bottleneck-skew'. The leading '#' is
|
|
@@ -45,7 +57,7 @@ export const METRIC_ANCHORS = {
|
|
|
45
57
|
'stage-duration': '#metric-stage-duration',
|
|
46
58
|
};
|
|
47
59
|
|
|
48
|
-
// Allowlist of every anchor that exists in the
|
|
60
|
+
// Allowlist of every anchor that exists in the generated tuning-reference markdown. DocsLink
|
|
49
61
|
// renders a link only for anchors here: a detector may declare a docAnchor for an unwritten
|
|
50
62
|
// section, and gating keeps that from becoming a dead "Learn more" link. docs-config.test.js
|
|
51
63
|
// asserts this stays a subset of the real ids so it can't drift.
|
|
@@ -62,6 +74,9 @@ export const KNOWN_DOC_ANCHORS = new Set([
|
|
|
62
74
|
'#bottleneck-memory-utilization', '#bottleneck-broadcast-sizing',
|
|
63
75
|
'#bottleneck-duplicate-plan-subtree', '#bottleneck-small-files',
|
|
64
76
|
'#bottleneck-stage-shape', '#bottleneck-stage-slowness',
|
|
77
|
+
'#bottleneck-partition-sizing', '#bottleneck-speculation-waste',
|
|
78
|
+
'#bottleneck-core-locality', '#bottleneck-caching-opportunity',
|
|
79
|
+
'#bottleneck-autoscaling-churn', '#bottleneck-cache-utilization',
|
|
65
80
|
// Config-audit sections
|
|
66
81
|
'#config-autoscale-bounds', '#config-memory-overhead', '#config-serializer',
|
|
67
82
|
'#config-shuffle-service',
|
|
@@ -69,7 +84,7 @@ export const KNOWN_DOC_ANCHORS = new Set([
|
|
|
69
84
|
...Object.values(METRIC_ANCHORS),
|
|
70
85
|
]);
|
|
71
86
|
|
|
72
|
-
// True when `anchor` resolves to a real section in the
|
|
87
|
+
// True when `anchor` resolves to a real section in the generated
|
|
73
88
|
// tuning-reference markdown.
|
|
74
89
|
export function isKnownDocAnchor(anchor ) {
|
|
75
90
|
return KNOWN_DOC_ANCHORS.has(String(anchor));
|
|
@@ -109,10 +124,21 @@ export function docAnchorForType(type ) {
|
|
|
109
124
|
return anchor && isKnownDocAnchor(anchor) ? anchor : undefined;
|
|
110
125
|
}
|
|
111
126
|
|
|
127
|
+
/** The docAnchor every finding in `findings` carries, or undefined when they disagree or any lacks
|
|
128
|
+
* one. For a badge standing for several findings (a widget header, a grouped row) whose type-level
|
|
129
|
+
* docAnchorForType can't pick one: configAudit's sub-checks each stamp their own anchor. */
|
|
130
|
+
export function sharedDocAnchor(findings ) {
|
|
131
|
+
const anchors = new Set(findings.map((f) => f.docAnchor));
|
|
132
|
+
if (anchors.size !== 1) return undefined;
|
|
133
|
+
const [anchor] = anchors;
|
|
134
|
+
return anchor;
|
|
135
|
+
}
|
|
136
|
+
|
|
112
137
|
// Resolves a docAnchorForType() result to the tuning-doc file slug under docs-content/tuning/.
|
|
113
138
|
// Reuses pageForAnchor's sub-anchor resolution so a sub-anchor like bottleneck-stage-shape maps
|
|
114
139
|
// to its owning page (skew.md), not a stage-shape.md that never exists. Returns null for anchors
|
|
115
|
-
// with no bottleneck tuning doc (metric-/config-prefixed,
|
|
140
|
+
// with no bottleneck tuning doc (metric-/config-prefixed, a page section like #memory-model, or a
|
|
141
|
+
// sub-anchor hosted on a chapter like #bottleneck-autoscaling-churn).
|
|
116
142
|
export function tuningDocSlugForAnchor(anchor ) {
|
|
117
143
|
const bare = String(anchor).replace(/^#/, '');
|
|
118
144
|
const page = pageForAnchor(bare);
|
|
@@ -58,6 +58,45 @@ When enabling off-heap memory, always pair `spark.memory.offHeap.enabled=true` w
|
|
|
58
58
|
|
|
59
59
|
> **PySpark:** if you need PySpark's own memory bounded rather than folded silently into the overhead, set `spark.executor.pyspark.memory` explicitly, though its enforcement relies on Python's `resource` module and won't work on Windows and won't actually limit anything on macOS[^4].
|
|
60
60
|
|
|
61
|
+
## Cache utilization {#bottleneck-cache-utilization}
|
|
62
|
+
|
|
63
|
+
<span class="tag">CSTOR</span>
|
|
64
|
+
|
|
65
|
+
Spark's event log carries no block-access or hit-rate events, so cache utilization is read
|
|
66
|
+
from periodic per-RDD storage snapshots instead: how much of a persisted RDD is actually
|
|
67
|
+
resident where it was asked to be.
|
|
68
|
+
|
|
69
|
+
### How it's detected
|
|
70
|
+
|
|
71
|
+
A persisted RDD, one whose storage level requests memory and/or disk with at least one
|
|
72
|
+
partition actually cached, is evaluated on two independent measures and can register on
|
|
73
|
+
both at once:
|
|
74
|
+
|
|
75
|
+
| Signal | Info | Warning |
|
|
76
|
+
|---|---|---|
|
|
77
|
+
| Cached ratio (cached partitions ÷ total partitions) | < 90% | < 50% |
|
|
78
|
+
| Disk ratio (disk bytes ÷ (memory bytes + disk bytes)), `MEMORY_AND_DISK*` levels only | > 15% | > 40% |
|
|
79
|
+
|
|
80
|
+
Both ratios come from a point-in-time storage snapshot taken at stage-submission events,
|
|
81
|
+
not a runtime block-access count, so confirm against the Spark UI's Storage tab before
|
|
82
|
+
acting. Confidence scales with the RDD's partition count: below 10 partitions the ratio
|
|
83
|
+
is noisy enough to call low confidence, at 50 or more it's high.
|
|
84
|
+
|
|
85
|
+
### Why it matters
|
|
86
|
+
|
|
87
|
+
A partially cached RDD still pays the recomputation cost for its evicted share on every
|
|
88
|
+
later read, undercutting the reason it was cached in the first place. Disk spillover
|
|
89
|
+
under a `MEMORY_AND_DISK*` level keeps that data around instead of forcing a recompute,
|
|
90
|
+
but a disk read is still far slower than serving it out of memory.
|
|
91
|
+
|
|
92
|
+
### How to fix it
|
|
93
|
+
|
|
94
|
+
- Increase executor memory, or reduce the size of the dataset being cached, so more of it
|
|
95
|
+
fits in the storage region without eviction.
|
|
96
|
+
- If the RDD is only cached for the occasional narrow scan, weigh whether persisting it
|
|
97
|
+
at all is worth the partial-cache overhead, versus recomputing the slice you actually
|
|
98
|
+
need.
|
|
99
|
+
|
|
61
100
|
## Sources
|
|
62
101
|
|
|
63
102
|
[^1]: [Task Memory Management in Spark](https://raw.githubusercontent.com/spoddutur/spark-notes/master/task_memory_management_in_spark.md)
|
|
@@ -160,6 +160,46 @@ the target executor count dynamic allocation would otherwise compute[^3]. Raise
|
|
|
160
160
|
so proportionally reduces concurrent task slots per executor
|
|
161
161
|
(`spark.executor.cores / spark.task.cpus`)[^3].
|
|
162
162
|
|
|
163
|
+
## Autoscaling churn {#bottleneck-autoscaling-churn}
|
|
164
|
+
|
|
165
|
+
<span class="tag">CHRN</span> <span class="tag">EXPERIMENTAL</span>
|
|
166
|
+
|
|
167
|
+
Dynamic allocation adds executors once tasks back up and frees them once they go
|
|
168
|
+
idle[^6]. Churn is a narrower failure mode inside that same mechanism: executors that get
|
|
169
|
+
stood up and torn down again before they've done much useful work, cycling through
|
|
170
|
+
provisioning and JVM startup cost repeatedly instead of running tasks.
|
|
171
|
+
|
|
172
|
+
### How it's detected
|
|
173
|
+
|
|
174
|
+
An executor counts as short-lived when its lifetime, from its `ExecutorAdded` event to
|
|
175
|
+
either its `ExecutorRemoved` event or the end of the run if it was never removed, is
|
|
176
|
+
under 2 minutes. This is evaluated once a run has added at least 5 executors, a floor
|
|
177
|
+
that keeps a two- or three-executor job from registering on ordinary scale-down.
|
|
178
|
+
|
|
179
|
+
| Signal | Warning | Critical |
|
|
180
|
+
|---|---|---|
|
|
181
|
+
| Share of added executors that are short-lived (< 2 min lifetime) | > 30% | > 60% |
|
|
182
|
+
|
|
183
|
+
The 2-minute lifetime cutoff and the 30%/60% split are unvalidated heuristics rather than
|
|
184
|
+
figures published in Spark's own documentation or benchmarks, so treat a finding here as
|
|
185
|
+
a prompt to look at the executor timeline rather than a calibrated verdict.
|
|
186
|
+
|
|
187
|
+
### Why it matters
|
|
188
|
+
|
|
189
|
+
Every short-lived executor pays the same provisioning and JVM-startup cost as a
|
|
190
|
+
long-lived one, but returns little task work in exchange. A run that keeps flapping
|
|
191
|
+
between scaling up and down spends more of its wall-clock time paying that repeated
|
|
192
|
+
overhead than one that settles into a stable executor count.
|
|
193
|
+
|
|
194
|
+
### How to fix it
|
|
195
|
+
|
|
196
|
+
- Raise `spark.dynamicAllocation.executorIdleTimeout` (default 60s[^3]) so an executor
|
|
197
|
+
survives a brief lull instead of being released the moment it goes idle, only to be
|
|
198
|
+
requested again shortly after.
|
|
199
|
+
- Widen the gap between `spark.dynamicAllocation.minExecutors` and `.maxExecutors`[^3]:
|
|
200
|
+
bounds set too close together force Spark to repeatedly add and remove executors to
|
|
201
|
+
track small fluctuations in the task backlog instead of settling into a stable range.
|
|
202
|
+
|
|
163
203
|
## Sources
|
|
164
204
|
|
|
165
205
|
[^1]: [Distribution of Executors, Cores and Memory for a Spark Application](https://raw.githubusercontent.com/spoddutur/spark-notes/master/distribution_of_executors_cores_and_memory_for_spark_application.md)
|
|
@@ -2,3 +2,5 @@
|
|
|
2
2
|
|
|
3
3
|
Tasks spend an unusually large share of time reclaiming memory. Reduce
|
|
4
4
|
object creation: use primitive types, avoid UDFs, or raise executor memory.
|
|
5
|
+
A stage with very little GC gets an informational note that executor memory
|
|
6
|
+
may be over-provisioned, only on stages that take at least 0.5% of the run.
|
|
@@ -2,4 +2,5 @@
|
|
|
2
2
|
|
|
3
3
|
One executor is much slower than its peers. It may just hold data locality
|
|
4
4
|
for its tasks or carry one heavy stage, rather than a hardware fault.
|
|
5
|
-
Enable `spark.speculation` to relaunch a lagging task automatically.
|
|
5
|
+
Enable `spark.speculation` to relaunch a lagging task automatically. Only
|
|
6
|
+
flagged on stages that take at least 0.5% of the run.
|
|
@@ -4,7 +4,9 @@ Flags patterns in the SQL execution plan worth reviewing. Four checks share
|
|
|
4
4
|
this tag:
|
|
5
5
|
|
|
6
6
|
- Duplicate plan subtree: the same subtree recomputed more than once in the
|
|
7
|
-
plan.
|
|
7
|
+
plan. When the repeats have the same shape but different filters, columns
|
|
8
|
+
or tables, the finding stays informational and claims no time. Only flagged
|
|
9
|
+
when the repeat's stages take at least 0.5% of the run.
|
|
8
10
|
- Small files: reading an excessive number of small files.
|
|
9
11
|
- Under-broadcast: the smaller side of a Sort Merge Join looks well under
|
|
10
12
|
the broadcast threshold; consider a `broadcast()` hint or raising
|
|
@@ -2,4 +2,5 @@
|
|
|
2
2
|
|
|
3
3
|
The stage has an inefficient task count, output shape, or task-to-stage
|
|
4
4
|
balance: for example, one straggler task taking a large fraction of the
|
|
5
|
-
stage's wall-clock time.
|
|
5
|
+
stage's wall-clock time. A too-low task count is only flagged on stages that
|
|
6
|
+
take at least 0.5% of the run.
|
|
@@ -1,4 +1,5 @@
|
|
|
1
1
|
### `SHFL`: Shuffle I/O {#shfl}
|
|
2
2
|
|
|
3
3
|
Tasks move a large amount of intermediate data between stages. Raise
|
|
4
|
-
`spark.sql.shuffle.partitions`, or add a broadcast join.
|
|
4
|
+
`spark.sql.shuffle.partitions`, or add a broadcast join. Only flagged on
|
|
5
|
+
stages that take at least 0.5% of the run.
|
|
@@ -4,4 +4,4 @@ Tasks are writing data out of memory, which slows execution. Two spill
|
|
|
4
4
|
patterns get flagged differently: skew spill, where a few heavy tasks spill
|
|
5
5
|
while most don't (rebalance partitioning), and volume spill, where most
|
|
6
6
|
tasks spill because the data genuinely exceeds available memory (add
|
|
7
|
-
partitions).
|
|
7
|
+
partitions). Only flagged on stages that take at least 0.5% of the run.
|
|
@@ -2,4 +2,5 @@
|
|
|
2
2
|
|
|
3
3
|
A few tasks run much slower than the rest of their stage. Rule out a GC
|
|
4
4
|
pause or a slow shuffle fetch before assuming a hardware issue; if a skewed
|
|
5
|
-
key is the real cause, that's a candidate for AQE's skew-join handling.
|
|
5
|
+
key is the real cause, that's a candidate for AQE's skew-join handling. Only
|
|
6
|
+
flagged on stages that take at least 0.5% of the run.
|
|
@@ -1,4 +1,5 @@
|
|
|
1
1
|
### `TINY`: Tiny tasks {#tiny}
|
|
2
2
|
|
|
3
3
|
Many very short tasks add scheduling overhead out of proportion to the work
|
|
4
|
-
each one does. Repartition to fewer, larger tasks.
|
|
4
|
+
each one does. Repartition to fewer, larger tasks. Only flagged on stages that
|
|
5
|
+
take at least 0.5% of the run.
|
|
@@ -56,7 +56,7 @@ Config-only levers for memory-driven `ExecutorLostFailure`s and mid-shuffle exec
|
|
|
56
56
|
|
|
57
57
|
```properties
|
|
58
58
|
# Cover off-heap + PySpark memory the default overhead budget does NOT include.
|
|
59
|
-
# Default is max(384m, 10% of executor memory); 2g is an example
|
|
59
|
+
# Default is max(384m, 10% of executor memory); 2g is an example — size to real off-heap/PySpark use.
|
|
60
60
|
spark.executor.memoryOverhead=2g
|
|
61
61
|
|
|
62
62
|
# Keep shuffle output alive when dynamic allocation reclaims an executor mid-shuffle
|
|
@@ -16,12 +16,19 @@ pauses and overstate how long the task actually ran.[^1]
|
|
|
16
16
|
## How it's detected
|
|
17
17
|
|
|
18
18
|
Because `jvmGCTime` sits inside `executorRunTime` rather than alongside it, the ratio between
|
|
19
|
-
the two gives a bounded read on how much of a task's wall-clock time went to garbage collection
|
|
19
|
+
the two gives a bounded read on how much of a task's wall-clock time went to garbage collection.
|
|
20
|
+
Both directions below apply once a stage's `executorRunTime` reaches 10 seconds, a floor that
|
|
21
|
+
keeps short stages from reading as noise:
|
|
20
22
|
|
|
21
|
-
| gcPct = jvmGCTime / executorRunTime |
|
|
23
|
+
| Signal (gcPct = jvmGCTime / executorRunTime) | Fires when |
|
|
22
24
|
|---|---|
|
|
23
|
-
| > 10% |
|
|
24
|
-
|
|
|
25
|
+
| High GC | > 10% |
|
|
26
|
+
| Low GC (cost signal) | < 5% |
|
|
27
|
+
|
|
28
|
+
A stage lands in at most one of these two opposite-direction bands at a time. Severity for
|
|
29
|
+
both tracks the estimated recoverable time as a share of the app's total runtime: ≥2% is
|
|
30
|
+
critical, ≥0.5% is warning, anything smaller is info. Low GC signals a possible memory
|
|
31
|
+
over-provisioning cost rather than time lost to garbage collection.
|
|
25
32
|
|
|
26
33
|
## Why it matters
|
|
27
34
|
|
|
@@ -26,11 +26,13 @@ read from local disk, and `totalBytesRead` is their sum[^6].
|
|
|
26
26
|
|
|
27
27
|
## How it's detected
|
|
28
28
|
|
|
29
|
-
|
|
|
29
|
+
| Signal | Fires when |
|
|
30
30
|
|---|---|
|
|
31
|
-
| > 50 MB |
|
|
32
|
-
|
|
33
|
-
|
|
31
|
+
| Shuffle read bytes in a stage | > 50 MB |
|
|
32
|
+
|
|
33
|
+
50 MB marks a stage as shuffle-heavy enough to flag. Severity then tracks the estimated
|
|
34
|
+
recoverable time as a share of the app's total runtime: ≥2% is critical, ≥0.5% is
|
|
35
|
+
warning, anything smaller is info.
|
|
34
36
|
|
|
35
37
|
Beyond raw byte volume, the executor-side wait is captured by `fetchWaitTime`: time a task
|
|
36
38
|
spends blocked on a remote shuffle block it needs next, not counting time spent prefetching
|
|
@@ -91,7 +93,25 @@ spark.shuffle.file.buffer=1m
|
|
|
91
93
|
spark.reducer.maxSizeInFlight=48m
|
|
92
94
|
```
|
|
93
95
|
|
|
94
|
-
## Partition sizing
|
|
96
|
+
## Partition sizing {#bottleneck-partition-sizing}
|
|
97
|
+
|
|
98
|
+
<span class="tag">PART</span>
|
|
99
|
+
|
|
100
|
+
### How it's detected
|
|
101
|
+
|
|
102
|
+
A stage's shuffle-read partition sizes surface three distinct problems:
|
|
103
|
+
|
|
104
|
+
| Signal | Fires when | Level |
|
|
105
|
+
|---|---|---|
|
|
106
|
+
| Largest partition vs. median | > 5× the median **and** > 256 MB | Warning |
|
|
107
|
+
| Low parallelism | ≥ 1 GB of shuffle read spread across ≤ 7 tasks | Warning |
|
|
108
|
+
| Oversized partition | Largest partition ≥ 5 GB | Critical |
|
|
109
|
+
|
|
110
|
+
Skew and low-parallelism severity track the estimated recoverable time as a share of the
|
|
111
|
+
app's total runtime, the same wall-clock model used across this reference. An oversized
|
|
112
|
+
partition is a fixed safety signal instead: it reports critical purely on its own size,
|
|
113
|
+
because a partition past 5 GB is an OOM/crash risk regardless of how much wall-clock time
|
|
114
|
+
fixing it would recover, so it stays critical even on a stage that barely dents the run.
|
|
95
115
|
|
|
96
116
|
Adaptive Query Execution re-optimizes the plan while the query runs: as each shuffle stage
|
|
97
117
|
materializes, it reads the real shuffle-file sizes and resizes partitions before launching the
|
|
@@ -13,10 +13,18 @@ more data than everyone else in the same stage.
|
|
|
13
13
|
|
|
14
14
|
## How it's detected
|
|
15
15
|
|
|
16
|
-
| Signal |
|
|
17
|
-
|
|
18
|
-
| P95 / median task duration
|
|
19
|
-
| Max / median task duration (
|
|
16
|
+
| Signal | Fires when |
|
|
17
|
+
|---|---|
|
|
18
|
+
| P95 / median task duration (stage has ≥ 20 tasks) | > 3× |
|
|
19
|
+
| Max / median task duration (stage has < 20 tasks) | > 3× |
|
|
20
|
+
|
|
21
|
+
A 3× ratio marks a stage as skewed. Beyond that ratio, the estimated recoverable time (the
|
|
22
|
+
P95-minus-median, or max-minus-median, delta) needs to clear 0.5% of the app's total
|
|
23
|
+
runtime before it registers, a floor that filters out a 3× ratio sitting on a few
|
|
24
|
+
milliseconds. Severity then tracks that same recoverable-time estimate as a share of the
|
|
25
|
+
app's total runtime: ≥2% is critical, ≥0.5% is warning, anything smaller is info. A 3×
|
|
26
|
+
ratio on a stage that barely dents an eight-hour run typically surfaces as info; the same
|
|
27
|
+
ratio on a stage that dominates a short run reads as critical.
|
|
20
28
|
|
|
21
29
|
## Why it matters
|
|
22
30
|
|
|
@@ -53,12 +61,12 @@ The AQE toggle, plus the manual salting fallback as runnable code:
|
|
|
53
61
|
```python
|
|
54
62
|
from pyspark.sql import functions as fn
|
|
55
63
|
|
|
56
|
-
# AQE skew-join handling
|
|
64
|
+
# AQE skew-join handling — active once AQE itself is enabled
|
|
57
65
|
spark.conf.set("spark.sql.adaptive.enabled", "true")
|
|
58
66
|
spark.conf.set("spark.sql.adaptive.skewJoin.enabled", "true")
|
|
59
67
|
|
|
60
68
|
# Manual salting fallback (pre-AQE): spread the skewed key across N salted variants.
|
|
61
|
-
# N is an example
|
|
69
|
+
# N is an example — size it to how badly the key is skewed.
|
|
62
70
|
N = 16
|
|
63
71
|
salted_big = big.withColumn("salt", (fn.rand() * N).cast("int"))
|
|
64
72
|
salted_small = small.withColumn(
|
|
@@ -21,15 +21,20 @@ inefficient to read a whole block when you only need a few rows[^5].
|
|
|
21
21
|
|
|
22
22
|
## How it's detected
|
|
23
23
|
|
|
24
|
-
|
|
25
|
-
|
|
26
|
-
|
|
27
|
-
|
|
24
|
+
Each SQL plan node carries file-count and file-size metrics Spark reports directly:
|
|
25
|
+
`number of files read`/`size of files read` on the read side, `number of written
|
|
26
|
+
files`/`written output` on the write side. A node reads as a small-files problem once
|
|
27
|
+
its file count on a given side is high and the resulting average file size is small.
|
|
28
28
|
|
|
29
|
-
| Signal
|
|
29
|
+
| Signal (per plan node, checked separately for read and write) | Fires when |
|
|
30
30
|
|---|---|
|
|
31
|
-
|
|
|
32
|
-
|
|
|
31
|
+
| File count | > 100 |
|
|
32
|
+
| Average file size (bytes ÷ file count) | < 3 MB |
|
|
33
|
+
|
|
34
|
+
Both conditions have to hold together, so a node with thousands of files that are each
|
|
35
|
+
big enough, or a handful of genuinely tiny ones, doesn't register. The signal comes
|
|
36
|
+
entirely from those file-count and file-size metrics, independent of
|
|
37
|
+
`spark.sql.files.maxPartitionBytes`.
|
|
33
38
|
|
|
34
39
|
## Why it matters
|
|
35
40
|
|
|
@@ -93,6 +93,40 @@ Validated.
|
|
|
93
93
|
|
|
94
94
|
The 4x-median rule flags a slow task, but slow is not the same as broken. The same threshold trips on a skewed key that simply has more data to process, or on a task that spent its time in a GC pause rather than doing extra work, so a flagged task is not automatically a slow host. Small stages make this worse: with only a handful of tasks the median is unstable, and one moderately slow task can look like a straggler against a median computed from too few peers.
|
|
95
95
|
|
|
96
|
+
## Speculation waste {#bottleneck-speculation-waste}
|
|
97
|
+
|
|
98
|
+
<span class="tag">SPEC</span>
|
|
99
|
+
|
|
100
|
+
A straggler is reported by whichever signal, straggler share or speculative-task count,
|
|
101
|
+
best explains it. Speculation waste is a separate, narrower signal: the executor time
|
|
102
|
+
speculative attempts burned without confirming a genuine straggler, whether the
|
|
103
|
+
speculative copy lost the race to the original attempt or the other way around. Either
|
|
104
|
+
way, the losing attempt's executor time is pure waste.
|
|
105
|
+
|
|
106
|
+
### How it's detected
|
|
107
|
+
|
|
108
|
+
| Signal | Fires when |
|
|
109
|
+
|---|---|
|
|
110
|
+
| Wasted speculative attempts in a stage | ≥ 5 |
|
|
111
|
+
| Wasted executor time from those attempts | ≥ 60 seconds |
|
|
112
|
+
|
|
113
|
+
Both conditions have to hold together. Severity tracks the estimated recoverable time as
|
|
114
|
+
a share of the app's total runtime, the same model used across this page: ≥2% is
|
|
115
|
+
critical, ≥0.5% is warning, anything smaller is info.
|
|
116
|
+
|
|
117
|
+
### Why it matters
|
|
118
|
+
|
|
119
|
+
Every wasted speculative attempt occupies an executor slot that could have run other
|
|
120
|
+
work, so a stage generating a lot of speculative waste is trading cluster capacity for
|
|
121
|
+
copies that never pay off.
|
|
122
|
+
|
|
123
|
+
### How to fix it
|
|
124
|
+
|
|
125
|
+
If task durations are naturally variable rather than genuine stragglers, speculation is
|
|
126
|
+
firing too eagerly: tune `spark.speculation.multiplier` (require a bigger gap from the
|
|
127
|
+
median before speculating) or `spark.speculation.quantile` (wait for more of the stage to
|
|
128
|
+
finish first) so fewer ordinary slow tasks get speculated in the first place.
|
|
129
|
+
|
|
96
130
|
## Related
|
|
97
131
|
|
|
98
132
|
- **When the real cause is a skewed key:** [Partitioning](#partitioning), [Adaptive Query Execution](#aqe)
|
|
@@ -70,7 +70,7 @@ The three sizing calls, side by side:
|
|
|
70
70
|
# Collapse many tiny output files without a shuffle (use when distribution is already even)
|
|
71
71
|
df.coalesce(100).write.parquet(path)
|
|
72
72
|
|
|
73
|
-
# Full reshuffle to a target count
|
|
73
|
+
# Full reshuffle to a target count — use when the distribution itself needs rebalancing
|
|
74
74
|
df = df.repartition(200)
|
|
75
75
|
|
|
76
76
|
# Middle ground: reduce partition count but still rebalance (pays a shuffle)
|
|
@@ -4,17 +4,17 @@
|
|
|
4
4
|
|
|
5
5
|
## What it is
|
|
6
6
|
|
|
7
|
-
Executor utilization measures how much of the cluster's allocated
|
|
7
|
+
Executor utilization measures how much of the cluster's allocated core-time a job actually keeps busy. When busy core-time trails the core-time available across the run, the cluster is holding compute (cores and memory) that isn't running any tasks.
|
|
8
8
|
|
|
9
9
|
## How it's detected
|
|
10
10
|
|
|
11
|
-
The signal is the ratio of
|
|
11
|
+
The signal is the ratio of busy executor core-time to the core-time available over the run: peak concurrent cores multiplied by the app's duration.
|
|
12
12
|
|
|
13
|
-
|
|
|
13
|
+
| Signal | Fires when |
|
|
14
14
|
|---|---|
|
|
15
|
-
| < 60% |
|
|
16
|
-
|
|
17
|
-
|
|
15
|
+
| busy core-time / available core-time | < 60% |
|
|
16
|
+
|
|
17
|
+
A finding here always reports at the info level.
|
|
18
18
|
|
|
19
19
|
## Why it matters
|
|
20
20
|
|
|
@@ -41,12 +41,62 @@ spark.conf.set("spark.dynamicAllocation.executorAllocationRatio", "0.5") # 0.5
|
|
|
41
41
|
|
|
42
42
|
A low average-to-peak ratio isn't always waste. A bursty or I/O-bound job legitimately holds executors while tasks wait on external systems rather than burning cores, and the ratio is sensitive to short stages, where a brief spike in allocation skews the average without meaning the cluster was genuinely idle.
|
|
43
43
|
|
|
44
|
-
##
|
|
44
|
+
## Core locality {#bottleneck-core-locality}
|
|
45
|
+
|
|
46
|
+
<span class="tag">LOCAL</span>
|
|
47
|
+
|
|
48
|
+
Idle cores are one half of wasted capacity; this is the other half. A core running a task
|
|
49
|
+
without process- or node-local data placement is still busy, but that task now has to
|
|
50
|
+
fetch its input across the network or from a different process instead of reading it in
|
|
51
|
+
place, work the core wouldn't have to do at all if it were scheduled on data it already
|
|
52
|
+
holds.
|
|
53
|
+
|
|
54
|
+
### How it's detected
|
|
55
|
+
|
|
56
|
+
The share of tasks that ran RACK_LOCAL or ANY, out of every task with a recorded locality
|
|
57
|
+
level, is the non-local task share. NO_PREF tasks stay in the denominator only, since
|
|
58
|
+
shuffle-read stages legitimately report that level without it indicating a placement
|
|
59
|
+
problem. This applies once a run has logged at least 50 total tasks.
|
|
60
|
+
|
|
61
|
+
| Signal | Warning | Critical |
|
|
62
|
+
|---|---|---|
|
|
63
|
+
| Non-local task share | ≥ 15% | ≥ 35% |
|
|
64
|
+
|
|
65
|
+
### Why it matters
|
|
66
|
+
|
|
67
|
+
A task denied process- or node-local placement pulls its input over the network or
|
|
68
|
+
through inter-process I/O instead of reading it from local memory or disk, adding latency
|
|
69
|
+
to every task that lands that way. Spread across a whole run, a high non-local share adds
|
|
70
|
+
up to a meaningful share of total task time spent moving data that a better-placed
|
|
71
|
+
schedule wouldn't have had to move.
|
|
72
|
+
|
|
73
|
+
### How to fix it
|
|
74
|
+
|
|
75
|
+
- Check `spark.locality.wait` (default 3s) and its per-level overrides
|
|
76
|
+
(`.process`/`.node`/`.rack`): a wait set too short gives Spark less time to find a
|
|
77
|
+
local slot before it falls back to a less-local one.
|
|
78
|
+
- Check executor and data colocation: if the executors are running far from where the
|
|
79
|
+
data actually lives (a different rack, a different zone), no amount of locality-wait
|
|
80
|
+
tuning fixes a placement that isn't available to begin with.
|
|
81
|
+
|
|
82
|
+
## Caching opportunity {#bottleneck-caching-opportunity}
|
|
45
83
|
|
|
46
84
|
<span class="tag">CACHE</span>
|
|
47
85
|
|
|
48
86
|
When the same DataFrame, RDD, or input is scanned more than once, low utilization can trace back to repeated recomputation rather than idle cores. Spark keeps nothing between actions: transformations only build a DAG, and once an action finishes its intermediate results are discarded[^6]. Call a second action on the same logic and Spark re-runs the whole DAG from the source, which can mean re-reading a terabyte from S3, re-reading Kafka, or repeating expensive decompression[^6]. Fork that logic into two pipeline branches and you sign up to recompute everything twice[^6].
|
|
49
87
|
|
|
88
|
+
### How it's detected
|
|
89
|
+
|
|
90
|
+
An input relation that recurs across at least 2 SQL executions in one run is flagged by
|
|
91
|
+
its scan format and path identity. A join or union subtree that recurs across at least 2
|
|
92
|
+
executions is matched by operator name, metric names, and its join or filter condition
|
|
93
|
+
normalized to strip per-analysis ids and canonicalize commutative operand order, while
|
|
94
|
+
columns and literal values are kept intact: two executions of the exact same join or
|
|
95
|
+
filter still match even when Spark's internal ids differ between runs, but a genuinely
|
|
96
|
+
different filter or join value does not. A qualifying join or union subtree is reported
|
|
97
|
+
as one finding covering everything beneath it. Every finding here reports at the info
|
|
98
|
+
level.
|
|
99
|
+
|
|
50
100
|
<img class="light-only" src="../diagrams/duplicate-plan-subtree.svg" alt="How two branches that repeat the same scan and operators each recompute it, until a shared cached or reused node lets both read one materialized result.">
|
|
51
101
|
<img class="dark-only" src="../diagrams/duplicate-plan-subtree.dark.svg" alt="How two branches that repeat the same scan and operators each recompute it, until a shared cached or reused node lets both read one materialized result.">
|
|
52
102
|
|