sparkforensics-mcp 0.2.2 → 0.2.4

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (42) hide show
  1. package/bin/sparkforensics-mcp.mjs +11 -5
  2. package/package.json +3 -3
  3. package/vendor-core/cli/collect-run.js +3 -2
  4. package/vendor-core/cli/native-zstd.js +351 -0
  5. package/vendor-core/detectors.js +243 -53
  6. package/vendor-core/docs-config.js +34 -8
  7. package/vendor-core/docs-content/chapters/03-memory-model.md +39 -0
  8. package/vendor-core/docs-content/chapters/11-cluster-config.md +40 -0
  9. package/vendor-core/docs-content/detection/gc.md +2 -0
  10. package/vendor-core/docs-content/detection/host.md +2 -1
  11. package/vendor-core/docs-content/detection/plan.md +3 -1
  12. package/vendor-core/docs-content/detection/shape.md +2 -1
  13. package/vendor-core/docs-content/detection/shfl.md +2 -1
  14. package/vendor-core/docs-content/detection/spill.md +1 -1
  15. package/vendor-core/docs-content/detection/strag.md +2 -1
  16. package/vendor-core/docs-content/detection/tiny.md +2 -1
  17. package/vendor-core/docs-content/tuning/failures.md +1 -1
  18. package/vendor-core/docs-content/tuning/gc.md +11 -4
  19. package/vendor-core/docs-content/tuning/shuffle.md +25 -5
  20. package/vendor-core/docs-content/tuning/skew.md +14 -6
  21. package/vendor-core/docs-content/tuning/small-files.md +12 -7
  22. package/vendor-core/docs-content/tuning/straggler.md +34 -0
  23. package/vendor-core/docs-content/tuning/tiny-tasks.md +1 -1
  24. package/vendor-core/docs-content/tuning/utilization.md +57 -7
  25. package/vendor-core/docs-content/upstream.json +4 -0
  26. package/vendor-core/event-handlers.js +321 -69
  27. package/vendor-core/event-schemas.js +8 -6
  28. package/vendor-core/evidence-report.js +3 -1
  29. package/vendor-core/impact-estimator.js +170 -34
  30. package/vendor-core/mcp-tools.js +20 -7
  31. package/vendor-core/occupancy.js +71 -2
  32. package/vendor-core/parser-worker.js +56 -24
  33. package/vendor-core/plan-summary.js +5 -1
  34. package/vendor-core/run-comparison.js +1 -15
  35. package/vendor-core/shs-fetch.js +18 -7
  36. package/vendor-core/shs-load.js +2 -1
  37. package/vendor-core/stage-quantiles.js +111 -3
  38. package/vendor-core/string-hash.js +15 -0
  39. package/vendor-core/types.js +14 -1
  40. package/vendor-core/vendor/fzstd.js +94 -18
  41. package/vendor-core/zstd-worker-client.js +180 -0
  42. package/vendor-core/zstd-worker.js +103 -0
@@ -4,15 +4,27 @@ import { detectorCatalog } from './detectors.js';
4
4
  // serves the tuning reference, relative to the app's origin.
5
5
  export const DOCS_BASE_DIR = 'docs/tuning-reference';
6
6
 
7
+ // Bottleneck sub-anchors that are sections of another entry's page, not pages of their own.
8
+ // Most live on a sibling bottleneck page; autoscaling-churn and cache-utilization live on the
9
+ // cluster-config and memory-model chapters. Keep in sync with spark-tuning-reference's anchor-map.
10
+ const SUB_ANCHOR_PAGES = {
11
+ 'bottleneck-stage-shape': 'bottleneck-skew',
12
+ 'bottleneck-stage-slowness': 'bottleneck-slow-host',
13
+ 'bottleneck-partition-sizing': 'bottleneck-shuffle',
14
+ 'bottleneck-speculation-waste': 'bottleneck-straggler',
15
+ 'bottleneck-core-locality': 'bottleneck-utilization',
16
+ 'bottleneck-caching-opportunity': 'bottleneck-utilization',
17
+ 'bottleneck-autoscaling-churn': 'cluster-config',
18
+ 'bottleneck-cache-utilization': 'memory-model',
19
+ };
20
+
7
21
  // Some anchors are in-page fragments on another entry's page: config-audit sub-findings and
8
- // metric-glossary entries live on the 'config'/'metrics' pages, and the "stage-*" sub-anchors
9
- // live on their owning bottleneck's page. Keep in sync with spark-tuning-reference's anchor-map.
22
+ // metric-glossary entries live on the 'config'/'metrics' pages, and SUB_ANCHOR_PAGES lists the
23
+ // bottleneck sub-anchors.
10
24
  export function pageForAnchor(anchor ) {
11
25
  if (anchor.startsWith('metric-')) return 'metrics';
12
26
  if (anchor.startsWith('config-')) return 'config';
13
- if (anchor === 'bottleneck-stage-shape') return 'bottleneck-skew';
14
- if (anchor === 'bottleneck-stage-slowness') return 'bottleneck-slow-host';
15
- return anchor;
27
+ return SUB_ANCHOR_PAGES[anchor] ?? anchor;
16
28
  }
17
29
 
18
30
  // Build a docs URL for an anchor like '#bottleneck-skew'. The leading '#' is
@@ -45,7 +57,7 @@ export const METRIC_ANCHORS = {
45
57
  'stage-duration': '#metric-stage-duration',
46
58
  };
47
59
 
48
- // Allowlist of every anchor that exists in the committed tuning-reference markdown. DocsLink
60
+ // Allowlist of every anchor that exists in the generated tuning-reference markdown. DocsLink
49
61
  // renders a link only for anchors here: a detector may declare a docAnchor for an unwritten
50
62
  // section, and gating keeps that from becoming a dead "Learn more" link. docs-config.test.js
51
63
  // asserts this stays a subset of the real ids so it can't drift.
@@ -62,6 +74,9 @@ export const KNOWN_DOC_ANCHORS = new Set([
62
74
  '#bottleneck-memory-utilization', '#bottleneck-broadcast-sizing',
63
75
  '#bottleneck-duplicate-plan-subtree', '#bottleneck-small-files',
64
76
  '#bottleneck-stage-shape', '#bottleneck-stage-slowness',
77
+ '#bottleneck-partition-sizing', '#bottleneck-speculation-waste',
78
+ '#bottleneck-core-locality', '#bottleneck-caching-opportunity',
79
+ '#bottleneck-autoscaling-churn', '#bottleneck-cache-utilization',
65
80
  // Config-audit sections
66
81
  '#config-autoscale-bounds', '#config-memory-overhead', '#config-serializer',
67
82
  '#config-shuffle-service',
@@ -69,7 +84,7 @@ export const KNOWN_DOC_ANCHORS = new Set([
69
84
  ...Object.values(METRIC_ANCHORS),
70
85
  ]);
71
86
 
72
- // True when `anchor` resolves to a real section in the committed
87
+ // True when `anchor` resolves to a real section in the generated
73
88
  // tuning-reference markdown.
74
89
  export function isKnownDocAnchor(anchor ) {
75
90
  return KNOWN_DOC_ANCHORS.has(String(anchor));
@@ -109,10 +124,21 @@ export function docAnchorForType(type ) {
109
124
  return anchor && isKnownDocAnchor(anchor) ? anchor : undefined;
110
125
  }
111
126
 
127
+ /** The docAnchor every finding in `findings` carries, or undefined when they disagree or any lacks
128
+ * one. For a badge standing for several findings (a widget header, a grouped row) whose type-level
129
+ * docAnchorForType can't pick one: configAudit's sub-checks each stamp their own anchor. */
130
+ export function sharedDocAnchor(findings ) {
131
+ const anchors = new Set(findings.map((f) => f.docAnchor));
132
+ if (anchors.size !== 1) return undefined;
133
+ const [anchor] = anchors;
134
+ return anchor;
135
+ }
136
+
112
137
  // Resolves a docAnchorForType() result to the tuning-doc file slug under docs-content/tuning/.
113
138
  // Reuses pageForAnchor's sub-anchor resolution so a sub-anchor like bottleneck-stage-shape maps
114
139
  // to its owning page (skew.md), not a stage-shape.md that never exists. Returns null for anchors
115
- // with no bottleneck tuning doc (metric-/config-prefixed, or a page section like #memory-model).
140
+ // with no bottleneck tuning doc (metric-/config-prefixed, a page section like #memory-model, or a
141
+ // sub-anchor hosted on a chapter like #bottleneck-autoscaling-churn).
116
142
  export function tuningDocSlugForAnchor(anchor ) {
117
143
  const bare = String(anchor).replace(/^#/, '');
118
144
  const page = pageForAnchor(bare);
@@ -58,6 +58,45 @@ When enabling off-heap memory, always pair `spark.memory.offHeap.enabled=true` w
58
58
 
59
59
  > **PySpark:** if you need PySpark's own memory bounded rather than folded silently into the overhead, set `spark.executor.pyspark.memory` explicitly, though its enforcement relies on Python's `resource` module and won't work on Windows and won't actually limit anything on macOS[^4].
60
60
 
61
+ ## Cache utilization {#bottleneck-cache-utilization}
62
+
63
+ <span class="tag">CSTOR</span>
64
+
65
+ Spark's event log carries no block-access or hit-rate events, so cache utilization is read
66
+ from periodic per-RDD storage snapshots instead: how much of a persisted RDD is actually
67
+ resident where it was asked to be.
68
+
69
+ ### How it's detected
70
+
71
+ A persisted RDD, one whose storage level requests memory and/or disk with at least one
72
+ partition actually cached, is evaluated on two independent measures and can register on
73
+ both at once:
74
+
75
+ | Signal | Info | Warning |
76
+ |---|---|---|
77
+ | Cached ratio (cached partitions ÷ total partitions) | < 90% | < 50% |
78
+ | Disk ratio (disk bytes ÷ (memory bytes + disk bytes)), `MEMORY_AND_DISK*` levels only | > 15% | > 40% |
79
+
80
+ Both ratios come from a point-in-time storage snapshot taken at stage-submission events,
81
+ not a runtime block-access count, so confirm against the Spark UI's Storage tab before
82
+ acting. Confidence scales with the RDD's partition count: below 10 partitions the ratio
83
+ is noisy enough to call low confidence, at 50 or more it's high.
84
+
85
+ ### Why it matters
86
+
87
+ A partially cached RDD still pays the recomputation cost for its evicted share on every
88
+ later read, undercutting the reason it was cached in the first place. Disk spillover
89
+ under a `MEMORY_AND_DISK*` level keeps that data around instead of forcing a recompute,
90
+ but a disk read is still far slower than serving it out of memory.
91
+
92
+ ### How to fix it
93
+
94
+ - Increase executor memory, or reduce the size of the dataset being cached, so more of it
95
+ fits in the storage region without eviction.
96
+ - If the RDD is only cached for the occasional narrow scan, weigh whether persisting it
97
+ at all is worth the partial-cache overhead, versus recomputing the slice you actually
98
+ need.
99
+
61
100
  ## Sources
62
101
 
63
102
  [^1]: [Task Memory Management in Spark](https://raw.githubusercontent.com/spoddutur/spark-notes/master/task_memory_management_in_spark.md)
@@ -160,6 +160,46 @@ the target executor count dynamic allocation would otherwise compute[^3]. Raise
160
160
  so proportionally reduces concurrent task slots per executor
161
161
  (`spark.executor.cores / spark.task.cpus`)[^3].
162
162
 
163
+ ## Autoscaling churn {#bottleneck-autoscaling-churn}
164
+
165
+ <span class="tag">CHRN</span> <span class="tag">EXPERIMENTAL</span>
166
+
167
+ Dynamic allocation adds executors once tasks back up and frees them once they go
168
+ idle[^6]. Churn is a narrower failure mode inside that same mechanism: executors that get
169
+ stood up and torn down again before they've done much useful work, cycling through
170
+ provisioning and JVM startup cost repeatedly instead of running tasks.
171
+
172
+ ### How it's detected
173
+
174
+ An executor counts as short-lived when its lifetime, from its `ExecutorAdded` event to
175
+ either its `ExecutorRemoved` event or the end of the run if it was never removed, is
176
+ under 2 minutes. This is evaluated once a run has added at least 5 executors, a floor
177
+ that keeps a two- or three-executor job from registering on ordinary scale-down.
178
+
179
+ | Signal | Warning | Critical |
180
+ |---|---|---|
181
+ | Share of added executors that are short-lived (< 2 min lifetime) | > 30% | > 60% |
182
+
183
+ The 2-minute lifetime cutoff and the 30%/60% split are unvalidated heuristics rather than
184
+ figures published in Spark's own documentation or benchmarks, so treat a finding here as
185
+ a prompt to look at the executor timeline rather than a calibrated verdict.
186
+
187
+ ### Why it matters
188
+
189
+ Every short-lived executor pays the same provisioning and JVM-startup cost as a
190
+ long-lived one, but returns little task work in exchange. A run that keeps flapping
191
+ between scaling up and down spends more of its wall-clock time paying that repeated
192
+ overhead than one that settles into a stable executor count.
193
+
194
+ ### How to fix it
195
+
196
+ - Raise `spark.dynamicAllocation.executorIdleTimeout` (default 60s[^3]) so an executor
197
+ survives a brief lull instead of being released the moment it goes idle, only to be
198
+ requested again shortly after.
199
+ - Widen the gap between `spark.dynamicAllocation.minExecutors` and `.maxExecutors`[^3]:
200
+ bounds set too close together force Spark to repeatedly add and remove executors to
201
+ track small fluctuations in the task backlog instead of settling into a stable range.
202
+
163
203
  ## Sources
164
204
 
165
205
  [^1]: [Distribution of Executors, Cores and Memory for a Spark Application](https://raw.githubusercontent.com/spoddutur/spark-notes/master/distribution_of_executors_cores_and_memory_for_spark_application.md)
@@ -2,3 +2,5 @@
2
2
 
3
3
  Tasks spend an unusually large share of time reclaiming memory. Reduce
4
4
  object creation: use primitive types, avoid UDFs, or raise executor memory.
5
+ A stage with very little GC gets an informational note that executor memory
6
+ may be over-provisioned, only on stages that take at least 0.5% of the run.
@@ -2,4 +2,5 @@
2
2
 
3
3
  One executor is much slower than its peers. It may just hold data locality
4
4
  for its tasks or carry one heavy stage, rather than a hardware fault.
5
- Enable `spark.speculation` to relaunch a lagging task automatically.
5
+ Enable `spark.speculation` to relaunch a lagging task automatically. Only
6
+ flagged on stages that take at least 0.5% of the run.
@@ -4,7 +4,9 @@ Flags patterns in the SQL execution plan worth reviewing. Four checks share
4
4
  this tag:
5
5
 
6
6
  - Duplicate plan subtree: the same subtree recomputed more than once in the
7
- plan.
7
+ plan. When the repeats have the same shape but different filters, columns
8
+ or tables, the finding stays informational and claims no time. Only flagged
9
+ when the repeat's stages take at least 0.5% of the run.
8
10
  - Small files: reading an excessive number of small files.
9
11
  - Under-broadcast: the smaller side of a Sort Merge Join looks well under
10
12
  the broadcast threshold; consider a `broadcast()` hint or raising
@@ -2,4 +2,5 @@
2
2
 
3
3
  The stage has an inefficient task count, output shape, or task-to-stage
4
4
  balance: for example, one straggler task taking a large fraction of the
5
- stage's wall-clock time.
5
+ stage's wall-clock time. A too-low task count is only flagged on stages that
6
+ take at least 0.5% of the run.
@@ -1,4 +1,5 @@
1
1
  ### `SHFL`: Shuffle I/O {#shfl}
2
2
 
3
3
  Tasks move a large amount of intermediate data between stages. Raise
4
- `spark.sql.shuffle.partitions`, or add a broadcast join.
4
+ `spark.sql.shuffle.partitions`, or add a broadcast join. Only flagged on
5
+ stages that take at least 0.5% of the run.
@@ -4,4 +4,4 @@ Tasks are writing data out of memory, which slows execution. Two spill
4
4
  patterns get flagged differently: skew spill, where a few heavy tasks spill
5
5
  while most don't (rebalance partitioning), and volume spill, where most
6
6
  tasks spill because the data genuinely exceeds available memory (add
7
- partitions).
7
+ partitions). Only flagged on stages that take at least 0.5% of the run.
@@ -2,4 +2,5 @@
2
2
 
3
3
  A few tasks run much slower than the rest of their stage. Rule out a GC
4
4
  pause or a slow shuffle fetch before assuming a hardware issue; if a skewed
5
- key is the real cause, that's a candidate for AQE's skew-join handling.
5
+ key is the real cause, that's a candidate for AQE's skew-join handling. Only
6
+ flagged on stages that take at least 0.5% of the run.
@@ -1,4 +1,5 @@
1
1
  ### `TINY`: Tiny tasks {#tiny}
2
2
 
3
3
  Many very short tasks add scheduling overhead out of proportion to the work
4
- each one does. Repartition to fewer, larger tasks.
4
+ each one does. Repartition to fewer, larger tasks. Only flagged on stages that
5
+ take at least 0.5% of the run.
@@ -56,7 +56,7 @@ Config-only levers for memory-driven `ExecutorLostFailure`s and mid-shuffle exec
56
56
 
57
57
  ```properties
58
58
  # Cover off-heap + PySpark memory the default overhead budget does NOT include.
59
- # Default is max(384m, 10% of executor memory); 2g is an example, size to real off-heap/PySpark use.
59
+ # Default is max(384m, 10% of executor memory); 2g is an example — size to real off-heap/PySpark use.
60
60
  spark.executor.memoryOverhead=2g
61
61
 
62
62
  # Keep shuffle output alive when dynamic allocation reclaims an executor mid-shuffle
@@ -16,12 +16,19 @@ pauses and overstate how long the task actually ran.[^1]
16
16
  ## How it's detected
17
17
 
18
18
  Because `jvmGCTime` sits inside `executorRunTime` rather than alongside it, the ratio between
19
- the two gives a bounded read on how much of a task's wall-clock time went to garbage collection:
19
+ the two gives a bounded read on how much of a task's wall-clock time went to garbage collection.
20
+ Both directions below apply once a stage's `executorRunTime` reaches 10 seconds, a floor that
21
+ keeps short stages from reading as noise:
20
22
 
21
- | gcPct = jvmGCTime / executorRunTime | Level |
23
+ | Signal (gcPct = jvmGCTime / executorRunTime) | Fires when |
22
24
  |---|---|
23
- | > 10% | Warning |
24
- | > 20% | Critical |
25
+ | High GC | > 10% |
26
+ | Low GC (cost signal) | < 5% |
27
+
28
+ A stage lands in at most one of these two opposite-direction bands at a time. Severity for
29
+ both tracks the estimated recoverable time as a share of the app's total runtime: ≥2% is
30
+ critical, ≥0.5% is warning, anything smaller is info. Low GC signals a possible memory
31
+ over-provisioning cost rather than time lost to garbage collection.
25
32
 
26
33
  ## Why it matters
27
34
 
@@ -26,11 +26,13 @@ read from local disk, and `totalBytesRead` is their sum[^6].
26
26
 
27
27
  ## How it's detected
28
28
 
29
- | Shuffle read/write bytes | Level |
29
+ | Signal | Fires when |
30
30
  |---|---|
31
- | > 50 MB | Info |
32
- | > 500 MB | Warning |
33
- | > 1 GB | Critical |
31
+ | Shuffle read bytes in a stage | > 50 MB |
32
+
33
+ 50 MB marks a stage as shuffle-heavy enough to flag. Severity then tracks the estimated
34
+ recoverable time as a share of the app's total runtime: ≥2% is critical, ≥0.5% is
35
+ warning, anything smaller is info.
34
36
 
35
37
  Beyond raw byte volume, the executor-side wait is captured by `fetchWaitTime`: time a task
36
38
  spends blocked on a remote shuffle block it needs next, not counting time spent prefetching
@@ -91,7 +93,25 @@ spark.shuffle.file.buffer=1m
91
93
  spark.reducer.maxSizeInFlight=48m
92
94
  ```
93
95
 
94
- ## Partition sizing <span class="tag">PART</span>
96
+ ## Partition sizing {#bottleneck-partition-sizing}
97
+
98
+ <span class="tag">PART</span>
99
+
100
+ ### How it's detected
101
+
102
+ A stage's shuffle-read partition sizes surface three distinct problems:
103
+
104
+ | Signal | Fires when | Level |
105
+ |---|---|---|
106
+ | Largest partition vs. median | > 5× the median **and** > 256 MB | Warning |
107
+ | Low parallelism | ≥ 1 GB of shuffle read spread across ≤ 7 tasks | Warning |
108
+ | Oversized partition | Largest partition ≥ 5 GB | Critical |
109
+
110
+ Skew and low-parallelism severity track the estimated recoverable time as a share of the
111
+ app's total runtime, the same wall-clock model used across this reference. An oversized
112
+ partition is a fixed safety signal instead: it reports critical purely on its own size,
113
+ because a partition past 5 GB is an OOM/crash risk regardless of how much wall-clock time
114
+ fixing it would recover, so it stays critical even on a stage that barely dents the run.
95
115
 
96
116
  Adaptive Query Execution re-optimizes the plan while the query runs: as each shuffle stage
97
117
  materializes, it reads the real shuffle-file sizes and resizes partitions before launching the
@@ -13,10 +13,18 @@ more data than everyone else in the same stage.
13
13
 
14
14
  ## How it's detected
15
15
 
16
- | Signal | Warning | Critical |
17
- |---|---|---|
18
- | P95 / median task duration | > 3× | > 5× |
19
- | Max / median task duration (task count < 20) | > 3× | > 5× |
16
+ | Signal | Fires when |
17
+ |---|---|
18
+ | P95 / median task duration (stage has ≥ 20 tasks) | > 3× |
19
+ | Max / median task duration (stage has < 20 tasks) | > 3× |
20
+
21
+ A 3× ratio marks a stage as skewed. Beyond that ratio, the estimated recoverable time (the
22
+ P95-minus-median, or max-minus-median, delta) needs to clear 0.5% of the app's total
23
+ runtime before it registers, a floor that filters out a 3× ratio sitting on a few
24
+ milliseconds. Severity then tracks that same recoverable-time estimate as a share of the
25
+ app's total runtime: ≥2% is critical, ≥0.5% is warning, anything smaller is info. A 3×
26
+ ratio on a stage that barely dents an eight-hour run typically surfaces as info; the same
27
+ ratio on a stage that dominates a short run reads as critical.
20
28
 
21
29
  ## Why it matters
22
30
 
@@ -53,12 +61,12 @@ The AQE toggle, plus the manual salting fallback as runnable code:
53
61
  ```python
54
62
  from pyspark.sql import functions as fn
55
63
 
56
- # AQE skew-join handling, active once AQE itself is enabled
64
+ # AQE skew-join handling — active once AQE itself is enabled
57
65
  spark.conf.set("spark.sql.adaptive.enabled", "true")
58
66
  spark.conf.set("spark.sql.adaptive.skewJoin.enabled", "true")
59
67
 
60
68
  # Manual salting fallback (pre-AQE): spread the skewed key across N salted variants.
61
- # N is an example, size it to how badly the key is skewed.
69
+ # N is an example — size it to how badly the key is skewed.
62
70
  N = 16
63
71
  salted_big = big.withColumn("salt", (fn.rand() * N).cast("int"))
64
72
  salted_small = small.withColumn(
@@ -21,15 +21,20 @@ inefficient to read a whole block when you only need a few rows[^5].
21
21
 
22
22
  ## How it's detected
23
23
 
24
- Reading the SQL plan surfaces write paths that will emit far more output files
25
- than the data warrants, keying off the one-file-per-output-partition rule[^4]. The signal is
26
- a high output-partition count against a modest data volume, which foreshadows a directory
27
- full of tiny files.
24
+ Each SQL plan node carries file-count and file-size metrics Spark reports directly:
25
+ `number of files read`/`size of files read` on the read side, `number of written
26
+ files`/`written output` on the write side. A node reads as a small-files problem once
27
+ its file count on a given side is high and the resulting average file size is small.
28
28
 
29
- | Signal | What it points to |
29
+ | Signal (per plan node, checked separately for read and write) | Fires when |
30
30
  |---|---|
31
- | Output partition count high vs. data volume | Many tiny files on write |
32
- | Read partitioning below `spark.sql.files.maxPartitionBytes` (128 MB) | Over-split input, excess filesystem I/O[^2][^3] |
31
+ | File count | > 100 |
32
+ | Average file size (bytes ÷ file count) | < 3 MB |
33
+
34
+ Both conditions have to hold together, so a node with thousands of files that are each
35
+ big enough, or a handful of genuinely tiny ones, doesn't register. The signal comes
36
+ entirely from those file-count and file-size metrics, independent of
37
+ `spark.sql.files.maxPartitionBytes`.
33
38
 
34
39
  ## Why it matters
35
40
 
@@ -93,6 +93,40 @@ Validated.
93
93
 
94
94
  The 4x-median rule flags a slow task, but slow is not the same as broken. The same threshold trips on a skewed key that simply has more data to process, or on a task that spent its time in a GC pause rather than doing extra work, so a flagged task is not automatically a slow host. Small stages make this worse: with only a handful of tasks the median is unstable, and one moderately slow task can look like a straggler against a median computed from too few peers.
95
95
 
96
+ ## Speculation waste {#bottleneck-speculation-waste}
97
+
98
+ <span class="tag">SPEC</span>
99
+
100
+ A straggler is reported by whichever signal, straggler share or speculative-task count,
101
+ best explains it. Speculation waste is a separate, narrower signal: the executor time
102
+ speculative attempts burned without confirming a genuine straggler, whether the
103
+ speculative copy lost the race to the original attempt or the other way around. Either
104
+ way, the losing attempt's executor time is pure waste.
105
+
106
+ ### How it's detected
107
+
108
+ | Signal | Fires when |
109
+ |---|---|
110
+ | Wasted speculative attempts in a stage | ≥ 5 |
111
+ | Wasted executor time from those attempts | ≥ 60 seconds |
112
+
113
+ Both conditions have to hold together. Severity tracks the estimated recoverable time as
114
+ a share of the app's total runtime, the same model used across this page: ≥2% is
115
+ critical, ≥0.5% is warning, anything smaller is info.
116
+
117
+ ### Why it matters
118
+
119
+ Every wasted speculative attempt occupies an executor slot that could have run other
120
+ work, so a stage generating a lot of speculative waste is trading cluster capacity for
121
+ copies that never pay off.
122
+
123
+ ### How to fix it
124
+
125
+ If task durations are naturally variable rather than genuine stragglers, speculation is
126
+ firing too eagerly: tune `spark.speculation.multiplier` (require a bigger gap from the
127
+ median before speculating) or `spark.speculation.quantile` (wait for more of the stage to
128
+ finish first) so fewer ordinary slow tasks get speculated in the first place.
129
+
96
130
  ## Related
97
131
 
98
132
  - **When the real cause is a skewed key:** [Partitioning](#partitioning), [Adaptive Query Execution](#aqe)
@@ -70,7 +70,7 @@ The three sizing calls, side by side:
70
70
  # Collapse many tiny output files without a shuffle (use when distribution is already even)
71
71
  df.coalesce(100).write.parquet(path)
72
72
 
73
- # Full reshuffle to a target count, use when the distribution itself needs rebalancing
73
+ # Full reshuffle to a target count — use when the distribution itself needs rebalancing
74
74
  df = df.repartition(200)
75
75
 
76
76
  # Middle ground: reduce partition count but still rebalance (pays a shuffle)
@@ -4,17 +4,17 @@
4
4
 
5
5
  ## What it is
6
6
 
7
- Executor utilization measures how much of the cluster's allocated executor capacity a job actually keeps busy. When the average number of active executors trails the peak number allocated, the cluster is holding compute (cores and memory) that isn't running any tasks.
7
+ Executor utilization measures how much of the cluster's allocated core-time a job actually keeps busy. When busy core-time trails the core-time available across the run, the cluster is holding compute (cores and memory) that isn't running any tasks.
8
8
 
9
9
  ## How it's detected
10
10
 
11
- The signal is the ratio of average active executors to the peak active executor count observed over the job's lifetime:
11
+ The signal is the ratio of busy executor core-time to the core-time available over the run: peak concurrent cores multiplied by the app's duration.
12
12
 
13
- | avg active executors / peak | Level |
13
+ | Signal | Fires when |
14
14
  |---|---|
15
- | < 60% | Info |
16
- | < 40% | Warning |
17
- | < 20% | Critical |
15
+ | busy core-time / available core-time | < 60% |
16
+
17
+ A finding here always reports at the info level.
18
18
 
19
19
  ## Why it matters
20
20
 
@@ -41,12 +41,62 @@ spark.conf.set("spark.dynamicAllocation.executorAllocationRatio", "0.5") # 0.5
41
41
 
42
42
  A low average-to-peak ratio isn't always waste. A bursty or I/O-bound job legitimately holds executors while tasks wait on external systems rather than burning cores, and the ratio is sensitive to short stages, where a brief spike in allocation skews the average without meaning the cluster was genuinely idle.
43
43
 
44
- ## Caching opportunity
44
+ ## Core locality {#bottleneck-core-locality}
45
+
46
+ <span class="tag">LOCAL</span>
47
+
48
+ Idle cores are one half of wasted capacity; this is the other half. A core running a task
49
+ without process- or node-local data placement is still busy, but that task now has to
50
+ fetch its input across the network or from a different process instead of reading it in
51
+ place, work the core wouldn't have to do at all if it were scheduled on data it already
52
+ holds.
53
+
54
+ ### How it's detected
55
+
56
+ The share of tasks that ran RACK_LOCAL or ANY, out of every task with a recorded locality
57
+ level, is the non-local task share. NO_PREF tasks stay in the denominator only, since
58
+ shuffle-read stages legitimately report that level without it indicating a placement
59
+ problem. This applies once a run has logged at least 50 total tasks.
60
+
61
+ | Signal | Warning | Critical |
62
+ |---|---|---|
63
+ | Non-local task share | ≥ 15% | ≥ 35% |
64
+
65
+ ### Why it matters
66
+
67
+ A task denied process- or node-local placement pulls its input over the network or
68
+ through inter-process I/O instead of reading it from local memory or disk, adding latency
69
+ to every task that lands that way. Spread across a whole run, a high non-local share adds
70
+ up to a meaningful share of total task time spent moving data that a better-placed
71
+ schedule wouldn't have had to move.
72
+
73
+ ### How to fix it
74
+
75
+ - Check `spark.locality.wait` (default 3s) and its per-level overrides
76
+ (`.process`/`.node`/`.rack`): a wait set too short gives Spark less time to find a
77
+ local slot before it falls back to a less-local one.
78
+ - Check executor and data colocation: if the executors are running far from where the
79
+ data actually lives (a different rack, a different zone), no amount of locality-wait
80
+ tuning fixes a placement that isn't available to begin with.
81
+
82
+ ## Caching opportunity {#bottleneck-caching-opportunity}
45
83
 
46
84
  <span class="tag">CACHE</span>
47
85
 
48
86
  When the same DataFrame, RDD, or input is scanned more than once, low utilization can trace back to repeated recomputation rather than idle cores. Spark keeps nothing between actions: transformations only build a DAG, and once an action finishes its intermediate results are discarded[^6]. Call a second action on the same logic and Spark re-runs the whole DAG from the source, which can mean re-reading a terabyte from S3, re-reading Kafka, or repeating expensive decompression[^6]. Fork that logic into two pipeline branches and you sign up to recompute everything twice[^6].
49
87
 
88
+ ### How it's detected
89
+
90
+ An input relation that recurs across at least 2 SQL executions in one run is flagged by
91
+ its scan format and path identity. A join or union subtree that recurs across at least 2
92
+ executions is matched by operator name, metric names, and its join or filter condition
93
+ normalized to strip per-analysis ids and canonicalize commutative operand order, while
94
+ columns and literal values are kept intact: two executions of the exact same join or
95
+ filter still match even when Spark's internal ids differ between runs, but a genuinely
96
+ different filter or join value does not. A qualifying join or union subtree is reported
97
+ as one finding covering everything beneath it. Every finding here reports at the info
98
+ level.
99
+
50
100
  <img class="light-only" src="../diagrams/duplicate-plan-subtree.svg" alt="How two branches that repeat the same scan and operators each recompute it, until a shared cached or reused node lets both read one materialized result.">
51
101
  <img class="dark-only" src="../diagrams/duplicate-plan-subtree.dark.svg" alt="How two branches that repeat the same scan and operators each recompute it, until a shared cached or reused node lets both read one materialized result.">
52
102
 
@@ -0,0 +1,4 @@
1
+ {
2
+ "repository": "https://github.com/shuffle-works/spark-tuning-reference.git",
3
+ "commit": "1d0f90dce8c0e04ad0ee6f1f724ae28d821b42cd"
4
+ }