sparkforensics-mcp 0.3.0 → 0.5.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/bin/sparkforensics-mcp.mjs +41 -10
- package/package.json +1 -1
- package/vendor-core/allocation.js +106 -0
- package/vendor-core/analyzer.js +168 -60
- package/vendor-core/check-coverage.js +88 -0
- package/vendor-core/cli/budgets.js +54 -27
- package/vendor-core/cli/collect-run.js +84 -32
- package/vendor-core/cli/regression-budgets.js +83 -0
- package/vendor-core/cli/threshold-config.js +28 -0
- package/vendor-core/comparison-verdict.js +177 -0
- package/vendor-core/core-source-hash.txt +1 -0
- package/vendor-core/core-usage-locality.js +56 -2
- package/vendor-core/detector-docs.js +58 -0
- package/vendor-core/detectors.js +1094 -500
- package/vendor-core/docs-config.js +0 -36
- package/vendor-core/docs-content/chapters/nav-index.json +31 -0
- package/vendor-core/docs-content/detection/cache.md +3 -2
- package/vendor-core/docs-content/detection/cfg.md +9 -8
- package/vendor-core/docs-content/detection/chrn.md +1 -2
- package/vendor-core/docs-content/detection/cold.md +4 -2
- package/vendor-core/docs-content/detection/cstor.md +9 -0
- package/vendor-core/docs-content/detection/fail.md +3 -2
- package/vendor-core/docs-content/detection/gc.md +3 -2
- package/vendor-core/docs-content/detection/host.md +2 -1
- package/vendor-core/docs-content/detection/local.md +1 -1
- package/vendor-core/docs-content/detection/mem.md +5 -2
- package/vendor-core/docs-content/detection/plan.md +2 -1
- package/vendor-core/docs-content/detection/sfail.md +2 -1
- package/vendor-core/docs-content/detection/shape.md +5 -4
- package/vendor-core/docs-content/detection/skew.md +3 -1
- package/vendor-core/docs-content/detection/slow.md +2 -2
- package/vendor-core/docs-content/detection/spec.md +2 -3
- package/vendor-core/docs-content/detection/spill.md +1 -1
- package/vendor-core/docs-site-config.js +3 -0
- package/vendor-core/effective-conf.js +107 -0
- package/vendor-core/efficiency-model.js +8 -6
- package/vendor-core/event-handlers.js +321 -44
- package/vendor-core/event-schemas.js +23 -0
- package/vendor-core/evidence-report.js +432 -115
- package/vendor-core/export-data.js +79 -6
- package/vendor-core/finding-action-label.js +9 -88
- package/vendor-core/finding-filter-predicate.js +9 -0
- package/vendor-core/finding-generic-recommendation.js +26 -105
- package/vendor-core/finding-names.js +28 -45
- package/vendor-core/finding-presentation.js +368 -0
- package/vendor-core/finding-tag-help.js +110 -0
- package/vendor-core/finding-types.js +373 -0
- package/vendor-core/findings-of-type.js +11 -0
- package/vendor-core/format-utils.js +96 -30
- package/vendor-core/html-export.js +51 -0
- package/vendor-core/impact-band.js +21 -8
- package/vendor-core/impact-estimator.js +25 -520
- package/vendor-core/impact-format.js +115 -0
- package/vendor-core/impact-model.js +197 -0
- package/vendor-core/ingest.js +6 -2
- package/vendor-core/intervals.js +13 -0
- package/vendor-core/list-runs.js +7 -5
- package/vendor-core/load-vendored.js +70 -5
- package/vendor-core/mcp-server-factory.js +14 -10
- package/vendor-core/mcp-tools.js +105 -45
- package/vendor-core/model-assembler.js +35 -1
- package/vendor-core/occupancy.js +1 -1
- package/vendor-core/parser-worker.js +2 -2
- package/vendor-core/plan-graph-model.js +3 -2
- package/vendor-core/plan-node-detail.js +1 -1
- package/vendor-core/proxy.js +3 -1
- package/vendor-core/python-stage.js +25 -0
- package/vendor-core/recommendation-rollup.js +70 -3
- package/vendor-core/redact.js +96 -37
- package/vendor-core/remediation.js +20 -0
- package/vendor-core/run-comparison.js +73 -29
- package/vendor-core/run-interpretation.js +291 -0
- package/vendor-core/run-metrics.js +198 -0
- package/vendor-core/run-outcome.js +74 -0
- package/vendor-core/run-payload.js +17 -0
- package/vendor-core/run-shape.js +40 -0
- package/vendor-core/run-totals.js +24 -0
- package/vendor-core/run-verdict.js +352 -0
- package/vendor-core/scaling-sim.js +4 -5
- package/vendor-core/scorecard-estimates.js +63 -0
- package/vendor-core/session-snapshot.js +7 -0
- package/vendor-core/shs-schemas.js +2 -2
- package/vendor-core/spark-memory.js +17 -0
- package/vendor-core/sql-stages.js +11 -0
- package/vendor-core/stage-plan-nodes.js +18 -0
- package/vendor-core/stage-quantiles.js +6 -0
- package/vendor-core/threshold-overrides.js +160 -0
- package/vendor-core/threshold-summary.js +11 -33
- package/vendor-core/types.js +54 -42
- package/vendor-core/wall-clock.js +1 -12
- package/vendor-core/wasted-core-hours.js +12 -9
- package/vendor-core/write-targets.js +312 -0
|
@@ -1,5 +1,3 @@
|
|
|
1
|
-
import { detectorCatalog } from './detectors.js';
|
|
2
|
-
|
|
3
1
|
// Single source of the docs-panel URL surface. DOCS_BASE_DIR is the built docs-site path that
|
|
4
2
|
// serves the tuning reference, relative to the app's origin.
|
|
5
3
|
export const DOCS_BASE_DIR = 'docs/tuning-reference';
|
|
@@ -90,40 +88,6 @@ export function isKnownDocAnchor(anchor ) {
|
|
|
90
88
|
return KNOWN_DOC_ANCHORS.has(String(anchor));
|
|
91
89
|
}
|
|
92
90
|
|
|
93
|
-
// Finding types that never appear as their own DETECTORS entry's type: the parent entry declares
|
|
94
|
-
// a different type because one plan-walk covers two rules (see broadcastSizing). Map to the parent.
|
|
95
|
-
const TYPE_ALIASES = {
|
|
96
|
-
underBroadcast: 'broadcastSizing',
|
|
97
|
-
overBroadcast: 'broadcastSizing',
|
|
98
|
-
};
|
|
99
|
-
|
|
100
|
-
// DETECTORS is static, so this grouping is built once (lazily) instead of re-scanning per
|
|
101
|
-
// docAnchorForType call (called once per TagBadge per render).
|
|
102
|
-
let anchorsByTypeCache ;
|
|
103
|
-
|
|
104
|
-
function anchorsByType() {
|
|
105
|
-
if (!anchorsByTypeCache) {
|
|
106
|
-
anchorsByTypeCache = new Map();
|
|
107
|
-
for (const entry of detectorCatalog()) {
|
|
108
|
-
const anchors = anchorsByTypeCache.get(entry.type) ?? new Set();
|
|
109
|
-
anchors.add(entry.docAnchor);
|
|
110
|
-
anchorsByTypeCache.set(entry.type, anchors);
|
|
111
|
-
}
|
|
112
|
-
}
|
|
113
|
-
return anchorsByTypeCache;
|
|
114
|
-
}
|
|
115
|
-
|
|
116
|
-
/** Resolves a finding `type` to its documented anchor from detectorCatalog(). Returns undefined
|
|
117
|
-
* when entries sharing the type disagree on docAnchor (only configAudit today), or when the
|
|
118
|
-
* resolved anchor isn't in the allowlist (isKnownDocAnchor, the same gate DocsLink uses). */
|
|
119
|
-
export function docAnchorForType(type ) {
|
|
120
|
-
const resolvedType = TYPE_ALIASES[type] ?? type;
|
|
121
|
-
const anchors = anchorsByType().get(resolvedType) ?? new Set();
|
|
122
|
-
if (anchors.size !== 1) return undefined;
|
|
123
|
-
const [anchor] = anchors;
|
|
124
|
-
return anchor && isKnownDocAnchor(anchor) ? anchor : undefined;
|
|
125
|
-
}
|
|
126
|
-
|
|
127
91
|
/** The docAnchor every finding in `findings` carries, or undefined when they disagree or any lacks
|
|
128
92
|
* one. For a badge standing for several findings (a widget header, a grouped row) whose type-level
|
|
129
93
|
* docAnchorForType can't pick one: configAudit's sub-checks each stamp their own anchor. */
|
|
@@ -3,6 +3,7 @@
|
|
|
3
3
|
"anchor": "intro",
|
|
4
4
|
"section": "Getting Started",
|
|
5
5
|
"title": "Introduction",
|
|
6
|
+
"brief": "Guide overview: what the reference covers, the severity-dot legend, and the tag system linking into the Bottleneck Reference.",
|
|
6
7
|
"keywords": [
|
|
7
8
|
"severity-dot",
|
|
8
9
|
"tag-system"
|
|
@@ -14,6 +15,7 @@
|
|
|
14
15
|
"anchor": "spark-architecture",
|
|
15
16
|
"section": "Optimization Guide",
|
|
16
17
|
"title": "Spark Execution Model",
|
|
18
|
+
"brief": "Spark's execution model: Driver/Executor split, DAGScheduler/TaskScheduler roles, stage boundaries at shuffles, and pipelining of narrow transformations.",
|
|
17
19
|
"keywords": [
|
|
18
20
|
"driver-executor",
|
|
19
21
|
"dagscheduler",
|
|
@@ -27,6 +29,7 @@
|
|
|
27
29
|
"anchor": "memory-model",
|
|
28
30
|
"section": "Optimization Guide",
|
|
29
31
|
"title": "Memory Management",
|
|
32
|
+
"brief": "Unified memory management: on-heap/off-heap execution vs. storage regions, spill triggers, and executor memory overhead sizing.",
|
|
30
33
|
"keywords": [
|
|
31
34
|
"unified-memory",
|
|
32
35
|
"execution-memory",
|
|
@@ -41,6 +44,7 @@
|
|
|
41
44
|
"anchor": "partitioning",
|
|
42
45
|
"section": "Optimization Guide",
|
|
43
46
|
"title": "Partitioning",
|
|
47
|
+
"brief": "Partition sizing and reshaping: repartition vs. coalesce mechanics, shuffle-partition tuning, and skew detection thresholds.",
|
|
44
48
|
"keywords": [
|
|
45
49
|
"repartition",
|
|
46
50
|
"coalesce",
|
|
@@ -53,6 +57,7 @@
|
|
|
53
57
|
"anchor": "joins",
|
|
54
58
|
"section": "Optimization Guide",
|
|
55
59
|
"title": "Join Optimization",
|
|
60
|
+
"brief": "Physical join strategy selection: broadcast vs. shuffle joins, join hints, bucketing, and AQE's runtime join-strategy switching.",
|
|
56
61
|
"keywords": [
|
|
57
62
|
"broadcast",
|
|
58
63
|
"shuffle-join",
|
|
@@ -66,6 +71,7 @@
|
|
|
66
71
|
"anchor": "shuffle",
|
|
67
72
|
"section": "Optimization Guide",
|
|
68
73
|
"title": "Shuffle",
|
|
74
|
+
"brief": "Shuffle internals: SortShuffleManager mechanics, shuffle-avoidance paths (bucketing, Storage Partition Join), and shuffle-tuning configs.",
|
|
69
75
|
"keywords": [
|
|
70
76
|
"sortshufflemanager",
|
|
71
77
|
"bucketing",
|
|
@@ -78,6 +84,7 @@
|
|
|
78
84
|
"anchor": "data-formats",
|
|
79
85
|
"section": "Optimization Guide",
|
|
80
86
|
"title": "Data Formats",
|
|
87
|
+
"brief": "Columnar file format tradeoffs: Parquet vs. ORC internals, splittability, predicate pushdown, and compression codec choice.",
|
|
81
88
|
"keywords": [
|
|
82
89
|
"parquet",
|
|
83
90
|
"orc",
|
|
@@ -91,6 +98,7 @@
|
|
|
91
98
|
"anchor": "table-formats",
|
|
92
99
|
"section": "Optimization Guide",
|
|
93
100
|
"title": "Table Formats",
|
|
101
|
+
"brief": "Lakehouse table-format tuning across Delta Lake, Iceberg, and Hudi: file sizing, compaction/OPTIMIZE, clustering (Z-order/liquid/sort), metadata and manifest overhead, snapshot/version expiry, and deletion vectors vs. merge-on-read/copy-on-write.",
|
|
94
102
|
"keywords": [
|
|
95
103
|
"delta",
|
|
96
104
|
"iceberg",
|
|
@@ -107,6 +115,7 @@
|
|
|
107
115
|
"anchor": "caching",
|
|
108
116
|
"section": "Optimization Guide",
|
|
109
117
|
"title": "Caching & Persistence",
|
|
118
|
+
"brief": "cache()/persist() semantics, storage levels, eviction behavior, and checkpointing as a lineage-truncation alternative.",
|
|
110
119
|
"keywords": [
|
|
111
120
|
"cache",
|
|
112
121
|
"persist",
|
|
@@ -120,6 +129,7 @@
|
|
|
120
129
|
"anchor": "pyspark",
|
|
121
130
|
"section": "Optimization Guide",
|
|
122
131
|
"title": "PySpark Specifics",
|
|
132
|
+
"brief": "PySpark-specific performance: UDF serialization cost, Arrow-optimized and pandas UDF types, and Python-worker memory configs.",
|
|
123
133
|
"keywords": [
|
|
124
134
|
"udf",
|
|
125
135
|
"arrow",
|
|
@@ -133,6 +143,7 @@
|
|
|
133
143
|
"anchor": "aqe",
|
|
134
144
|
"section": "Optimization Guide",
|
|
135
145
|
"title": "Adaptive Query Execution",
|
|
146
|
+
"brief": "AQE's runtime re-optimization loop: post-shuffle partition coalescing, join-strategy promotion, and skew-partition splitting.",
|
|
136
147
|
"keywords": [
|
|
137
148
|
"partition-coalescing",
|
|
138
149
|
"join-strategy-promotion",
|
|
@@ -145,6 +156,7 @@
|
|
|
145
156
|
"anchor": "cluster-config",
|
|
146
157
|
"section": "Optimization Guide",
|
|
147
158
|
"title": "Cluster Tuning",
|
|
159
|
+
"brief": "Cluster-level sizing: executor core/memory formulas, container memory budgeting, dynamic allocation, and locality-wait tuning.",
|
|
148
160
|
"keywords": [
|
|
149
161
|
"executor-sizing",
|
|
150
162
|
"container-memory",
|
|
@@ -158,6 +170,7 @@
|
|
|
158
170
|
"anchor": "anti-patterns",
|
|
159
171
|
"section": "Optimization Guide",
|
|
160
172
|
"title": "Anti-Patterns",
|
|
173
|
+
"brief": "Checklist of a dozen recurring Spark performance anti-patterns, each with what it is, how to detect it, why it hurts, and how to fix it.",
|
|
161
174
|
"keywords": [
|
|
162
175
|
"checklist",
|
|
163
176
|
"anti-pattern"
|
|
@@ -169,6 +182,7 @@
|
|
|
169
182
|
"anchor": "bottleneck-skew",
|
|
170
183
|
"section": "Detector Catalog",
|
|
171
184
|
"title": "Task Skew",
|
|
185
|
+
"brief": "Detecting and mitigating task skew: oversized shuffle partitions vs. the median, AQE skew-join splitting, and manual salting as the pre-AQE fallback.",
|
|
172
186
|
"keywords": [
|
|
173
187
|
"skew",
|
|
174
188
|
"aqe-skew-join",
|
|
@@ -181,6 +195,7 @@
|
|
|
181
195
|
"anchor": "bottleneck-shuffle",
|
|
182
196
|
"section": "Detector Catalog",
|
|
183
197
|
"title": "Shuffle I/O",
|
|
198
|
+
"brief": "Detecting and mitigating shuffle I/O bottlenecks: read/write byte thresholds, fetch-wait time, and shuffle-avoidance and tuning levers.",
|
|
184
199
|
"keywords": [
|
|
185
200
|
"shuffle-bytes",
|
|
186
201
|
"fetch-wait"
|
|
@@ -192,6 +207,7 @@
|
|
|
192
207
|
"anchor": "bottleneck-spill",
|
|
193
208
|
"section": "Detector Catalog",
|
|
194
209
|
"title": "Memory / Disk Spill",
|
|
210
|
+
"brief": "Detecting and mitigating memory/disk spill: TaskMemoryManager pressure, the skew-vs-volume spill classification, and matching the fix to the cause.",
|
|
195
211
|
"keywords": [
|
|
196
212
|
"spill",
|
|
197
213
|
"taskmemorymanager"
|
|
@@ -203,6 +219,7 @@
|
|
|
203
219
|
"anchor": "bottleneck-gc",
|
|
204
220
|
"section": "Detector Catalog",
|
|
205
221
|
"title": "GC Pressure",
|
|
222
|
+
"brief": "Detecting and mitigating GC pressure: the gcPct warning/critical thresholds, oversized executors and memory.fraction as causes, and off-heap/collector fixes.",
|
|
206
223
|
"keywords": [
|
|
207
224
|
"gc-pressure",
|
|
208
225
|
"memory-fraction",
|
|
@@ -215,6 +232,7 @@
|
|
|
215
232
|
"anchor": "bottleneck-cold-start",
|
|
216
233
|
"section": "Detector Catalog",
|
|
217
234
|
"title": "Cold Start",
|
|
235
|
+
"brief": "Detecting and mitigating cold start: the firstStageSubmittedAt gap before the first stage is submitted, and executor-sizing/dynamic-allocation readiness fixes.",
|
|
218
236
|
"keywords": [
|
|
219
237
|
"cold-start",
|
|
220
238
|
"dynamic-allocation"
|
|
@@ -226,6 +244,7 @@
|
|
|
226
244
|
"anchor": "bottleneck-utilization",
|
|
227
245
|
"section": "Detector Catalog",
|
|
228
246
|
"title": "Executor Utilization",
|
|
247
|
+
"brief": "Detecting and mitigating low executor utilization: the avg/peak active-executor ratio, under-partitioning and coalesce-caused parallelism loss, and dynamic-allocation over-provisioning.",
|
|
229
248
|
"keywords": [
|
|
230
249
|
"executor-utilization",
|
|
231
250
|
"under-partitioning",
|
|
@@ -238,6 +257,7 @@
|
|
|
238
257
|
"anchor": "bottleneck-slow-host",
|
|
239
258
|
"section": "Detector Catalog",
|
|
240
259
|
"title": "Slow Host",
|
|
260
|
+
"brief": "Detecting and mitigating a slow host: host-wide task duration inflation versus the cluster median, and speculative-execution or locality-wait fixes.",
|
|
241
261
|
"keywords": [
|
|
242
262
|
"slow-host",
|
|
243
263
|
"speculative-execution"
|
|
@@ -249,6 +269,7 @@
|
|
|
249
269
|
"anchor": "bottleneck-failures",
|
|
250
270
|
"section": "Detector Catalog",
|
|
251
271
|
"title": "Task Failures",
|
|
272
|
+
"brief": "Detecting and mitigating task failures: the TaskEndReason taxonomy, failed-task-share thresholds, and distinguishing memory-driven ExecutorLostFailure from application-code ExceptionFailure.",
|
|
252
273
|
"keywords": [
|
|
253
274
|
"taskendreason",
|
|
254
275
|
"executorlostfailure",
|
|
@@ -261,6 +282,7 @@
|
|
|
261
282
|
"anchor": "bottleneck-straggler",
|
|
262
283
|
"section": "Detector Catalog",
|
|
263
284
|
"title": "Stragglers",
|
|
285
|
+
"brief": "Detecting and mitigating straggler tasks: the 4x-median-duration rule, distinguishing stragglers from GC pauses or skew, and speculative-execution tuning.",
|
|
264
286
|
"keywords": [
|
|
265
287
|
"straggler",
|
|
266
288
|
"speculative-execution"
|
|
@@ -272,6 +294,7 @@
|
|
|
272
294
|
"anchor": "bottleneck-retry-waste",
|
|
273
295
|
"section": "Detector Catalog",
|
|
274
296
|
"title": "Retry Waste",
|
|
297
|
+
"brief": "Detecting and mitigating wasted retry compute: executorRunTime burned by superseded task attempts, and distinguishing executor-loss from fetch-failure causes.",
|
|
275
298
|
"keywords": [
|
|
276
299
|
"retry-waste",
|
|
277
300
|
"fetch-failure"
|
|
@@ -283,6 +306,7 @@
|
|
|
283
306
|
"anchor": "bottleneck-tiny-tasks",
|
|
284
307
|
"section": "Detector Catalog",
|
|
285
308
|
"title": "Tiny Tasks",
|
|
309
|
+
"brief": "Detecting and mitigating tiny-task overhead: scheduling/serialization cost dominating once partitions shrink too far, and coalesce/repartition sizing fixes.",
|
|
286
310
|
"keywords": [
|
|
287
311
|
"tiny-tasks",
|
|
288
312
|
"scheduling-overhead"
|
|
@@ -294,6 +318,7 @@
|
|
|
294
318
|
"anchor": "bottleneck-job-failure-rate",
|
|
295
319
|
"section": "Detector Catalog",
|
|
296
320
|
"title": "Job Failure Rate",
|
|
321
|
+
"brief": "Detecting and mitigating job-level failure: the task/stage/executor retry-budget escalation ladder that turns isolated failures into an aborted job.",
|
|
297
322
|
"keywords": [
|
|
298
323
|
"job-failure-rate",
|
|
299
324
|
"retry-budget"
|
|
@@ -305,6 +330,7 @@
|
|
|
305
330
|
"anchor": "bottleneck-memory-utilization",
|
|
306
331
|
"section": "Detector Catalog",
|
|
307
332
|
"title": "Memory Utilization",
|
|
333
|
+
"brief": "Detecting and mitigating held-but-idle executor memory: cores sitting idle below allocated slots, dynamic-allocation idle timeouts that reclaim an executor's memory band, and the low-confidence peak-vs-allocated waste estimate.",
|
|
308
334
|
"keywords": [
|
|
309
335
|
"memory-utilization",
|
|
310
336
|
"idle-timeout"
|
|
@@ -316,6 +342,7 @@
|
|
|
316
342
|
"anchor": "bottleneck-duplicate-plan-subtree",
|
|
317
343
|
"section": "Detector Catalog",
|
|
318
344
|
"title": "Duplicate Plan Subtree",
|
|
345
|
+
"brief": "Detecting a repeated scan or exchange subtree in a physical plan, how exchange reuse and cache()/persist() avoid recomputing it, and why inferring duplication from the plan is a medium-confidence signal.",
|
|
319
346
|
"keywords": [
|
|
320
347
|
"duplicate-plan-subtree",
|
|
321
348
|
"exchange-reuse"
|
|
@@ -327,6 +354,7 @@
|
|
|
327
354
|
"anchor": "bottleneck-small-files",
|
|
328
355
|
"section": "Detector Catalog",
|
|
329
356
|
"title": "Small Files",
|
|
357
|
+
"brief": "Detecting and mitigating the small-files problem on read and write: metadata and I/O overhead of many tiny files, and coalesce()/repartition()/compaction and AQE advisoryPartitionSizeInBytes as output-size controls.",
|
|
330
358
|
"keywords": [
|
|
331
359
|
"small-files",
|
|
332
360
|
"advisorypartitionsizeinbytes"
|
|
@@ -338,6 +366,7 @@
|
|
|
338
366
|
"anchor": "bottleneck-broadcast-sizing",
|
|
339
367
|
"section": "Detector Catalog",
|
|
340
368
|
"title": "Broadcast Sizing",
|
|
369
|
+
"brief": "Detecting under- and over-broadcast joins: spark.sql.autoBroadcastJoinThreshold and its default, when to broadcast the smaller side versus shuffle, and the driver-collection and executor-memory failure modes of broadcasting too large a table.",
|
|
341
370
|
"keywords": [
|
|
342
371
|
"broadcast-threshold",
|
|
343
372
|
"driver-collection"
|
|
@@ -349,6 +378,7 @@
|
|
|
349
378
|
"anchor": "metrics",
|
|
350
379
|
"section": "Reference",
|
|
351
380
|
"title": "Metrics Glossary",
|
|
381
|
+
"brief": "Glossary of every metric surfaced elsewhere on the site: what each one measures, its event-log source field, and what a problematic value looks like.",
|
|
352
382
|
"keywords": [
|
|
353
383
|
"metrics-glossary",
|
|
354
384
|
"event-log"
|
|
@@ -360,6 +390,7 @@
|
|
|
360
390
|
"anchor": "config",
|
|
361
391
|
"section": "Reference",
|
|
362
392
|
"title": "Spark Config Quick-Reference",
|
|
393
|
+
"brief": "Cross-reference of the Spark/PySpark config properties and JVM GC flags discussed elsewhere in the guide, with defaults and version notes.",
|
|
363
394
|
"keywords": [
|
|
364
395
|
"spark-config",
|
|
365
396
|
"gc-flags"
|
|
@@ -1,7 +1,8 @@
|
|
|
1
1
|
### `CACHE`: Caching opportunity {#cache}
|
|
2
2
|
|
|
3
|
-
A reusable dataset (re-read via the same SQL relation more than once
|
|
4
|
-
|
|
3
|
+
A reusable dataset (re-read via the same SQL relation more than once, or a
|
|
4
|
+
join/union result recomputed by two or more executions, matched by plan
|
|
5
|
+
shape) may be worth persisting between stages. Self-flags a confidence that scales with
|
|
5
6
|
how many executions reuse the same relation: reuse is only inferred, from
|
|
6
7
|
plan-scan identity across SQL executions, so confirm the reads really do
|
|
7
8
|
hit the same data before you cache anything.
|
|
@@ -1,15 +1,16 @@
|
|
|
1
1
|
### `CFG`: Configuration audit {#cfg}
|
|
2
2
|
|
|
3
3
|
Flags configuration settings that may cause reliability or efficiency
|
|
4
|
-
problems, independent of any one stage's behavior. Four
|
|
5
|
-
audited today:
|
|
4
|
+
problems, independent of any one stage's behavior. Four checks run:
|
|
6
5
|
|
|
7
6
|
- `spark.shuffle.service.enabled`: flagged when dynamic allocation is on
|
|
8
7
|
but the external shuffle service is off, since shuffle data won't survive
|
|
9
8
|
executor removal.
|
|
10
|
-
- `spark.dynamicAllocation.maxExecutors`:
|
|
11
|
-
|
|
12
|
-
|
|
13
|
-
|
|
14
|
-
|
|
15
|
-
|
|
9
|
+
- `spark.dynamicAllocation.minExecutors`/`maxExecutors`: with dynamic
|
|
10
|
+
allocation on, flagged when min exceeds max (reported on `minExecutors`)
|
|
11
|
+
or when no max is set.
|
|
12
|
+
- `spark.serializer`: flagged when not set to Kryo (the default is the Java
|
|
13
|
+
serializer); `org.apache.spark.serializer.KryoSerializer` is faster and
|
|
14
|
+
produces smaller buffers.
|
|
15
|
+
- `spark.executor.memoryOverhead`: flagged when set below max(384 MiB, 10%
|
|
16
|
+
of executor memory).
|
|
@@ -5,5 +5,4 @@ re-provisioning churn rather than normal scale-down. Raise
|
|
|
5
5
|
`spark.dynamicAllocation.executorIdleTimeout`, or widen the
|
|
6
6
|
`minExecutors`/`maxExecutors` bounds to reduce flapping. Self-flags a
|
|
7
7
|
confidence that scales with how far the short-lived-executor share sits
|
|
8
|
-
past the threshold
|
|
9
|
-
validated against real-world runs.
|
|
8
|
+
past the threshold.
|
|
@@ -1,4 +1,6 @@
|
|
|
1
1
|
### `COLD`: Executor cold start {#cold}
|
|
2
2
|
|
|
3
|
-
|
|
4
|
-
or
|
|
3
|
+
The first stage waited more than 30 s for an executor. Keep a warm pool of
|
|
4
|
+
executors, or, with dynamic allocation, raise
|
|
5
|
+
`spark.dynamicAllocation.minExecutors`/`initialExecutors` so the app doesn't
|
|
6
|
+
scale up from zero.
|
|
@@ -2,3 +2,12 @@
|
|
|
2
2
|
|
|
3
3
|
A persisted dataset is not fully cached in memory, or is spilling to disk.
|
|
4
4
|
Raise executor memory, or shrink the cached dataset.
|
|
5
|
+
|
|
6
|
+
The cached-partition counts and sizes come from `SparkListenerBlockUpdated`
|
|
7
|
+
events, which Spark writes only when
|
|
8
|
+
`spark.eventLog.logBlockUpdates.enabled=true`. Since Spark 2.3 the RDD
|
|
9
|
+
storage figures in stage-submission events are always 0, and they are
|
|
10
|
+
used only as a fallback. When a Spark 2.3+ run persists RDDs but its log has
|
|
11
|
+
neither and block-update logging was off, the check reports that the cache could not be
|
|
12
|
+
checked instead of passing it: turn `spark.eventLog.logBlockUpdates.enabled`
|
|
13
|
+
on and rerun to measure eviction and disk spillover.
|
|
@@ -5,5 +5,6 @@ instability or data-driven errors. The finding names the dominant error: the
|
|
|
5
5
|
exception class, or the executor loss reason (for example "Container killed
|
|
6
6
|
by YARN for exceeding memory limits"). It lists up to five distinct failures,
|
|
7
7
|
each with its message and a short stack excerpt. With redaction on, messages
|
|
8
|
-
and
|
|
9
|
-
paths and data values; class names
|
|
8
|
+
become `[redacted]` and message lines inside excerpts are dropped, since they
|
|
9
|
+
can carry file paths and data values; class names, stack frames and the
|
|
10
|
+
executor loss reason stay (hosts in it are pseudonymized).
|
|
@@ -1,6 +1,7 @@
|
|
|
1
1
|
### `GC`: Garbage collection pressure {#gc}
|
|
2
2
|
|
|
3
|
-
Tasks spend
|
|
3
|
+
Tasks spend more than 10% of executor run time reclaiming memory. Reduce
|
|
4
4
|
object creation: use primitive types, avoid UDFs, or raise executor memory.
|
|
5
|
-
A stage with
|
|
5
|
+
A stage with GC below 5% gets an informational note that executor memory
|
|
6
6
|
may be over-provisioned, only on stages that take at least 0.5% of the run.
|
|
7
|
+
Both need at least 10 s of executor run time on the stage.
|
|
@@ -1,6 +1,7 @@
|
|
|
1
1
|
### `HOST`: Slow host {#host}
|
|
2
2
|
|
|
3
|
-
One executor is much slower than its peers
|
|
3
|
+
One executor is much slower than its peers, or carries most of the stage's
|
|
4
|
+
task time or bytes. It may just hold data locality
|
|
4
5
|
for its tasks or carry one heavy stage, rather than a hardware fault.
|
|
5
6
|
Enable `spark.speculation` to relaunch a lagging task automatically. Only
|
|
6
7
|
flagged on stages that take at least 0.5% of the run.
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
### `LOCAL`: Core usage locality {#local}
|
|
2
2
|
|
|
3
3
|
Tasks run without process- or node-local data placement more often than
|
|
4
|
-
expected. Check
|
|
4
|
+
expected. Check executor/data colocation.
|
|
5
5
|
Self-flags a confidence that scales with the non-local ratio and sample
|
|
6
6
|
size: the thresholds are our own noise floor for this metric.
|
|
@@ -1,10 +1,13 @@
|
|
|
1
1
|
### `MEM`: Memory utilization {#mem}
|
|
2
2
|
|
|
3
|
-
Executor memory or core capacity may be over- or under-provisioned
|
|
3
|
+
Executor memory or core capacity may be over- or under-provisioned: more
|
|
4
|
+
than 50% of available core time ran no task, an executor's heap peaked above
|
|
5
|
+
95% of its allocation, or it stayed below 70%. Some
|
|
4
6
|
detail here needs `spark.eventLog.logStageExecutorMetrics=true` on the run
|
|
5
7
|
being analyzed; without it, per-executor memory usage can't be broken down.
|
|
6
8
|
Review `spark.executor.memory` and executor count if allocated memory sat
|
|
7
9
|
largely idle over the run. That idle-memory variant self-flags a confidence
|
|
8
10
|
that scales with how far the estimated waste sits past a 1.5x buffer: it
|
|
9
|
-
estimates waste from allocated
|
|
11
|
+
estimates waste from allocated memory-time versus task run time (not
|
|
12
|
+
measured heap usage). Check it against
|
|
10
13
|
the Spark UI before resizing anything.
|
|
@@ -7,7 +7,8 @@ this tag:
|
|
|
7
7
|
plan. When the repeats have the same shape but different filters, columns
|
|
8
8
|
or tables, the finding stays informational and claims no time. Only flagged
|
|
9
9
|
when the repeat's stages take at least 0.5% of the run.
|
|
10
|
-
- Small files:
|
|
10
|
+
- Small files: one plan node reads or writes more than 100 files averaging
|
|
11
|
+
under 3 MB. Compact upstream output, or coalesce before writing.
|
|
11
12
|
- Under-broadcast: the smaller side of a Sort Merge Join looks well under
|
|
12
13
|
the broadcast threshold; consider a `broadcast()` hint or raising
|
|
13
14
|
`spark.sql.autoBroadcastJoinThreshold`.
|
|
@@ -2,4 +2,5 @@
|
|
|
2
2
|
|
|
3
3
|
A stage attempt failed outright rather than losing individual tasks within
|
|
4
4
|
it. Inspect the driver log for the failure reason and the job that triggered
|
|
5
|
-
it.
|
|
5
|
+
it. With redaction on, the failure reason becomes `[redacted]`, since it can
|
|
6
|
+
carry file paths and data values.
|
|
@@ -1,6 +1,7 @@
|
|
|
1
1
|
### `SHAPE`: Stage shape {#shape}
|
|
2
2
|
|
|
3
|
-
The stage has an inefficient task count, output shape
|
|
4
|
-
balance:
|
|
5
|
-
stage's wall-clock time
|
|
6
|
-
|
|
3
|
+
The stage has an inefficient task count, output shape (output more than 10×
|
|
4
|
+
input), or task-to-stage balance: one straggler task running for more than
|
|
5
|
+
half the stage's wall-clock time and over 3× the median task, so it alone
|
|
6
|
+
sets when the stage ends. A too-low task count and a straggler are only
|
|
7
|
+
flagged on stages that take at least 0.5% of the run.
|
|
@@ -3,4 +3,6 @@
|
|
|
3
3
|
A small number of tasks take much longer than their peers in the same
|
|
4
4
|
stage. For join-driven skew, enable AQE skew-join handling
|
|
5
5
|
(`spark.sql.adaptive.skewJoin.enabled`); otherwise salt the key or
|
|
6
|
-
repartition on a better key.
|
|
6
|
+
repartition on a better key. Flagged when P95 task time (the longest task,
|
|
7
|
+
on a stage with fewer than 20 tasks) exceeds 3x the median and the
|
|
8
|
+
recoverable tail is at least 0.5% of the run.
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
### `SLOW`: Stage slowness {#slow}
|
|
2
2
|
|
|
3
|
-
A stage ran
|
|
4
|
-
Often a partition-count problem: raise parallelism via
|
|
3
|
+
A stage ran for 15 minutes or more and no slow host was flagged on it. It
|
|
4
|
+
can appear alongside other findings on the same stage. Often a partition-count problem: raise parallelism via
|
|
5
5
|
`spark.sql.shuffle.partitions` or `spark.default.parallelism`, or check for a
|
|
6
6
|
large per-task data volume driving heavy shuffle and spill.
|
|
@@ -2,7 +2,6 @@
|
|
|
2
2
|
|
|
3
3
|
Speculative task attempts used a lot of executor time without confirming a
|
|
4
4
|
genuine straggler. Self-flags a confidence that scales with how far the
|
|
5
|
-
wasted time sits past the threshold
|
|
6
|
-
|
|
7
|
-
just naturally variable rather than genuine stragglers, tune
|
|
5
|
+
wasted time sits past the threshold. If task durations are just naturally
|
|
6
|
+
variable rather than genuine stragglers, tune
|
|
8
7
|
`spark.speculation.multiplier`/`spark.speculation.quantile`.
|
|
@@ -4,4 +4,4 @@ Tasks are writing data out of memory, which slows execution. Two spill
|
|
|
4
4
|
patterns get flagged differently: skew spill, where a few heavy tasks spill
|
|
5
5
|
while most don't (rebalance partitioning), and volume spill, where most
|
|
6
6
|
tasks spill because the data genuinely exceeds available memory (add
|
|
7
|
-
partitions). Only flagged on stages that take at least 0.5% of the run.
|
|
7
|
+
partitions or executor memory). Only flagged on stages that take at least 0.5% of the run.
|
|
@@ -14,3 +14,6 @@ import { typeTag } from './format-utils.js';
|
|
|
14
14
|
export function findingGuideUrl(type ) {
|
|
15
15
|
return `docs/user-guide/understanding-findings.html#${typeTag(type).toLowerCase()}`;
|
|
16
16
|
}
|
|
17
|
+
|
|
18
|
+
// How to find a Spark event log: enabling event logging, the History Server, managed platforms, SSH bastions.
|
|
19
|
+
export const ALTERNATIVE_LOG_RETRIEVAL_URL = 'docs/user-guide/alternative-log-retrieval.html';
|
|
@@ -0,0 +1,107 @@
|
|
|
1
|
+
// The effective Spark configuration of a run for the CLI's JSON output, so a caller can check
|
|
2
|
+
// that a --conf overlay took effect. Every key is listed; a value is withheld when the key or the
|
|
3
|
+
// value matches a secret pattern, and credentials inside URL-like values are stripped. A withheld
|
|
4
|
+
// key is shown as present with no value, and no derivative of the value (no hash, length or
|
|
5
|
+
// prefix) is emitted.
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
export const EFFECTIVE_CONF_SCHEMA_VERSION = 1;
|
|
9
|
+
|
|
10
|
+
/** Spark's default spark.redaction.regex, in JS syntax (the JVM pattern is `(?i)` + this). */
|
|
11
|
+
export const DEFAULT_SECRET_PATTERN = 'secret|password|token|access[.]?key';
|
|
12
|
+
/** Further credential names no Spark default covers, tested on keys (Azure `fs.azure.account.key.*`,
|
|
13
|
+
* `apiKey`, `pwd`, a bare `pass` or `sas` segment, `credential`). */
|
|
14
|
+
export const EXTRA_SECRET_KEY_PATTERN = 'passwd|pwd|(?:^|[._-])pass(?:[._-]|$)|api[._-]?key|account[._-]?key|private[._-]?key|credential|(?:^|[._-])sas(?:[._-]|$)|(?:^|[._-])sig(?:nature)?(?:[._-]|$)';
|
|
15
|
+
const REDACTED = '[redacted]';
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
/** Compiles a JVM regex, which may start with inline flags such as (?i), as a JS RegExp. Null when
|
|
35
|
+
* it uses syntax JS lacks. */
|
|
36
|
+
export function compileJvmPattern(source ) {
|
|
37
|
+
const lead = /^\(\?([a-z]+)\)/.exec(source);
|
|
38
|
+
const flags = new Set ();
|
|
39
|
+
let body = source;
|
|
40
|
+
if (lead) {
|
|
41
|
+
body = source.slice(lead[0].length);
|
|
42
|
+
for (const f of lead[1]) {
|
|
43
|
+
if (f === 'i' || f === 'm' || f === 's') flags.add(f);
|
|
44
|
+
else if (f !== 'u') return null;
|
|
45
|
+
}
|
|
46
|
+
}
|
|
47
|
+
try { return new RegExp(body, [...flags].join('')); } catch { return null; }
|
|
48
|
+
}
|
|
49
|
+
|
|
50
|
+
// Sensitive query/parameter names in a URL-like value: signatures of pre-signed and SAS URLs.
|
|
51
|
+
const SIGNATURE_PARAMS = 'sig|signature|x-amz-signature|x-amz-credential|x-amz-security-token|x-goog-signature|x-goog-credential';
|
|
52
|
+
|
|
53
|
+
// A `name=value` credential parameter in any value (a JDBC or ODBC string, a JVM option, a query
|
|
54
|
+
// string): `password`, `pwd`, `pass`, `apikey`, `accountkey`, `sas`, `sig` and the like, alone or as
|
|
55
|
+
// the tail of a longer name (`-Djavax.net.ssl.keyStorePassword=`).
|
|
56
|
+
const CREDENTIAL_PARAM = new RegExp(
|
|
57
|
+
'((?:^|[\\s;&,?:"\'(]|-D)(?:[\\w.-]*?(?:password|passwd|pwd|secret|token|api[_.-]?key|account[_.-]?key|access[_.-]?key|private[_.-]?key)|pass|sas|credential|'
|
|
58
|
+
+ `${SIGNATURE_PARAMS})\\s*=\\s*)("[^"]*"|'[^']*'|[^;&,\\s"']*)`, 'gi');
|
|
59
|
+
|
|
60
|
+
/** Strips credentials from a value: `name=value` credential parameters anywhere (JDBC
|
|
61
|
+
* `password=`/`pwd=`/`pass=`, `apiKey=`, signature parameters such as Azure SAS `sig=`), and, in a
|
|
62
|
+
* value that looks like a URL, userinfo (user:password@host) and the Oracle thin form
|
|
63
|
+
* (jdbc:oracle:thin:user/password@host). */
|
|
64
|
+
export function stripUrlCredentials(value ) {
|
|
65
|
+
const urlLike = /^[a-z][a-z0-9+.-]*:/i.test(value) || value.includes('://');
|
|
66
|
+
const withoutUserinfo = urlLike
|
|
67
|
+
? value
|
|
68
|
+
.replace(/(:\/\/)[^/\s@?#,]*@/g, `$1${REDACTED}@`)
|
|
69
|
+
.replace(/^((?:[a-z][a-z0-9+.-]*:)+[^\s:/@]+\/)[^\s@]+@/i, `$1${REDACTED}@`)
|
|
70
|
+
: value;
|
|
71
|
+
return withoutUserinfo.replace(CREDENTIAL_PARAM, `$1${REDACTED}`);
|
|
72
|
+
}
|
|
73
|
+
|
|
74
|
+
/** The run's Spark Properties as the CLI reports them; null when the log recorded none. */
|
|
75
|
+
export function buildEffectiveConf(app , options = {}) {
|
|
76
|
+
const config = app?.config;
|
|
77
|
+
if (config == null) return null;
|
|
78
|
+
const defaultPattern = new RegExp(DEFAULT_SECRET_PATTERN, 'i');
|
|
79
|
+
const extraKeyPattern = new RegExp(EXTRA_SECRET_KEY_PATTERN, 'i');
|
|
80
|
+
const jobSource = config['spark.redaction.regex'] ?? null;
|
|
81
|
+
const jobPattern = jobSource != null ? compileJvmPattern(jobSource) : null;
|
|
82
|
+
// A job pattern this runtime cannot evaluate withholds every value rather than guess.
|
|
83
|
+
const jobPatternUsable = jobSource == null || jobPattern != null;
|
|
84
|
+
const userSource = options.userPattern ?? null;
|
|
85
|
+
const userPattern = userSource != null ? compileJvmPattern(userSource) : null;
|
|
86
|
+
if (userSource != null && userPattern == null) throw new Error(`invalid secret pattern: ${userSource}`);
|
|
87
|
+
|
|
88
|
+
const wanted = options.keys ? new Set(options.keys) : null;
|
|
89
|
+
const values = {};
|
|
90
|
+
const maskedKeys = [];
|
|
91
|
+
const matches = (pattern , key , value ) => (pattern?.test(key) ?? false) || (pattern?.test(value) ?? false);
|
|
92
|
+
for (const key of Object.keys(config).sort()) {
|
|
93
|
+
if (wanted && !wanted.has(key)) continue;
|
|
94
|
+
const value = config[key];
|
|
95
|
+
// Spark applies its redaction pattern to the key and the value.
|
|
96
|
+
const masked = !jobPatternUsable
|
|
97
|
+
|| matches(defaultPattern, key, value) || extraKeyPattern.test(key) || matches(jobPattern, key, value) || matches(userPattern, key, value);
|
|
98
|
+
if (masked) maskedKeys.push(key);
|
|
99
|
+
else values[key] = stripUrlCredentials(value);
|
|
100
|
+
}
|
|
101
|
+
return {
|
|
102
|
+
schemaVersion: EFFECTIVE_CONF_SCHEMA_VERSION,
|
|
103
|
+
values,
|
|
104
|
+
maskedKeys,
|
|
105
|
+
absentKeys: wanted ? [...wanted].filter((k) => !(k in config)).sort() : [],
|
|
106
|
+
};
|
|
107
|
+
}
|
|
@@ -1,13 +1,14 @@
|
|
|
1
1
|
// §5 Efficiency/wastage model. DESIGN SPIKE:
|
|
2
2
|
// decomposes available compute-hours into driver-bound vs executor-bound waste, plus two floors.
|
|
3
3
|
import { computeWallClock } from './wall-clock.js';
|
|
4
|
-
import {
|
|
4
|
+
import { computePeakConcurrentCores } from './core-count.js';
|
|
5
5
|
|
|
6
6
|
|
|
7
|
-
export function computeEfficiencyModel({ app, stages, executorsAdded, runAggregates }
|
|
7
|
+
export function computeEfficiencyModel({ app, stages, executorsAdded, executorsRemoved, runAggregates }
|
|
8
8
|
|
|
9
9
|
|
|
10
10
|
|
|
11
|
+
|
|
11
12
|
|
|
12
13
|
)
|
|
13
14
|
|
|
@@ -17,10 +18,11 @@ export function computeEfficiencyModel({ app, stages, executorsAdded, runAggrega
|
|
|
17
18
|
|
|
18
19
|
|
|
19
20
|
{
|
|
20
|
-
//
|
|
21
|
-
//
|
|
22
|
-
//
|
|
23
|
-
|
|
21
|
+
// Capacity is the peak concurrent core count, the one the utilization and idle-cores detectors
|
|
22
|
+
// use, so the Unused core time tile matches the verdict's idle figure. Summing every addition
|
|
23
|
+
// (computeTotalCores) counts a replaced executor's cores alongside its replacement's.
|
|
24
|
+
// `app ?? {}`: callers tolerate a null app (malformed logs); `app!` would crash on app.resources.
|
|
25
|
+
const totalCores = computePeakConcurrentCores(app ?? {}, executorsAdded, executorsRemoved);
|
|
24
26
|
const appDurationMs = (app?.endTime ?? 0) - (app?.startTime ?? 0);
|
|
25
27
|
const wc = computeWallClock(app, stages );
|
|
26
28
|
|