sparkforensics-mcp 0.3.0 → 0.4.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/bin/sparkforensics-mcp.mjs +41 -10
- package/package.json +1 -1
- package/vendor-core/analyzer.js +156 -48
- package/vendor-core/check-coverage.js +88 -0
- package/vendor-core/cli/budgets.js +31 -18
- package/vendor-core/cli/collect-run.js +76 -31
- package/vendor-core/cli/threshold-config.js +28 -0
- package/vendor-core/comparison-verdict.js +177 -0
- package/vendor-core/core-source-hash.txt +1 -0
- package/vendor-core/core-usage-locality.js +56 -2
- package/vendor-core/detector-docs.js +58 -0
- package/vendor-core/detectors.js +918 -458
- package/vendor-core/docs-config.js +0 -36
- package/vendor-core/docs-content/chapters/nav-index.json +31 -0
- package/vendor-core/docs-content/detection/cstor.md +9 -0
- package/vendor-core/docs-site-config.js +3 -0
- package/vendor-core/event-handlers.js +170 -6
- package/vendor-core/event-schemas.js +21 -0
- package/vendor-core/evidence-report.js +421 -112
- package/vendor-core/export-data.js +79 -6
- package/vendor-core/finding-action-label.js +9 -88
- package/vendor-core/finding-filter-predicate.js +9 -0
- package/vendor-core/finding-generic-recommendation.js +6 -104
- package/vendor-core/finding-names.js +21 -45
- package/vendor-core/finding-presentation.js +333 -0
- package/vendor-core/finding-tag-help.js +110 -0
- package/vendor-core/finding-types.js +361 -0
- package/vendor-core/findings-of-type.js +11 -0
- package/vendor-core/format-utils.js +92 -27
- package/vendor-core/html-export.js +51 -0
- package/vendor-core/impact-band.js +21 -8
- package/vendor-core/impact-estimator.js +8 -521
- package/vendor-core/impact-format.js +114 -0
- package/vendor-core/impact-model.js +175 -0
- package/vendor-core/ingest.js +2 -0
- package/vendor-core/intervals.js +13 -0
- package/vendor-core/list-runs.js +2 -3
- package/vendor-core/load-vendored.js +70 -5
- package/vendor-core/mcp-server-factory.js +14 -10
- package/vendor-core/mcp-tools.js +105 -45
- package/vendor-core/model-assembler.js +12 -0
- package/vendor-core/occupancy.js +1 -1
- package/vendor-core/parser-worker.js +1 -1
- package/vendor-core/plan-graph-model.js +3 -2
- package/vendor-core/plan-node-detail.js +1 -1
- package/vendor-core/recommendation-rollup.js +63 -3
- package/vendor-core/redact.js +45 -27
- package/vendor-core/run-comparison.js +32 -7
- package/vendor-core/run-interpretation.js +290 -0
- package/vendor-core/run-outcome.js +74 -0
- package/vendor-core/run-payload.js +17 -0
- package/vendor-core/run-shape.js +40 -0
- package/vendor-core/run-verdict.js +353 -0
- package/vendor-core/scaling-sim.js +4 -5
- package/vendor-core/scorecard-estimates.js +62 -0
- package/vendor-core/sql-stages.js +11 -0
- package/vendor-core/stage-quantiles.js +2 -0
- package/vendor-core/threshold-overrides.js +160 -0
- package/vendor-core/threshold-summary.js +11 -33
- package/vendor-core/types.js +6 -42
- package/vendor-core/wall-clock.js +1 -12
- package/vendor-core/wasted-core-hours.js +2 -2
|
@@ -1,5 +1,3 @@
|
|
|
1
|
-
import { detectorCatalog } from './detectors.js';
|
|
2
|
-
|
|
3
1
|
// Single source of the docs-panel URL surface. DOCS_BASE_DIR is the built docs-site path that
|
|
4
2
|
// serves the tuning reference, relative to the app's origin.
|
|
5
3
|
export const DOCS_BASE_DIR = 'docs/tuning-reference';
|
|
@@ -90,40 +88,6 @@ export function isKnownDocAnchor(anchor ) {
|
|
|
90
88
|
return KNOWN_DOC_ANCHORS.has(String(anchor));
|
|
91
89
|
}
|
|
92
90
|
|
|
93
|
-
// Finding types that never appear as their own DETECTORS entry's type: the parent entry declares
|
|
94
|
-
// a different type because one plan-walk covers two rules (see broadcastSizing). Map to the parent.
|
|
95
|
-
const TYPE_ALIASES = {
|
|
96
|
-
underBroadcast: 'broadcastSizing',
|
|
97
|
-
overBroadcast: 'broadcastSizing',
|
|
98
|
-
};
|
|
99
|
-
|
|
100
|
-
// DETECTORS is static, so this grouping is built once (lazily) instead of re-scanning per
|
|
101
|
-
// docAnchorForType call (called once per TagBadge per render).
|
|
102
|
-
let anchorsByTypeCache ;
|
|
103
|
-
|
|
104
|
-
function anchorsByType() {
|
|
105
|
-
if (!anchorsByTypeCache) {
|
|
106
|
-
anchorsByTypeCache = new Map();
|
|
107
|
-
for (const entry of detectorCatalog()) {
|
|
108
|
-
const anchors = anchorsByTypeCache.get(entry.type) ?? new Set();
|
|
109
|
-
anchors.add(entry.docAnchor);
|
|
110
|
-
anchorsByTypeCache.set(entry.type, anchors);
|
|
111
|
-
}
|
|
112
|
-
}
|
|
113
|
-
return anchorsByTypeCache;
|
|
114
|
-
}
|
|
115
|
-
|
|
116
|
-
/** Resolves a finding `type` to its documented anchor from detectorCatalog(). Returns undefined
|
|
117
|
-
* when entries sharing the type disagree on docAnchor (only configAudit today), or when the
|
|
118
|
-
* resolved anchor isn't in the allowlist (isKnownDocAnchor, the same gate DocsLink uses). */
|
|
119
|
-
export function docAnchorForType(type ) {
|
|
120
|
-
const resolvedType = TYPE_ALIASES[type] ?? type;
|
|
121
|
-
const anchors = anchorsByType().get(resolvedType) ?? new Set();
|
|
122
|
-
if (anchors.size !== 1) return undefined;
|
|
123
|
-
const [anchor] = anchors;
|
|
124
|
-
return anchor && isKnownDocAnchor(anchor) ? anchor : undefined;
|
|
125
|
-
}
|
|
126
|
-
|
|
127
91
|
/** The docAnchor every finding in `findings` carries, or undefined when they disagree or any lacks
|
|
128
92
|
* one. For a badge standing for several findings (a widget header, a grouped row) whose type-level
|
|
129
93
|
* docAnchorForType can't pick one: configAudit's sub-checks each stamp their own anchor. */
|
|
@@ -3,6 +3,7 @@
|
|
|
3
3
|
"anchor": "intro",
|
|
4
4
|
"section": "Getting Started",
|
|
5
5
|
"title": "Introduction",
|
|
6
|
+
"brief": "Guide overview: what the reference covers, the severity-dot legend, and the tag system linking into the Bottleneck Reference.",
|
|
6
7
|
"keywords": [
|
|
7
8
|
"severity-dot",
|
|
8
9
|
"tag-system"
|
|
@@ -14,6 +15,7 @@
|
|
|
14
15
|
"anchor": "spark-architecture",
|
|
15
16
|
"section": "Optimization Guide",
|
|
16
17
|
"title": "Spark Execution Model",
|
|
18
|
+
"brief": "Spark's execution model: Driver/Executor split, DAGScheduler/TaskScheduler roles, stage boundaries at shuffles, and pipelining of narrow transformations.",
|
|
17
19
|
"keywords": [
|
|
18
20
|
"driver-executor",
|
|
19
21
|
"dagscheduler",
|
|
@@ -27,6 +29,7 @@
|
|
|
27
29
|
"anchor": "memory-model",
|
|
28
30
|
"section": "Optimization Guide",
|
|
29
31
|
"title": "Memory Management",
|
|
32
|
+
"brief": "Unified memory management: on-heap/off-heap execution vs. storage regions, spill triggers, and executor memory overhead sizing.",
|
|
30
33
|
"keywords": [
|
|
31
34
|
"unified-memory",
|
|
32
35
|
"execution-memory",
|
|
@@ -41,6 +44,7 @@
|
|
|
41
44
|
"anchor": "partitioning",
|
|
42
45
|
"section": "Optimization Guide",
|
|
43
46
|
"title": "Partitioning",
|
|
47
|
+
"brief": "Partition sizing and reshaping: repartition vs. coalesce mechanics, shuffle-partition tuning, and skew detection thresholds.",
|
|
44
48
|
"keywords": [
|
|
45
49
|
"repartition",
|
|
46
50
|
"coalesce",
|
|
@@ -53,6 +57,7 @@
|
|
|
53
57
|
"anchor": "joins",
|
|
54
58
|
"section": "Optimization Guide",
|
|
55
59
|
"title": "Join Optimization",
|
|
60
|
+
"brief": "Physical join strategy selection: broadcast vs. shuffle joins, join hints, bucketing, and AQE's runtime join-strategy switching.",
|
|
56
61
|
"keywords": [
|
|
57
62
|
"broadcast",
|
|
58
63
|
"shuffle-join",
|
|
@@ -66,6 +71,7 @@
|
|
|
66
71
|
"anchor": "shuffle",
|
|
67
72
|
"section": "Optimization Guide",
|
|
68
73
|
"title": "Shuffle",
|
|
74
|
+
"brief": "Shuffle internals: SortShuffleManager mechanics, shuffle-avoidance paths (bucketing, Storage Partition Join), and shuffle-tuning configs.",
|
|
69
75
|
"keywords": [
|
|
70
76
|
"sortshufflemanager",
|
|
71
77
|
"bucketing",
|
|
@@ -78,6 +84,7 @@
|
|
|
78
84
|
"anchor": "data-formats",
|
|
79
85
|
"section": "Optimization Guide",
|
|
80
86
|
"title": "Data Formats",
|
|
87
|
+
"brief": "Columnar file format tradeoffs: Parquet vs. ORC internals, splittability, predicate pushdown, and compression codec choice.",
|
|
81
88
|
"keywords": [
|
|
82
89
|
"parquet",
|
|
83
90
|
"orc",
|
|
@@ -91,6 +98,7 @@
|
|
|
91
98
|
"anchor": "table-formats",
|
|
92
99
|
"section": "Optimization Guide",
|
|
93
100
|
"title": "Table Formats",
|
|
101
|
+
"brief": "Lakehouse table-format tuning across Delta Lake, Iceberg, and Hudi: file sizing, compaction/OPTIMIZE, clustering (Z-order/liquid/sort), metadata and manifest overhead, snapshot/version expiry, and deletion vectors vs. merge-on-read/copy-on-write.",
|
|
94
102
|
"keywords": [
|
|
95
103
|
"delta",
|
|
96
104
|
"iceberg",
|
|
@@ -107,6 +115,7 @@
|
|
|
107
115
|
"anchor": "caching",
|
|
108
116
|
"section": "Optimization Guide",
|
|
109
117
|
"title": "Caching & Persistence",
|
|
118
|
+
"brief": "cache()/persist() semantics, storage levels, eviction behavior, and checkpointing as a lineage-truncation alternative.",
|
|
110
119
|
"keywords": [
|
|
111
120
|
"cache",
|
|
112
121
|
"persist",
|
|
@@ -120,6 +129,7 @@
|
|
|
120
129
|
"anchor": "pyspark",
|
|
121
130
|
"section": "Optimization Guide",
|
|
122
131
|
"title": "PySpark Specifics",
|
|
132
|
+
"brief": "PySpark-specific performance: UDF serialization cost, Arrow-optimized and pandas UDF types, and Python-worker memory configs.",
|
|
123
133
|
"keywords": [
|
|
124
134
|
"udf",
|
|
125
135
|
"arrow",
|
|
@@ -133,6 +143,7 @@
|
|
|
133
143
|
"anchor": "aqe",
|
|
134
144
|
"section": "Optimization Guide",
|
|
135
145
|
"title": "Adaptive Query Execution",
|
|
146
|
+
"brief": "AQE's runtime re-optimization loop: post-shuffle partition coalescing, join-strategy promotion, and skew-partition splitting.",
|
|
136
147
|
"keywords": [
|
|
137
148
|
"partition-coalescing",
|
|
138
149
|
"join-strategy-promotion",
|
|
@@ -145,6 +156,7 @@
|
|
|
145
156
|
"anchor": "cluster-config",
|
|
146
157
|
"section": "Optimization Guide",
|
|
147
158
|
"title": "Cluster Tuning",
|
|
159
|
+
"brief": "Cluster-level sizing: executor core/memory formulas, container memory budgeting, dynamic allocation, and locality-wait tuning.",
|
|
148
160
|
"keywords": [
|
|
149
161
|
"executor-sizing",
|
|
150
162
|
"container-memory",
|
|
@@ -158,6 +170,7 @@
|
|
|
158
170
|
"anchor": "anti-patterns",
|
|
159
171
|
"section": "Optimization Guide",
|
|
160
172
|
"title": "Anti-Patterns",
|
|
173
|
+
"brief": "Checklist of a dozen recurring Spark performance anti-patterns, each with what it is, how to detect it, why it hurts, and how to fix it.",
|
|
161
174
|
"keywords": [
|
|
162
175
|
"checklist",
|
|
163
176
|
"anti-pattern"
|
|
@@ -169,6 +182,7 @@
|
|
|
169
182
|
"anchor": "bottleneck-skew",
|
|
170
183
|
"section": "Detector Catalog",
|
|
171
184
|
"title": "Task Skew",
|
|
185
|
+
"brief": "Detecting and mitigating task skew: oversized shuffle partitions vs. the median, AQE skew-join splitting, and manual salting as the pre-AQE fallback.",
|
|
172
186
|
"keywords": [
|
|
173
187
|
"skew",
|
|
174
188
|
"aqe-skew-join",
|
|
@@ -181,6 +195,7 @@
|
|
|
181
195
|
"anchor": "bottleneck-shuffle",
|
|
182
196
|
"section": "Detector Catalog",
|
|
183
197
|
"title": "Shuffle I/O",
|
|
198
|
+
"brief": "Detecting and mitigating shuffle I/O bottlenecks: read/write byte thresholds, fetch-wait time, and shuffle-avoidance and tuning levers.",
|
|
184
199
|
"keywords": [
|
|
185
200
|
"shuffle-bytes",
|
|
186
201
|
"fetch-wait"
|
|
@@ -192,6 +207,7 @@
|
|
|
192
207
|
"anchor": "bottleneck-spill",
|
|
193
208
|
"section": "Detector Catalog",
|
|
194
209
|
"title": "Memory / Disk Spill",
|
|
210
|
+
"brief": "Detecting and mitigating memory/disk spill: TaskMemoryManager pressure, the skew-vs-volume spill classification, and matching the fix to the cause.",
|
|
195
211
|
"keywords": [
|
|
196
212
|
"spill",
|
|
197
213
|
"taskmemorymanager"
|
|
@@ -203,6 +219,7 @@
|
|
|
203
219
|
"anchor": "bottleneck-gc",
|
|
204
220
|
"section": "Detector Catalog",
|
|
205
221
|
"title": "GC Pressure",
|
|
222
|
+
"brief": "Detecting and mitigating GC pressure: the gcPct warning/critical thresholds, oversized executors and memory.fraction as causes, and off-heap/collector fixes.",
|
|
206
223
|
"keywords": [
|
|
207
224
|
"gc-pressure",
|
|
208
225
|
"memory-fraction",
|
|
@@ -215,6 +232,7 @@
|
|
|
215
232
|
"anchor": "bottleneck-cold-start",
|
|
216
233
|
"section": "Detector Catalog",
|
|
217
234
|
"title": "Cold Start",
|
|
235
|
+
"brief": "Detecting and mitigating cold start: the firstStageSubmittedAt gap before the first stage is submitted, and executor-sizing/dynamic-allocation readiness fixes.",
|
|
218
236
|
"keywords": [
|
|
219
237
|
"cold-start",
|
|
220
238
|
"dynamic-allocation"
|
|
@@ -226,6 +244,7 @@
|
|
|
226
244
|
"anchor": "bottleneck-utilization",
|
|
227
245
|
"section": "Detector Catalog",
|
|
228
246
|
"title": "Executor Utilization",
|
|
247
|
+
"brief": "Detecting and mitigating low executor utilization: the avg/peak active-executor ratio, under-partitioning and coalesce-caused parallelism loss, and dynamic-allocation over-provisioning.",
|
|
229
248
|
"keywords": [
|
|
230
249
|
"executor-utilization",
|
|
231
250
|
"under-partitioning",
|
|
@@ -238,6 +257,7 @@
|
|
|
238
257
|
"anchor": "bottleneck-slow-host",
|
|
239
258
|
"section": "Detector Catalog",
|
|
240
259
|
"title": "Slow Host",
|
|
260
|
+
"brief": "Detecting and mitigating a slow host: host-wide task duration inflation versus the cluster median, and speculative-execution or locality-wait fixes.",
|
|
241
261
|
"keywords": [
|
|
242
262
|
"slow-host",
|
|
243
263
|
"speculative-execution"
|
|
@@ -249,6 +269,7 @@
|
|
|
249
269
|
"anchor": "bottleneck-failures",
|
|
250
270
|
"section": "Detector Catalog",
|
|
251
271
|
"title": "Task Failures",
|
|
272
|
+
"brief": "Detecting and mitigating task failures: the TaskEndReason taxonomy, failed-task-share thresholds, and distinguishing memory-driven ExecutorLostFailure from application-code ExceptionFailure.",
|
|
252
273
|
"keywords": [
|
|
253
274
|
"taskendreason",
|
|
254
275
|
"executorlostfailure",
|
|
@@ -261,6 +282,7 @@
|
|
|
261
282
|
"anchor": "bottleneck-straggler",
|
|
262
283
|
"section": "Detector Catalog",
|
|
263
284
|
"title": "Stragglers",
|
|
285
|
+
"brief": "Detecting and mitigating straggler tasks: the 4x-median-duration rule, distinguishing stragglers from GC pauses or skew, and speculative-execution tuning.",
|
|
264
286
|
"keywords": [
|
|
265
287
|
"straggler",
|
|
266
288
|
"speculative-execution"
|
|
@@ -272,6 +294,7 @@
|
|
|
272
294
|
"anchor": "bottleneck-retry-waste",
|
|
273
295
|
"section": "Detector Catalog",
|
|
274
296
|
"title": "Retry Waste",
|
|
297
|
+
"brief": "Detecting and mitigating wasted retry compute: executorRunTime burned by superseded task attempts, and distinguishing executor-loss from fetch-failure causes.",
|
|
275
298
|
"keywords": [
|
|
276
299
|
"retry-waste",
|
|
277
300
|
"fetch-failure"
|
|
@@ -283,6 +306,7 @@
|
|
|
283
306
|
"anchor": "bottleneck-tiny-tasks",
|
|
284
307
|
"section": "Detector Catalog",
|
|
285
308
|
"title": "Tiny Tasks",
|
|
309
|
+
"brief": "Detecting and mitigating tiny-task overhead: scheduling/serialization cost dominating once partitions shrink too far, and coalesce/repartition sizing fixes.",
|
|
286
310
|
"keywords": [
|
|
287
311
|
"tiny-tasks",
|
|
288
312
|
"scheduling-overhead"
|
|
@@ -294,6 +318,7 @@
|
|
|
294
318
|
"anchor": "bottleneck-job-failure-rate",
|
|
295
319
|
"section": "Detector Catalog",
|
|
296
320
|
"title": "Job Failure Rate",
|
|
321
|
+
"brief": "Detecting and mitigating job-level failure: the task/stage/executor retry-budget escalation ladder that turns isolated failures into an aborted job.",
|
|
297
322
|
"keywords": [
|
|
298
323
|
"job-failure-rate",
|
|
299
324
|
"retry-budget"
|
|
@@ -305,6 +330,7 @@
|
|
|
305
330
|
"anchor": "bottleneck-memory-utilization",
|
|
306
331
|
"section": "Detector Catalog",
|
|
307
332
|
"title": "Memory Utilization",
|
|
333
|
+
"brief": "Detecting and mitigating held-but-idle executor memory: cores sitting idle below allocated slots, dynamic-allocation idle timeouts that reclaim an executor's memory band, and the low-confidence peak-vs-allocated waste estimate.",
|
|
308
334
|
"keywords": [
|
|
309
335
|
"memory-utilization",
|
|
310
336
|
"idle-timeout"
|
|
@@ -316,6 +342,7 @@
|
|
|
316
342
|
"anchor": "bottleneck-duplicate-plan-subtree",
|
|
317
343
|
"section": "Detector Catalog",
|
|
318
344
|
"title": "Duplicate Plan Subtree",
|
|
345
|
+
"brief": "Detecting a repeated scan or exchange subtree in a physical plan, how exchange reuse and cache()/persist() avoid recomputing it, and why inferring duplication from the plan is a medium-confidence signal.",
|
|
319
346
|
"keywords": [
|
|
320
347
|
"duplicate-plan-subtree",
|
|
321
348
|
"exchange-reuse"
|
|
@@ -327,6 +354,7 @@
|
|
|
327
354
|
"anchor": "bottleneck-small-files",
|
|
328
355
|
"section": "Detector Catalog",
|
|
329
356
|
"title": "Small Files",
|
|
357
|
+
"brief": "Detecting and mitigating the small-files problem on read and write: metadata and I/O overhead of many tiny files, and coalesce()/repartition()/compaction and AQE advisoryPartitionSizeInBytes as output-size controls.",
|
|
330
358
|
"keywords": [
|
|
331
359
|
"small-files",
|
|
332
360
|
"advisorypartitionsizeinbytes"
|
|
@@ -338,6 +366,7 @@
|
|
|
338
366
|
"anchor": "bottleneck-broadcast-sizing",
|
|
339
367
|
"section": "Detector Catalog",
|
|
340
368
|
"title": "Broadcast Sizing",
|
|
369
|
+
"brief": "Detecting under- and over-broadcast joins: spark.sql.autoBroadcastJoinThreshold and its default, when to broadcast the smaller side versus shuffle, and the driver-collection and executor-memory failure modes of broadcasting too large a table.",
|
|
341
370
|
"keywords": [
|
|
342
371
|
"broadcast-threshold",
|
|
343
372
|
"driver-collection"
|
|
@@ -349,6 +378,7 @@
|
|
|
349
378
|
"anchor": "metrics",
|
|
350
379
|
"section": "Reference",
|
|
351
380
|
"title": "Metrics Glossary",
|
|
381
|
+
"brief": "Glossary of every metric surfaced elsewhere on the site: what each one measures, its event-log source field, and what a problematic value looks like.",
|
|
352
382
|
"keywords": [
|
|
353
383
|
"metrics-glossary",
|
|
354
384
|
"event-log"
|
|
@@ -360,6 +390,7 @@
|
|
|
360
390
|
"anchor": "config",
|
|
361
391
|
"section": "Reference",
|
|
362
392
|
"title": "Spark Config Quick-Reference",
|
|
393
|
+
"brief": "Cross-reference of the Spark/PySpark config properties and JVM GC flags discussed elsewhere in the guide, with defaults and version notes.",
|
|
363
394
|
"keywords": [
|
|
364
395
|
"spark-config",
|
|
365
396
|
"gc-flags"
|
|
@@ -2,3 +2,12 @@
|
|
|
2
2
|
|
|
3
3
|
A persisted dataset is not fully cached in memory, or is spilling to disk.
|
|
4
4
|
Raise executor memory, or shrink the cached dataset.
|
|
5
|
+
|
|
6
|
+
The cached-partition counts and sizes come from `SparkListenerBlockUpdated`
|
|
7
|
+
events, which Spark writes only when
|
|
8
|
+
`spark.eventLog.logBlockUpdates.enabled=true`. Since Spark 2.3 the RDD
|
|
9
|
+
storage figures in stage-submission events are always 0, and they are
|
|
10
|
+
used only as a fallback. When a Spark 2.3+ run persists RDDs but its log has
|
|
11
|
+
neither and block-update logging was off, the check reports that the cache could not be
|
|
12
|
+
checked instead of passing it: turn `spark.eventLog.logBlockUpdates.enabled`
|
|
13
|
+
on and rerun to measure eviction and disk spillover.
|
|
@@ -14,3 +14,6 @@ import { typeTag } from './format-utils.js';
|
|
|
14
14
|
export function findingGuideUrl(type ) {
|
|
15
15
|
return `docs/user-guide/understanding-findings.html#${typeTag(type).toLowerCase()}`;
|
|
16
16
|
}
|
|
17
|
+
|
|
18
|
+
// Every other way to get a log (cloud consoles, bastions, copying from storage).
|
|
19
|
+
export const ALTERNATIVE_LOG_RETRIEVAL_URL = 'docs/user-guide/alternative-log-retrieval.html';
|
|
@@ -14,6 +14,7 @@ import {
|
|
|
14
14
|
DriverAccumUpdatesEventSchema,
|
|
15
15
|
ExecutorAddedEventSchema,
|
|
16
16
|
ExecutorRemovedEventSchema,
|
|
17
|
+
BlockUpdatedEventSchema,
|
|
17
18
|
|
|
18
19
|
|
|
19
20
|
} from './event-schemas.js';
|
|
@@ -62,7 +63,22 @@ import { computeRunAggregates } from './run-aggregates.js';
|
|
|
62
63
|
|
|
63
64
|
|
|
64
65
|
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
|
|
65
70
|
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
// Live per-RDD block residency rebuilt from SparkListenerBlockUpdated. Keyed by partition and
|
|
74
|
+
// executor because a block's status is per BlockManager: replicas and re-caches on another
|
|
75
|
+
// executor are separate entries, as in Spark's own AppStatusListener.
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
|
|
81
|
+
|
|
66
82
|
|
|
67
83
|
|
|
68
84
|
|
|
@@ -147,6 +163,12 @@ const MAX_TASK_SAMPLES = 20;
|
|
|
147
163
|
|
|
148
164
|
|
|
149
165
|
|
|
166
|
+
|
|
167
|
+
|
|
168
|
+
|
|
169
|
+
|
|
170
|
+
|
|
171
|
+
|
|
150
172
|
|
|
151
173
|
|
|
152
174
|
|
|
@@ -178,6 +200,10 @@ const MAX_TASK_SAMPLES = 20;
|
|
|
178
200
|
|
|
179
201
|
|
|
180
202
|
|
|
203
|
+
|
|
204
|
+
|
|
205
|
+
|
|
206
|
+
|
|
181
207
|
|
|
182
208
|
|
|
183
209
|
|
|
@@ -367,6 +393,8 @@ export function createState() {
|
|
|
367
393
|
skippedLines: 0,
|
|
368
394
|
accumState: new Map(),
|
|
369
395
|
rddInfo: new Map(),
|
|
396
|
+
rddBlocks: new Map(),
|
|
397
|
+
rddBlockUpdates: 0,
|
|
370
398
|
taskAccumStages: new Map(),
|
|
371
399
|
pendingAdaptiveUpdates: new Map(),
|
|
372
400
|
resolvedPlanExecutions: new Set(),
|
|
@@ -478,6 +506,7 @@ function snapshotEvidenceInputs(state ) {
|
|
|
478
506
|
function appMessage(state ) {
|
|
479
507
|
const evidenceInputs = snapshotEvidenceInputs(state);
|
|
480
508
|
state.app .evidenceInputs = evidenceInputs;
|
|
509
|
+
state.app .rddBlockUpdates = state.rddBlockUpdates;
|
|
481
510
|
return {
|
|
482
511
|
type: 'app',
|
|
483
512
|
data: { ...state.app , rddInfo: snapshotRddInfo(state.rddInfo) },
|
|
@@ -509,8 +538,12 @@ export function accumulateTask(event , state
|
|
|
509
538
|
const stage = state.stages.get(stageId);
|
|
510
539
|
if (!stage) return null;
|
|
511
540
|
// Late TaskEnd for a stage whose StageCompleted already freed taskAttempts (finalizeStage): its
|
|
512
|
-
// stats are already baked into the finalized stage, don't re-add.
|
|
513
|
-
|
|
541
|
+
// stats are already baked into the finalized stage, don't re-add. The one exception is a losing
|
|
542
|
+
// speculative attempt, whose wasted time the finalized stage never saw.
|
|
543
|
+
if (stage.taskAttempts === null) {
|
|
544
|
+
accountLateSpeculativeLoser(event, stage);
|
|
545
|
+
return null;
|
|
546
|
+
}
|
|
514
547
|
|
|
515
548
|
state.evidenceInputs.taskRecords++;
|
|
516
549
|
|
|
@@ -566,6 +599,7 @@ export function accumulateTask(event , state
|
|
|
566
599
|
|
|
567
600
|
if (!existing) {
|
|
568
601
|
stage.taskAttempts.set(key, record);
|
|
602
|
+
if (record.speculative && !record.failed) stage.speculativeWinners.add(key);
|
|
569
603
|
} else if (existing.failed && !record.failed) {
|
|
570
604
|
// A retry succeeded where the earlier attempt failed: the earlier attempt's time was wasted.
|
|
571
605
|
// Spark marks only the speculative COPY's Speculative flag, never the original it raced, so
|
|
@@ -581,6 +615,7 @@ export function accumulateTask(event , state
|
|
|
581
615
|
}
|
|
582
616
|
}
|
|
583
617
|
stage.taskAttempts.set(key, record);
|
|
618
|
+
if (record.speculative) stage.speculativeWinners.add(key);
|
|
584
619
|
} else {
|
|
585
620
|
// Non-winning duplicate (both failed, or a race where a winner is
|
|
586
621
|
// already recorded): its time is waste, its metrics are discarded.
|
|
@@ -599,6 +634,21 @@ export function accumulateTask(event , state
|
|
|
599
634
|
return null;
|
|
600
635
|
}
|
|
601
636
|
|
|
637
|
+
// Spark kills the losing copy of a speculative race only once the stage finishes ("Stage
|
|
638
|
+
// cancelled: Stage finished"), so that loser's TaskEnd normally lands after StageCompleted. Count
|
|
639
|
+
// its time as speculation waste, pairing it the same way accumulateTask does: the late attempt is
|
|
640
|
+
// the speculative copy itself, or the original that a speculative winner beat. Every other stat
|
|
641
|
+
// of a late attempt stays excluded, as the finalized stage already posted them.
|
|
642
|
+
function accountLateSpeculativeLoser(event , stage ) {
|
|
643
|
+
const info = event['Task Info'];
|
|
644
|
+
if (info?.['Index'] == null) return;
|
|
645
|
+
const key = `${event['Stage Attempt ID'] ?? 0}:${info['Index']}`;
|
|
646
|
+
if (info['Speculative'] !== true && !stage.speculativeWinners.has(key)) return;
|
|
647
|
+
stage.speculationWasteMs += (info['Finish Time'] ?? 0) - (info['Launch Time'] ?? 0);
|
|
648
|
+
stage.speculationWastedAttempts++;
|
|
649
|
+
stage.lateSpeculationWaste = true;
|
|
650
|
+
}
|
|
651
|
+
|
|
602
652
|
export function resolvePlanTree(
|
|
603
653
|
rootInfo ,
|
|
604
654
|
accumMap ,
|
|
@@ -828,6 +878,8 @@ export function submitStage(event , st
|
|
|
828
878
|
wastedAttempts: 0,
|
|
829
879
|
speculationWasteMs: 0,
|
|
830
880
|
speculationWastedAttempts: 0,
|
|
881
|
+
speculativeWinners: new Set(),
|
|
882
|
+
lateSpeculationWaste: false,
|
|
831
883
|
executorMetrics: new Map(),
|
|
832
884
|
});
|
|
833
885
|
mergeStageRddInfo(info, id, state);
|
|
@@ -847,11 +899,16 @@ export function mergeStageRddInfo(
|
|
|
847
899
|
const prev = state.rddInfo.get(rddId);
|
|
848
900
|
const stageIds = prev?.stageIds ?? new Set ();
|
|
849
901
|
stageIds.add(id);
|
|
902
|
+
// Block updates are the authoritative source once seen: never let a later RDD Info snapshot
|
|
903
|
+
// (0 on Spark 2.3+) replace them, nor the NONE level an unpersist() leaves on ancestor RDDs
|
|
904
|
+
// listed by later stages.
|
|
905
|
+
const fromBlocks = prev?.storageSource === 'blockUpdates';
|
|
906
|
+
const persisted = Boolean(sl['Use Disk'] || sl['Use Memory']);
|
|
850
907
|
state.rddInfo.set(rddId, {
|
|
851
908
|
id: rddId,
|
|
852
909
|
name: rdd['Name'] ?? '',
|
|
853
910
|
callsite: rdd['Callsite'] ?? '',
|
|
854
|
-
storageLevel: {
|
|
911
|
+
storageLevel: fromBlocks && !persisted ? prev.storageLevel : {
|
|
855
912
|
useDisk: sl['Use Disk'] ?? false,
|
|
856
913
|
useMemory: sl['Use Memory'] ?? false,
|
|
857
914
|
deserialized: sl['Deserialized'] ?? false,
|
|
@@ -861,14 +918,88 @@ export function mergeStageRddInfo(
|
|
|
861
918
|
// Merge forward, don't overwrite: an RDD cached for the first time in THIS stage legitimately
|
|
862
919
|
// reports 0 (snapshot reflects BlockManager state at submission). Keep the last real value on
|
|
863
920
|
// a resubmission instead of regressing to 0.
|
|
864
|
-
numCachedPartitions: rdd['Number of Cached Partitions'] || prev?.numCachedPartitions || 0,
|
|
865
|
-
memorySize: rdd['Memory Size'] || prev?.memorySize || 0,
|
|
866
|
-
diskSize: rdd['Disk Size'] || prev?.diskSize || 0,
|
|
921
|
+
numCachedPartitions: fromBlocks ? prev.numCachedPartitions : rdd['Number of Cached Partitions'] || prev?.numCachedPartitions || 0,
|
|
922
|
+
memorySize: fromBlocks ? prev.memorySize : rdd['Memory Size'] || prev?.memorySize || 0,
|
|
923
|
+
diskSize: fromBlocks ? prev.diskSize : rdd['Disk Size'] || prev?.diskSize || 0,
|
|
924
|
+
storageSource: prev?.storageSource ?? 'rddInfo',
|
|
867
925
|
stageIds,
|
|
868
926
|
});
|
|
869
927
|
}
|
|
870
928
|
}
|
|
871
929
|
|
|
930
|
+
const RDD_BLOCK_ID = /^rdd_(\d+)_(\d+)$/;
|
|
931
|
+
|
|
932
|
+
function dropRddBlock(rdd , key ) {
|
|
933
|
+
const prev = rdd.blocks.get(key);
|
|
934
|
+
if (!prev) return;
|
|
935
|
+
rdd.memorySize -= prev.memorySize;
|
|
936
|
+
rdd.diskSize -= prev.diskSize;
|
|
937
|
+
const replicas = (rdd.replicasByPartition.get(prev.partition) ?? 1) - 1;
|
|
938
|
+
if (replicas > 0) rdd.replicasByPartition.set(prev.partition, replicas);
|
|
939
|
+
else rdd.replicasByPartition.delete(prev.partition);
|
|
940
|
+
rdd.blocks.delete(key);
|
|
941
|
+
}
|
|
942
|
+
|
|
943
|
+
/**
|
|
944
|
+
* Folds one SparkListenerBlockUpdated into its RDD's live residency, then publishes the RDD's
|
|
945
|
+
* peak state to rddInfo: the most partitions resident at once, with the memory/disk bytes at the
|
|
946
|
+
* latest moment that peak held. A peak rather than the final state, because an unpersist() (or
|
|
947
|
+
* the app's own cleanup) removes every block before the log ends, and a final snapshot would
|
|
948
|
+
* read as "nothing was cached". Ties refresh, so a partition dropping from memory to disk after
|
|
949
|
+
* the peak still shows up in diskSize. Non-RDD blocks (broadcast, shuffle, task results) are ignored.
|
|
950
|
+
*/
|
|
951
|
+
export function recordBlockUpdate(event , state ) {
|
|
952
|
+
const info = event['Block Updated Info'];
|
|
953
|
+
const match = RDD_BLOCK_ID.exec(info['Block ID']);
|
|
954
|
+
if (!match) return null;
|
|
955
|
+
state.rddBlockUpdates++;
|
|
956
|
+
const rddId = Number(match[1]);
|
|
957
|
+
const partition = Number(match[2]);
|
|
958
|
+
const sl = info['Storage Level'] ?? {};
|
|
959
|
+
// Spark's StorageLevel.isValid: a removal or eviction reports level NONE.
|
|
960
|
+
const resident = Boolean(sl['Use Memory'] || sl['Use Disk']) && (sl['Replication'] ?? 1) > 0;
|
|
961
|
+
|
|
962
|
+
let rdd = state.rddBlocks.get(rddId);
|
|
963
|
+
if (!rdd) {
|
|
964
|
+
rdd = { blocks: new Map(), replicasByPartition: new Map(), memorySize: 0, diskSize: 0, peakCachedPartitions: 0 };
|
|
965
|
+
state.rddBlocks.set(rddId, rdd);
|
|
966
|
+
}
|
|
967
|
+
const key = `${partition}@${info['Block Manager ID']?.['Executor ID'] ?? ''}`;
|
|
968
|
+
dropRddBlock(rdd, key);
|
|
969
|
+
if (resident) {
|
|
970
|
+
// Sizes count only where the level says the block lives, as Spark's AppStatusListener does: a
|
|
971
|
+
// drop from memory to disk reports Use Memory false but still carries the dropped bytes as
|
|
972
|
+
// Memory Size (BlockManager reports max(memSize, droppedMemorySize)).
|
|
973
|
+
const block = {
|
|
974
|
+
partition,
|
|
975
|
+
memorySize: sl['Use Memory'] ? info['Memory Size'] ?? 0 : 0,
|
|
976
|
+
diskSize: sl['Use Disk'] ? info['Disk Size'] ?? 0 : 0,
|
|
977
|
+
};
|
|
978
|
+
rdd.blocks.set(key, block);
|
|
979
|
+
rdd.memorySize += block.memorySize;
|
|
980
|
+
rdd.diskSize += block.diskSize;
|
|
981
|
+
rdd.replicasByPartition.set(partition, (rdd.replicasByPartition.get(partition) ?? 0) + 1);
|
|
982
|
+
}
|
|
983
|
+
|
|
984
|
+
const record = state.rddInfo.get(rddId) ?? {
|
|
985
|
+
// A block reported before any stage listed its RDD: name and partition count arrive with
|
|
986
|
+
// the next StageSubmitted (mergeStageRddInfo keeps the block-derived sizes).
|
|
987
|
+
id: rddId, name: '', callsite: '',
|
|
988
|
+
storageLevel: { useDisk: Boolean(sl['Use Disk']), useMemory: Boolean(sl['Use Memory']), deserialized: false, replication: sl['Replication'] ?? 1 },
|
|
989
|
+
numPartitions: 0, numCachedPartitions: 0, memorySize: 0, diskSize: 0,
|
|
990
|
+
storageSource: 'blockUpdates' , stageIds: new Set (),
|
|
991
|
+
};
|
|
992
|
+
record.storageSource = 'blockUpdates';
|
|
993
|
+
if (rdd.replicasByPartition.size >= rdd.peakCachedPartitions) {
|
|
994
|
+
rdd.peakCachedPartitions = rdd.replicasByPartition.size;
|
|
995
|
+
record.numCachedPartitions = rdd.peakCachedPartitions;
|
|
996
|
+
record.memorySize = rdd.memorySize;
|
|
997
|
+
record.diskSize = rdd.diskSize;
|
|
998
|
+
}
|
|
999
|
+
state.rddInfo.set(rddId, record);
|
|
1000
|
+
return null;
|
|
1001
|
+
}
|
|
1002
|
+
|
|
872
1003
|
// Spark logs these AFTER SparkListenerStageCompleted, so finalizeStage has already posted the
|
|
873
1004
|
// stage with an empty executorMetrics Map. Keep accumulating worker-side, then re-post all maps
|
|
874
1005
|
// once via `stageExecutorMetrics` just before `done` so main-thread stages are patched before analyze().
|
|
@@ -982,6 +1113,14 @@ export function removeExecutor(event
|
|
|
982
1113
|
reason: event['Removed Reason'] ?? '',
|
|
983
1114
|
};
|
|
984
1115
|
state.executors.removed.push(ev);
|
|
1116
|
+
// Blocks lost with their executor get no BlockUpdated; drop them as Spark's AppStatusListener
|
|
1117
|
+
// does, so a partition re-cached elsewhere isn't counted twice.
|
|
1118
|
+
const suffix = `@${ev.executorId}`;
|
|
1119
|
+
for (const rdd of state.rddBlocks.values()) {
|
|
1120
|
+
for (const key of rdd.blocks.keys()) {
|
|
1121
|
+
if (key.endsWith(suffix)) dropRddBlock(rdd, key);
|
|
1122
|
+
}
|
|
1123
|
+
}
|
|
985
1124
|
return { type: 'executor', data: ev };
|
|
986
1125
|
}
|
|
987
1126
|
|
|
@@ -1058,6 +1197,9 @@ export function processEvent(event , state ) {
|
|
|
1058
1197
|
case 'SparkListenerExecutorRemoved':
|
|
1059
1198
|
return removeExecutor(event, state);
|
|
1060
1199
|
|
|
1200
|
+
case 'SparkListenerBlockUpdated':
|
|
1201
|
+
return recordBlockUpdate(event, state);
|
|
1202
|
+
|
|
1061
1203
|
default:
|
|
1062
1204
|
return assertNever(event);
|
|
1063
1205
|
}
|
|
@@ -1215,7 +1357,14 @@ export function dispatchLine(
|
|
|
1215
1357
|
parseAndDispatch(line, state, emit);
|
|
1216
1358
|
}
|
|
1217
1359
|
|
|
1360
|
+
// With logBlockUpdates on, most BlockUpdated lines are broadcast/shuffle blocks recordBlockUpdate
|
|
1361
|
+
// ignores. Spark writes "Event" first and "Block ID" as a plain string, so those lines can be
|
|
1362
|
+
// dropped on a substring test before paying for JSON.parse.
|
|
1363
|
+
const BLOCK_UPDATED_PREFIX = '{"Event":"SparkListenerBlockUpdated",';
|
|
1364
|
+
const RDD_BLOCK_ID_FRAGMENT = '"Block ID":"rdd_';
|
|
1365
|
+
|
|
1218
1366
|
function parseAndDispatch(line , state , emit ) {
|
|
1367
|
+
if (line.startsWith(BLOCK_UPDATED_PREFIX) && !line.includes(RDD_BLOCK_ID_FRAGMENT)) return;
|
|
1219
1368
|
let parsed ;
|
|
1220
1369
|
try {
|
|
1221
1370
|
parsed = parseTaskEnd(line) ?? JSON.parse(stripPlanDescription(line));
|
|
@@ -1265,11 +1414,26 @@ export function collectStageExecutorMetrics(state )
|
|
|
1265
1414
|
return out;
|
|
1266
1415
|
}
|
|
1267
1416
|
|
|
1417
|
+
// Speculation totals of every stage a late TaskEnd added waste to (accountLateSpeculativeLoser),
|
|
1418
|
+
// re-posted once before `done`: the stage message posted at completion carried the earlier totals.
|
|
1419
|
+
export function collectLateSpeculationWaste(
|
|
1420
|
+
state ,
|
|
1421
|
+
) {
|
|
1422
|
+
const out = new Map ();
|
|
1423
|
+
for (const [id, stage] of state.stages) {
|
|
1424
|
+
if (stage.lateSpeculationWaste) {
|
|
1425
|
+
out.set(id, { speculationWasteMs: stage.speculationWasteMs, speculationWastedAttempts: stage.speculationWastedAttempts });
|
|
1426
|
+
}
|
|
1427
|
+
}
|
|
1428
|
+
return out;
|
|
1429
|
+
}
|
|
1430
|
+
|
|
1268
1431
|
export function emitParseCompletion(state , emit , linesProcessed ) {
|
|
1269
1432
|
// Executions that never ended keep their latest AQE update, as they did before it was deferred.
|
|
1270
1433
|
for (const executionId of [...state.pendingAdaptiveUpdates.keys()]) flushAdaptiveUpdate(executionId, state, emit);
|
|
1271
1434
|
emit({ type: 'progress', pct: 1, linesProcessed });
|
|
1272
1435
|
emit({ type: 'runAggregates', data: computeRunAggregates(state.taskStore) });
|
|
1436
|
+
emit({ type: 'stageSpeculationWaste', data: collectLateSpeculationWaste(state) });
|
|
1273
1437
|
emit({ type: 'stageExecutorMetrics', data: collectStageExecutorMetrics(state) });
|
|
1274
1438
|
emit(appMessage(state));
|
|
1275
1439
|
emit({ type: 'done', skippedLines: state.skippedLines });
|
|
@@ -209,6 +209,26 @@ export const StageSubmittedEventSchema = z.object({
|
|
|
209
209
|
}),
|
|
210
210
|
});
|
|
211
211
|
|
|
212
|
+
// recordBlockUpdate. Written only with spark.eventLog.logBlockUpdates.enabled=true; one event per
|
|
213
|
+
// block status change reported to the driver's BlockManagerMaster (a removal or eviction carries
|
|
214
|
+
// storage level NONE with zero sizes). Only 'rdd_<rddId>_<partition>' blocks are read.
|
|
215
|
+
export const BlockUpdatedEventSchema = z.object({
|
|
216
|
+
Event: z.literal('SparkListenerBlockUpdated'),
|
|
217
|
+
'Block Updated Info': z.object({
|
|
218
|
+
'Block Manager ID': z.object({
|
|
219
|
+
'Executor ID': z.string().optional(),
|
|
220
|
+
}).optional(),
|
|
221
|
+
'Block ID': z.string(),
|
|
222
|
+
'Storage Level': z.object({
|
|
223
|
+
'Use Disk': z.boolean().optional(),
|
|
224
|
+
'Use Memory': z.boolean().optional(),
|
|
225
|
+
Replication: z.number().optional(),
|
|
226
|
+
}).optional(),
|
|
227
|
+
'Memory Size': z.number().optional(),
|
|
228
|
+
'Disk Size': z.number().optional(),
|
|
229
|
+
}),
|
|
230
|
+
});
|
|
231
|
+
|
|
212
232
|
// inline StageCompleted case.
|
|
213
233
|
export const StageCompletedEventSchema = z.object({
|
|
214
234
|
Event: z.literal('SparkListenerStageCompleted'),
|
|
@@ -345,5 +365,6 @@ export const SparkEventSchema = z.discriminatedUnion('Event', [
|
|
|
345
365
|
DriverAccumUpdatesEventSchema,
|
|
346
366
|
ExecutorAddedEventSchema,
|
|
347
367
|
ExecutorRemovedEventSchema,
|
|
368
|
+
BlockUpdatedEventSchema,
|
|
348
369
|
]);
|
|
349
370
|
|