sparkforensics-mcp 0.2.4 → 0.4.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -0
- package/README.md +7 -1
- package/bin/sparkforensics-mcp.mjs +41 -10
- package/package.json +1 -1
- package/vendor-core/analyzer.js +156 -48
- package/vendor-core/check-coverage.js +88 -0
- package/vendor-core/cli/budgets.js +31 -18
- package/vendor-core/cli/collect-run.js +76 -31
- package/vendor-core/cli/native-zstd.js +2 -2
- package/vendor-core/cli/threshold-config.js +28 -0
- package/vendor-core/comparison-verdict.js +177 -0
- package/vendor-core/core-source-hash.txt +1 -0
- package/vendor-core/core-usage-locality.js +56 -2
- package/vendor-core/detector-docs.js +58 -0
- package/vendor-core/detectors.js +933 -459
- package/vendor-core/docs-config.js +0 -36
- package/vendor-core/docs-content/chapters/nav-index.json +31 -0
- package/vendor-core/docs-content/detection/cstor.md +9 -0
- package/vendor-core/docs-content/detection/fail.md +6 -2
- package/vendor-core/docs-site-config.js +3 -0
- package/vendor-core/event-handlers.js +191 -6
- package/vendor-core/event-schemas.js +29 -0
- package/vendor-core/evidence-report.js +440 -112
- package/vendor-core/export-data.js +79 -6
- package/vendor-core/finding-action-label.js +9 -88
- package/vendor-core/finding-filter-predicate.js +9 -0
- package/vendor-core/finding-generic-recommendation.js +6 -104
- package/vendor-core/finding-names.js +21 -45
- package/vendor-core/finding-presentation.js +333 -0
- package/vendor-core/finding-tag-help.js +110 -0
- package/vendor-core/finding-types.js +361 -0
- package/vendor-core/findings-of-type.js +11 -0
- package/vendor-core/format-utils.js +92 -27
- package/vendor-core/html-export.js +51 -0
- package/vendor-core/impact-band.js +21 -8
- package/vendor-core/impact-estimator.js +8 -521
- package/vendor-core/impact-format.js +114 -0
- package/vendor-core/impact-model.js +175 -0
- package/vendor-core/ingest.js +2 -0
- package/vendor-core/intervals.js +13 -0
- package/vendor-core/list-runs.js +2 -3
- package/vendor-core/load-vendored.js +70 -5
- package/vendor-core/mcp-server-factory.js +14 -10
- package/vendor-core/mcp-tools.js +105 -45
- package/vendor-core/model-assembler.js +12 -0
- package/vendor-core/occupancy.js +1 -1
- package/vendor-core/parser-worker.js +22 -5
- package/vendor-core/plan-graph-model.js +3 -2
- package/vendor-core/plan-node-detail.js +1 -1
- package/vendor-core/recommendation-rollup.js +63 -3
- package/vendor-core/redact.js +68 -28
- package/vendor-core/run-comparison.js +40 -7
- package/vendor-core/run-interpretation.js +290 -0
- package/vendor-core/run-outcome.js +74 -0
- package/vendor-core/run-payload.js +17 -0
- package/vendor-core/run-shape.js +40 -0
- package/vendor-core/run-verdict.js +353 -0
- package/vendor-core/scaling-sim.js +4 -5
- package/vendor-core/scorecard-estimates.js +62 -0
- package/vendor-core/shs-fetch.js +175 -65
- package/vendor-core/shs-load.js +1 -1
- package/vendor-core/sql-stages.js +11 -0
- package/vendor-core/stage-quantiles.js +14 -0
- package/vendor-core/task-failure.js +151 -0
- package/vendor-core/threshold-overrides.js +160 -0
- package/vendor-core/threshold-summary.js +11 -33
- package/vendor-core/types.js +6 -42
- package/vendor-core/vendor/fflate.js +1 -1
- package/vendor-core/wall-clock.js +1 -12
- package/vendor-core/wasted-core-hours.js +2 -2
- package/vendor-core/zip-archive.js +167 -0
package/vendor-core/detectors.js
CHANGED
|
@@ -3,15 +3,32 @@ import { scanRelationId } from './plan-summary.js';
|
|
|
3
3
|
import { computePeakConcurrentCores, computePeakConcurrentExecutorCount } from './core-count.js';
|
|
4
4
|
import { walkPlanTree } from './plan-tree-walk.js';
|
|
5
5
|
import { computeCoreLocalityRatio } from './core-locality-ratio.js';
|
|
6
|
-
import {
|
|
6
|
+
import { tailRecoveryMs, tailRemovedWorkMs, stragglerFixLongestTaskMs, } from './occupancy.js';
|
|
7
|
+
import { IMPACT_FLOOR_PCT_WARN, IMPACT_FLOOR_PCT_CRIT, appDurationMs } from './impact-band.js';
|
|
8
|
+
import {
|
|
9
|
+
BROADCAST_BANDWIDTH_BPS, EXECUTOR_STARTUP_OVERHEAD_MS, FILE_OPEN_OVERHEAD_MS, IDEAL_BYTES_PER_PARTITION_TASK,
|
|
10
|
+
NETWORK_FETCH_PENALTY_MS, RE_READ_THROUGHPUT_BPS, SHUFFLE_THROUGHPUT_BPS, SPILL_IO_THROUGHPUT_BPS, TAIL_CLAIM,
|
|
11
|
+
TASK_SCHEDULING_OVERHEAD_MS, costOnly, fetchWaitWallClockMs, measuredTaskOverhead, multiStageImpact, noWasteModel,
|
|
12
|
+
retryWallClockMs, singleStageImpact, stageIoParallelism, stageMappableWasteOrCostOnly, tasksMostlyIdle,
|
|
13
|
+
|
|
14
|
+
} from './impact-model.js';
|
|
7
15
|
import { isExchangeNode, isBroadcastExchangeNode } from './plan-node-detail.js';
|
|
16
|
+
import { stageIdsForSqlExec } from './sql-stages.js';
|
|
8
17
|
import { cyrb53 } from './string-hash.js';
|
|
9
|
-
|
|
18
|
+
import { MAX_FAILURE_GROUPS, describeTaskFailure, } from './task-failure.js';
|
|
19
|
+
|
|
20
|
+
|
|
10
21
|
|
|
11
22
|
const MB = 1024 * 1024;
|
|
12
23
|
const GB = 1024 * MB;
|
|
13
24
|
const TB = 1024 * GB;
|
|
14
25
|
|
|
26
|
+
// Labels a byte threshold in this file's binary units, so a 1 GiB default reads '1 GB'.
|
|
27
|
+
function binaryThresholdLabel(bytes ) {
|
|
28
|
+
const [unit, size] = ([['GB', GB], ['MB', MB], ['KB', 1024]] ).find(([, u]) => bytes >= u) ?? ['bytes', 1];
|
|
29
|
+
return `${Math.round(bytes / size * 10) / 10} ${unit}`;
|
|
30
|
+
}
|
|
31
|
+
|
|
15
32
|
// Local runtime shapes.
|
|
16
33
|
//
|
|
17
34
|
// types.ts's Stage/SqlExecution/SparkAppInfo describe the posted AppModel surface for the view
|
|
@@ -26,16 +43,7 @@ const TB = 1024 * GB;
|
|
|
26
43
|
|
|
27
44
|
|
|
28
45
|
|
|
29
|
-
|
|
30
|
-
|
|
31
|
-
|
|
32
|
-
|
|
33
|
-
|
|
34
|
-
|
|
35
|
-
|
|
36
|
-
|
|
37
|
-
|
|
38
|
-
|
|
46
|
+
|
|
39
47
|
// Per-executor snapshot from a StageExecutorMetrics event: a loose bag of Spark's
|
|
40
48
|
// ExecutorMetrics field names, only a few of which any detector reads.
|
|
41
49
|
|
|
@@ -73,6 +81,7 @@ const TB = 1024 * GB;
|
|
|
73
81
|
|
|
74
82
|
|
|
75
83
|
|
|
84
|
+
|
|
76
85
|
|
|
77
86
|
|
|
78
87
|
|
|
@@ -101,6 +110,7 @@ const TB = 1024 * GB;
|
|
|
101
110
|
|
|
102
111
|
|
|
103
112
|
|
|
113
|
+
|
|
104
114
|
|
|
105
115
|
|
|
106
116
|
|
|
@@ -113,7 +123,10 @@ const TB = 1024 * GB;
|
|
|
113
123
|
|
|
114
124
|
|
|
115
125
|
|
|
126
|
+
|
|
116
127
|
|
|
128
|
+
|
|
129
|
+
|
|
117
130
|
|
|
118
131
|
|
|
119
132
|
|
|
@@ -126,25 +139,22 @@ const TB = 1024 * GB;
|
|
|
126
139
|
|
|
127
140
|
|
|
128
141
|
|
|
129
|
-
// The full context analyze() passes as every stage/sql detect()'s second arg, and as the
|
|
130
|
-
// arg for 'app' scope. 'config' scope gets a narrower `{ app }` (auditConfig)
|
|
142
|
+
// The full context analyze() passes as every stage/sql detect()'s second arg, and as the first
|
|
143
|
+
// arg for 'app' scope. 'config' scope gets a narrower `{ app }` (auditConfig) instead.
|
|
131
144
|
//
|
|
132
|
-
// `app` is
|
|
133
|
-
//
|
|
134
|
-
// malformed/incomplete logs. Despite the type annotation, app may be null in edge cases, and
|
|
135
|
-
// these detectors handle it gracefully. auditConfig's config-scope target keeps `app` nullable
|
|
136
|
-
// instead.
|
|
145
|
+
// `app` is nullable as on AppModel.app: a malformed or cut-short log may have no app record, so
|
|
146
|
+
// every app-reading detector guards it. `runAggregates.busyCoreMs` is optional for the same reason.
|
|
137
147
|
|
|
138
|
-
|
|
148
|
+
|
|
139
149
|
|
|
140
150
|
|
|
141
151
|
|
|
142
152
|
|
|
143
153
|
|
|
144
|
-
|
|
145
|
-
|
|
154
|
+
|
|
146
155
|
|
|
147
|
-
|
|
156
|
+
|
|
157
|
+
|
|
148
158
|
|
|
149
159
|
|
|
150
160
|
// auditConfig(app) calls every config-scope detect({ app }) with whatever appModel.app is
|
|
@@ -188,18 +198,6 @@ function pickDominantReason(reasons )
|
|
|
188
198
|
return [...reasons].sort((a, b) => b.count - a.count)[0].reason;
|
|
189
199
|
}
|
|
190
200
|
|
|
191
|
-
// Shared by every scope:'sql' detector. sql.get(id).stageIds is always empty (parser-worker
|
|
192
|
-
// never populates it), so stage linkage is derived from each stage's own sqlExecutionId.
|
|
193
|
-
// Minimal param shape (not DetectorStage): external callers pass Map<StageId, Stage>.
|
|
194
|
-
export function stageIdsForSqlExec(
|
|
195
|
-
executionId ,
|
|
196
|
-
stages ,
|
|
197
|
-
) {
|
|
198
|
-
const out = [];
|
|
199
|
-
for (const s of stages.values()) if (s.sqlExecutionId === executionId) out.push(s.id);
|
|
200
|
-
return out;
|
|
201
|
-
}
|
|
202
|
-
|
|
203
201
|
// Shared by the three Plan Advisor detectors: union the given nodes' own stageIds, or fall back
|
|
204
202
|
// to the whole execution's stage set when none have coverage (never a partial blend). `fallback`
|
|
205
203
|
// is a param so callers compute stageIdsForSqlExec once per detect() and reuse it per finding.
|
|
@@ -548,15 +546,19 @@ function maxMedianRatio(
|
|
|
548
546
|
}
|
|
549
547
|
|
|
550
548
|
|
|
551
|
-
|
|
549
|
+
|
|
552
550
|
|
|
553
|
-
|
|
554
|
-
|
|
551
|
+
|
|
552
|
+
|
|
553
|
+
|
|
555
554
|
|
|
555
|
+
|
|
556
|
+
|
|
556
557
|
|
|
557
558
|
|
|
558
559
|
// Machine-readable detector metadata for the evidence report (no `detect` closure), so a
|
|
559
|
-
// portable report records which detector + thresholds produced each finding.
|
|
560
|
+
// portable report records which detector + thresholds produced each finding. These are the
|
|
561
|
+
// defaults; tunedDetectorCatalog() (threshold-overrides.ts) is the same rows under overrides.
|
|
560
562
|
export function detectorCatalog() {
|
|
561
563
|
return DETECTORS.map((d) => ({
|
|
562
564
|
type: d.type,
|
|
@@ -585,49 +587,74 @@ export function computeSkewRatio(
|
|
|
585
587
|
|
|
586
588
|
// Absolute-magnitude floor as a % of app runtime, not a fixed ms constant: a skew/straggler
|
|
587
589
|
// ratio on a few ms is noise in an hours-long run but real in a seconds-long one; a fixed-ms
|
|
588
|
-
// floor can't scale. Used by skew/straggler, gated
|
|
589
|
-
//
|
|
590
|
+
// floor can't scale. Used by skew/straggler, gated on the same occupancy-clipped tail claim their
|
|
591
|
+
// estimate() displays as savings (tailClaimImpact).
|
|
590
592
|
// NOT SOURCED: floor percentages are our own noise floor, unvalidated.
|
|
591
|
-
function computeAppDurationMs(ctx ) {
|
|
592
|
-
const app = ctx?.app;
|
|
593
|
-
if (app?.startTime == null || app?.endTime == null) return null;
|
|
594
|
-
const durationMs = app.endTime - app.startTime;
|
|
595
|
-
return durationMs > 0 ? durationMs : null;
|
|
596
|
-
}
|
|
597
|
-
|
|
598
593
|
// Unknown app timing never suppresses a finding; it just skips the floor gate.
|
|
599
|
-
function meetsRuntimeFloor(wasteMs ,
|
|
600
|
-
return
|
|
594
|
+
function meetsRuntimeFloor(wasteMs , runMs , floorPct ) {
|
|
595
|
+
return runMs == null || wasteMs >= runMs * floorPct;
|
|
601
596
|
}
|
|
602
597
|
|
|
603
598
|
// A stage that ran for less than floorPct of the run (a known duration): an estimate clipped to
|
|
604
599
|
// the stage can't reach floorPct, so every finding there grades info. A zero-length stage (no
|
|
605
600
|
// submission time on older Spark) is not skipped: it gets no estimate and keeps its own band.
|
|
606
|
-
function stageBelowRuntimeFloor(stage , ctx
|
|
601
|
+
function stageBelowRuntimeFloor(stage , ctx , floorPct ) {
|
|
607
602
|
const stageDurationMs = (stage.completedAt ?? 0) - (stage.submittedAt ?? 0);
|
|
608
|
-
const
|
|
609
|
-
return
|
|
603
|
+
const runMs = appDurationMs(ctx.app);
|
|
604
|
+
return runMs != null && stageDurationMs > 0 && stageDurationMs < runMs * floorPct;
|
|
610
605
|
}
|
|
611
606
|
|
|
612
|
-
//
|
|
613
|
-
//
|
|
614
|
-
//
|
|
615
|
-
//
|
|
616
|
-
|
|
617
|
-
|
|
618
|
-
|
|
619
|
-
|
|
620
|
-
|
|
621
|
-
|
|
622
|
-
|
|
623
|
-
|
|
624
|
-
return
|
|
607
|
+
// What a skew or straggler fix claims off its stage: the wall-clock its slow tail costs, the task
|
|
608
|
+
// time the fix removes, and the longest task it leaves. One figure, read twice: detect() gates its
|
|
609
|
+
// runtime floor on the claim's clipped estimate and estimate() reports that same estimate as the
|
|
610
|
+
// savings, so the firing floor and the displayed figure can't disagree.
|
|
611
|
+
|
|
612
|
+
|
|
613
|
+
|
|
614
|
+
|
|
615
|
+
|
|
616
|
+
|
|
617
|
+
// singleDelta: the slowest task's own excess, the fallback when the stage has no task replay.
|
|
618
|
+
function tailClaim(stage , singleDelta , longestTaskAfterFixMs ) {
|
|
619
|
+
return {
|
|
620
|
+
wasteMs: tailRecoveryMs(stage, singleDelta),
|
|
621
|
+
removedCoreWorkMs: tailRemovedWorkMs(stage, singleDelta),
|
|
622
|
+
longestTaskAfterFixMs,
|
|
623
|
+
};
|
|
624
|
+
}
|
|
625
|
+
|
|
626
|
+
// skew's delta is the task computeSkewRatio's metric sampled (P95 or max) over the median. Fixing
|
|
627
|
+
// the skew still waits on the longest task it leaves, as for straggler.
|
|
628
|
+
function skewTailClaim(stage , usesP95Branch ) {
|
|
629
|
+
const p50 = stage.taskDurationP50 ?? 0;
|
|
630
|
+
const singleDelta = Math.max(0, usesP95Branch ? (stage.taskDurationP95 ?? 0) - p50 : (stage.taskDurationMax ?? 0) - p50);
|
|
631
|
+
return tailClaim(stage, singleDelta, stragglerFixLongestTaskMs(stage));
|
|
632
|
+
}
|
|
633
|
+
|
|
634
|
+
// straggler's delta is the slowest task over the longest one the fix leaves.
|
|
635
|
+
function stragglerTailClaim(stage ) {
|
|
636
|
+
const longestTaskAfterFixMs = stragglerFixLongestTaskMs(stage);
|
|
637
|
+
return tailClaim(stage, Math.max(0, (stage.taskDurationMax ?? 0) - longestTaskAfterFixMs), longestTaskAfterFixMs);
|
|
638
|
+
}
|
|
639
|
+
|
|
640
|
+
// A tail claim shortens the stage's longest task, hence TAIL_CLAIM (see occupancy.ts).
|
|
641
|
+
function tailClaimImpact(claim , stageId , ctx ) {
|
|
642
|
+
return singleStageImpact(claim.wasteMs, stageId, ctx, 'measured', { value: claim.wasteMs, unit: 'ms' },
|
|
643
|
+
{ ...TAIL_CLAIM, removedCoreWorkMs: claim.removedCoreWorkMs, longestTaskAfterFixMs: claim.longestTaskAfterFixMs });
|
|
625
644
|
}
|
|
626
645
|
|
|
627
|
-
//
|
|
628
|
-
//
|
|
629
|
-
|
|
630
|
-
|
|
646
|
+
// The figure a runtime floor checks: the claim's recoverable wall-clock, not a delta a physical
|
|
647
|
+
// floor leaves unrecoverable. Falls back to the raw claim when occupancy data is unavailable.
|
|
648
|
+
function tailClaimFloorMs(claim , stageId , ctx ) {
|
|
649
|
+
return tailClaimImpact(claim, stageId, ctx.impact).wallClock?.high ?? claim.wasteMs;
|
|
650
|
+
}
|
|
651
|
+
|
|
652
|
+
// Shared by cacheUtilization's two variants, worded per storage source: neither is a runtime
|
|
653
|
+
// block-access read-count.
|
|
654
|
+
const CACHE_UTILIZATION_VALIDATION = {
|
|
655
|
+
rddInfo: "This ratio is a point-in-time storage snapshot from stage-submission events, not a runtime read-count. Confirm against the Spark UI's Storage tab before acting.",
|
|
656
|
+
blockUpdates: "This ratio is the RDD's peak cache residency rebuilt from block-update events, not a runtime read-count. Confirm against the Spark UI's Storage tab before acting.",
|
|
657
|
+
} ;
|
|
631
658
|
|
|
632
659
|
// Both cachedRatio and diskRatio are percentages of an RDD's partition/byte count: with few
|
|
633
660
|
// partitions, one partition flipping cached/evicted (or disk-resident/memory-resident) swings the
|
|
@@ -651,7 +678,7 @@ function partialCacheFinding(rdd , cachedRatio , impactBa
|
|
|
651
678
|
rddId: rdd.id, rddName,
|
|
652
679
|
impactBand, metric: 'cachedRatio', value: cachedPct,
|
|
653
680
|
confidence: cacheSampleConfidence(rdd.numPartitions),
|
|
654
|
-
validationRequired: CACHE_UTILIZATION_VALIDATION,
|
|
681
|
+
validationRequired: CACHE_UTILIZATION_VALIDATION[rdd.storageSource ?? 'rddInfo'],
|
|
655
682
|
memorySize: rdd.memorySize, diskSize: rdd.diskSize,
|
|
656
683
|
numCachedPartitions: rdd.numCachedPartitions, numPartitions: rdd.numPartitions,
|
|
657
684
|
recommendation: `RDD ${rddName} is ${evictedPct}% evicted from cache (${cachedPct}% of partitions cached). Increase executor memory or reduce the cached dataset size.`,
|
|
@@ -666,35 +693,164 @@ function diskSpilloverFinding(rdd , diskRatio , impactBan
|
|
|
666
693
|
rddId: rdd.id, rddName,
|
|
667
694
|
impactBand, metric: 'diskRatio', value: diskPct,
|
|
668
695
|
confidence: cacheSampleConfidence(rdd.numPartitions),
|
|
669
|
-
validationRequired: CACHE_UTILIZATION_VALIDATION,
|
|
696
|
+
validationRequired: CACHE_UTILIZATION_VALIDATION[rdd.storageSource ?? 'rddInfo'],
|
|
670
697
|
memorySize: rdd.memorySize, diskSize: rdd.diskSize,
|
|
671
698
|
numCachedPartitions: rdd.numCachedPartitions, numPartitions: rdd.numPartitions,
|
|
672
699
|
recommendation: `RDD ${rddName} is ${diskPct}% spilled to disk despite requesting MEMORY_AND_DISK. Executor memory may be too small for this cached dataset.`,
|
|
673
700
|
};
|
|
674
701
|
}
|
|
675
702
|
|
|
676
|
-
//
|
|
677
|
-
//
|
|
678
|
-
//
|
|
679
|
-
|
|
680
|
-
|
|
681
|
-
|
|
682
|
-
|
|
683
|
-
|
|
703
|
+
// Persisted RDDs with no storage evidence at all: no block updates in the log, and RDD Info's
|
|
704
|
+
// sizes are the 0 that Spark 2.3+ always writes. Reports the gap instead of a clean result, the
|
|
705
|
+
// same missing-evidence shape as memoryUtilization's dataUnavailable caveat.
|
|
706
|
+
function storageUnobservedFinding(persistedRddCount ) {
|
|
707
|
+
const rdds = persistedRddCount === 1 ? '1 persisted RDD has' : `${persistedRddCount} persisted RDDs have`;
|
|
708
|
+
return {
|
|
709
|
+
type: 'cacheUtilization', variant: 'storageUnobserved', stageId: null,
|
|
710
|
+
impactBand: 'info', metric: 'persistedRdds', value: persistedRddCount, dataUnavailable: true,
|
|
711
|
+
recommendation: `${rdds} no cache-storage evidence in this log, so eviction and disk spillover can't be checked: Spark 2.3+ records cached sizes only as block updates, which need spark.eventLog.logBlockUpdates.enabled=true.`,
|
|
712
|
+
};
|
|
713
|
+
}
|
|
714
|
+
|
|
715
|
+
/** A detector entry's threshold set. number[] too: slowHost's ratioTiers and broadcastSizing's
|
|
716
|
+
* tiers are tier tables its detect() indexes by position. */
|
|
717
|
+
|
|
718
|
+
|
|
719
|
+
|
|
720
|
+
|
|
721
|
+
// What each scope's detect() is handed. `thresholds` is the entry's own set, or the caller's
|
|
722
|
+
// overrides merged over it (analyze()'s `thresholds` option). 'config' has no DetectorCtx:
|
|
723
|
+
// auditConfig() runs those entries on the app alone.
|
|
724
|
+
|
|
725
|
+
|
|
726
|
+
|
|
727
|
+
|
|
728
|
+
|
|
729
|
+
|
|
730
|
+
|
|
731
|
+
|
|
732
|
+
|
|
733
|
+
// The same calls with the thresholds already bound: what a runner holds after withThresholds().
|
|
734
|
+
|
|
735
|
+
|
|
736
|
+
|
|
737
|
+
|
|
738
|
+
|
|
739
|
+
|
|
740
|
+
|
|
741
|
+
// The fields a define*Detector() call spells out. `detect` is a property, not a method, so its
|
|
742
|
+
// parameters are checked contravariantly: a detect() that reads a threshold the entry doesn't
|
|
743
|
+
// declare, or expects a different target, fails to compile.
|
|
744
|
+
|
|
745
|
+
|
|
746
|
+
|
|
747
|
+
|
|
748
|
+
|
|
749
|
+
|
|
750
|
+
|
|
751
|
+
|
|
752
|
+
|
|
753
|
+
|
|
684
754
|
|
|
685
755
|
|
|
686
756
|
|
|
687
757
|
|
|
688
|
-
|
|
689
|
-
|
|
690
|
-
|
|
691
|
-
|
|
692
|
-
|
|
758
|
+
|
|
693
759
|
|
|
694
760
|
|
|
695
|
-
|
|
761
|
+
|
|
696
762
|
|
|
763
|
+
|
|
764
|
+
|
|
765
|
+
|
|
766
|
+
|
|
767
|
+
|
|
768
|
+
|
|
697
769
|
|
|
770
|
+
|
|
771
|
+
|
|
772
|
+
|
|
773
|
+
|
|
774
|
+
|
|
775
|
+
|
|
776
|
+
|
|
777
|
+
|
|
778
|
+
|
|
779
|
+
|
|
780
|
+
|
|
781
|
+
|
|
782
|
+
/** How runners (analyze(), auditConfig(), the catalog helpers) see any DETECTORS entry: its
|
|
783
|
+
* thresholds type erased and detect() left off, so they reach it only through withThresholds().
|
|
784
|
+
* estimate() takes any Finding here: estimateImpact() only hands an entry the types it emits. */
|
|
785
|
+
|
|
786
|
+
|
|
787
|
+
|
|
788
|
+
|
|
789
|
+
|
|
790
|
+
|
|
791
|
+
// Same-shape overrides merged over the defaults. analyze()'s callers validate overrides at their
|
|
792
|
+
// own boundary (threshold-overrides.ts); this re-checks so a programmatic caller can't hand a
|
|
793
|
+
// detector a threshold of the wrong shape, which is what makes the cast below sound.
|
|
794
|
+
function mergeThresholds (type , defaults , overrides ) {
|
|
795
|
+
for (const [name, value] of Object.entries(overrides)) {
|
|
796
|
+
// Own keys only: an inherited name such as `constructor` or `toString` is no threshold.
|
|
797
|
+
if (!Object.hasOwn(defaults, name)) throw new Error(`Detector ${type} has no threshold "${name}".`);
|
|
798
|
+
const fallback = defaults[name];
|
|
799
|
+
const sameShape = Array.isArray(fallback)
|
|
800
|
+
? Array.isArray(value) && value.length === fallback.length
|
|
801
|
+
: typeof value === 'number';
|
|
802
|
+
if (!sameShape) throw new Error(`Threshold ${type}.${name} must have the same shape as its default.`);
|
|
803
|
+
}
|
|
804
|
+
return Object.freeze({ ...defaults, ...overrides }) ;
|
|
805
|
+
}
|
|
806
|
+
|
|
807
|
+
function defineDetector
|
|
808
|
+
|
|
809
|
+
|
|
810
|
+
(
|
|
811
|
+
scope , spec ,
|
|
812
|
+
bind ,
|
|
813
|
+
) {
|
|
814
|
+
// Frozen: the defaults are the specification, never a knob to mutate in place.
|
|
815
|
+
const defaults = Object.freeze({ ...spec.thresholds }) ;
|
|
816
|
+
const boundDefaults = bind(defaults);
|
|
817
|
+
return {
|
|
818
|
+
...spec, scope, thresholds: defaults,
|
|
819
|
+
withThresholds: (overrides) => (overrides && Object.keys(overrides).length > 0
|
|
820
|
+
? bind(mergeThresholds(spec.type, defaults, overrides))
|
|
821
|
+
: boundDefaults),
|
|
822
|
+
};
|
|
823
|
+
}
|
|
824
|
+
|
|
825
|
+
// One helper per scope: each infers the entry's thresholds type from its `thresholds` literal
|
|
826
|
+
// and hands detect() exactly that type, with the scope's target and a required context.
|
|
827
|
+
export function defineStageDetector
|
|
828
|
+
|
|
829
|
+
|
|
830
|
+
(spec ) {
|
|
831
|
+
return defineDetector('stage', spec, (t) => (stage, ctx) => spec.detect(stage, ctx, t));
|
|
832
|
+
}
|
|
833
|
+
|
|
834
|
+
export function defineSqlDetector
|
|
835
|
+
|
|
836
|
+
|
|
837
|
+
(spec ) {
|
|
838
|
+
return defineDetector('sql', spec, (t) => (sqlExec, ctx) => spec.detect(sqlExec, ctx, t));
|
|
839
|
+
}
|
|
840
|
+
|
|
841
|
+
export function defineAppDetector
|
|
842
|
+
|
|
843
|
+
|
|
844
|
+
(spec ) {
|
|
845
|
+
return defineDetector('app', spec, (t) => (ctx) => spec.detect(ctx, t));
|
|
846
|
+
}
|
|
847
|
+
|
|
848
|
+
export function defineConfigDetector
|
|
849
|
+
|
|
850
|
+
|
|
851
|
+
(spec ) {
|
|
852
|
+
return defineDetector('config', spec, (t) => (target) => spec.detect(target, t));
|
|
853
|
+
}
|
|
698
854
|
|
|
699
855
|
// Threshold field naming convention:
|
|
700
856
|
// *Pct = 0–1 fraction (normalized)
|
|
@@ -702,11 +858,6 @@ function diskSpilloverFinding(rdd , diskRatio , impactBan
|
|
|
702
858
|
// *Ratio = multiplicative factor
|
|
703
859
|
// *Share/*Rate/*Util = 0–1 fraction (normalized)
|
|
704
860
|
|
|
705
|
-
// straggler's noise-floor thresholds (NOT SOURCED: unvalidated), exported so impact-band.ts
|
|
706
|
-
// reuses the same figures instead of hand-copying.
|
|
707
|
-
export const STRAGGLER_FLOOR_PCT_WARN = 0.005;
|
|
708
|
-
export const STRAGGLER_FLOOR_PCT_CRIT = 0.02;
|
|
709
|
-
|
|
710
861
|
// A ratio just past ratioWarn is the case most likely to be ordinary task-duration variance
|
|
711
862
|
// rather than real skew; a ratio many multiples past it (a 50x P95/median vs. a 3.1x one) is
|
|
712
863
|
// unambiguous. 1.5x/5x mirror this file's other "just past the floor vs. clearly past it" splits
|
|
@@ -804,62 +955,71 @@ function cachingReuseConfidence(occurrences , minExecutions )
|
|
|
804
955
|
return 'medium';
|
|
805
956
|
}
|
|
806
957
|
|
|
807
|
-
|
|
808
|
-
|
|
809
|
-
|
|
958
|
+
// A share threshold as caveat text states it: 0.005 -> "0.5%". Rounded to 4 decimals of a percent
|
|
959
|
+
// so float noise (0.07 * 100) never prints.
|
|
960
|
+
function shareLabel(share ) {
|
|
961
|
+
return `${Math.round(share * 1e6) / 1e4}%`;
|
|
962
|
+
}
|
|
963
|
+
|
|
964
|
+
// Caveats that name a threshold read it from the thresholds the detector ran with, so a tuned run
|
|
965
|
+
// states the floor it actually used.
|
|
966
|
+
function gcValidation(minRunTimeMs ) {
|
|
967
|
+
return `This finding is gated by a ${minRunTimeMs / 1000}-second minimum-runtime floor, our own noise floor for this metric.`;
|
|
968
|
+
}
|
|
969
|
+
|
|
970
|
+
const INCOMPLETE_RUN_RECOMMENDATION = 'This event log never recorded an ApplicationEnd event: the capture stopped before the run finished (an in-flight job, a rotated-away log, or a cut-short capture). Findings and metrics elsewhere on this board reflect only what was captured up to that point, not the full run.';
|
|
971
|
+
|
|
972
|
+
export const DETECTORS = [
|
|
973
|
+
defineStageDetector({
|
|
974
|
+
type: 'skew', order: 30, fixEffort: 'code', version: 1,
|
|
975
|
+
emits: ['skew'],
|
|
810
976
|
docAnchor: '#bottleneck-skew',
|
|
811
|
-
thresholds: { ratioWarn: 3, minTasksForP95: 20, floorPctWarn:
|
|
812
|
-
detect(
|
|
813
|
-
|
|
814
|
-
|
|
815
|
-
|
|
816
|
-
stage ,
|
|
817
|
-
ctx ,
|
|
818
|
-
) {
|
|
819
|
-
const result = computeSkewRatio(stage, this.thresholds.minTasksForP95);
|
|
977
|
+
thresholds: { ratioWarn: 3, minTasksForP95: 20, floorPctWarn: IMPACT_FLOOR_PCT_WARN },
|
|
978
|
+
detect(stage, ctx, thresholds) {
|
|
979
|
+
const result = computeSkewRatio(stage, thresholds.minTasksForP95);
|
|
820
980
|
if (result === null) return null;
|
|
821
981
|
const { ratio, metric } = result;
|
|
822
|
-
if (ratio <=
|
|
823
|
-
//
|
|
824
|
-
|
|
825
|
-
|
|
826
|
-
const wasteMs = tailRecoveryMs(stage, singleDelta);
|
|
827
|
-
const appDurationMs = computeAppDurationMs(ctx);
|
|
828
|
-
const floorWasteMs = clippedWasteMs(wasteMs, stage.id, ctx, tailRemovedWorkMs(stage, singleDelta), stragglerFixLongestTaskMs(stage));
|
|
829
|
-
if (!meetsRuntimeFloor(floorWasteMs, appDurationMs, this.thresholds.floorPctWarn)) return null;
|
|
982
|
+
if (ratio <= thresholds.ratioWarn) return null;
|
|
983
|
+
// The claim estimate() reports as savings, clipped the same way, so the gate agrees with it.
|
|
984
|
+
const floorWasteMs = tailClaimFloorMs(skewTailClaim(stage, metric === 'P95/median'), stage.id, ctx);
|
|
985
|
+
if (!meetsRuntimeFloor(floorWasteMs, appDurationMs(ctx.app), thresholds.floorPctWarn)) return null;
|
|
830
986
|
const value = Math.round(ratio * 10) / 10;
|
|
831
987
|
return {
|
|
832
988
|
type: 'skew', stageId: stage.id,
|
|
833
989
|
impactBand: 'warning',
|
|
834
990
|
metric, value,
|
|
835
|
-
confidence: skewConfidence(ratio,
|
|
836
|
-
validationRequired:
|
|
991
|
+
confidence: skewConfidence(ratio, thresholds.ratioWarn),
|
|
992
|
+
validationRequired: `This finding is gated by a ${shareLabel(thresholds.floorPctWarn)} runtime-floor threshold, our own noise floor for this metric.`,
|
|
837
993
|
recommendation: `Task duration ratio (${metric}) is ${value}×: for join-driven skew, enable AQE skew-join handling (spark.sql.adaptive.skewJoin.enabled); otherwise salt the key or repartition on a better key to reduce task skew.`,
|
|
838
994
|
};
|
|
839
995
|
},
|
|
840
|
-
|
|
841
|
-
|
|
842
|
-
|
|
996
|
+
estimate(finding, ctx) {
|
|
997
|
+
if (finding.stageId == null) return null;
|
|
998
|
+
const stage = ctx.stages.get(finding.stageId);
|
|
999
|
+
if (!stage) return null;
|
|
1000
|
+
// computeSkewRatio's own metric labels: 'P95/median' or 'max/median'.
|
|
1001
|
+
return tailClaimImpact(skewTailClaim(stage, finding.metric === 'P95/median'), finding.stageId, ctx);
|
|
1002
|
+
},
|
|
1003
|
+
}),
|
|
1004
|
+
defineStageDetector({
|
|
1005
|
+
type: 'stageShape', order: 35, fixEffort: 'code', version: 1,
|
|
1006
|
+
emits: ['stageShape'],
|
|
843
1007
|
docAnchor: '#bottleneck-stage-shape',
|
|
844
1008
|
// lowParallelismFloorPct: the same 0.5% runtime floor the tiered detectors use. Parallelizing a
|
|
845
1009
|
// stage can't save more than the stage's own duration, so a shorter stage can't clear it; on
|
|
846
1010
|
// the 14 real logs that was 2839 of 3005 lowParallelism findings (2168 on sub-second stages).
|
|
847
1011
|
// App-wide idle capacity stays covered by utilization.
|
|
848
1012
|
thresholds: { pRatioMax: 0.5, oiRatioMax: 10, skewWarn: 3, lowParallelismFloorPct: 0.005 },
|
|
849
|
-
detect(
|
|
850
|
-
|
|
851
|
-
stage ,
|
|
852
|
-
ctx ,
|
|
853
|
-
) {
|
|
1013
|
+
detect(stage, ctx, thresholds) {
|
|
854
1014
|
const out = [];
|
|
855
1015
|
const execCount = (stage.executorStats ?? []).length;
|
|
856
|
-
const cores = ctx
|
|
1016
|
+
const cores = ctx.app?.resources?.executor?.cores ?? 1;
|
|
857
1017
|
const totalCores = execCount * cores;
|
|
858
1018
|
const stageDurationMs = (stage.completedAt ?? 0) - (stage.submittedAt ?? 0);
|
|
859
1019
|
// PRatio: under-parallelization.
|
|
860
|
-
if (totalCores > 0 && !stageBelowRuntimeFloor(stage, ctx,
|
|
1020
|
+
if (totalCores > 0 && !stageBelowRuntimeFloor(stage, ctx, thresholds.lowParallelismFloorPct)) {
|
|
861
1021
|
const pRatio = stage.taskCount / totalCores;
|
|
862
|
-
if (pRatio <
|
|
1022
|
+
if (pRatio < thresholds.pRatioMax) {
|
|
863
1023
|
out.push({
|
|
864
1024
|
type: 'stageShape', stageId: stage.id, impactBand: 'info',
|
|
865
1025
|
rule: 'lowParallelism', metric: 'pRatio', value: Math.round(pRatio * 100) / 100,
|
|
@@ -872,7 +1032,7 @@ export const DETECTORS = [
|
|
|
872
1032
|
// OIRatio: data explosion. Skip when inputBytes is 0 (Infinity guard).
|
|
873
1033
|
if (stage.inputBytes > 0) {
|
|
874
1034
|
const oiRatio = stage.outputBytes / stage.inputBytes;
|
|
875
|
-
if (oiRatio >
|
|
1035
|
+
if (oiRatio > thresholds.oiRatioMax) {
|
|
876
1036
|
out.push({
|
|
877
1037
|
type: 'stageShape', stageId: stage.id, impactBand: 'info',
|
|
878
1038
|
rule: 'dataExplosion', metric: 'oiRatio', value: Math.round(oiRatio * 10) / 10,
|
|
@@ -885,7 +1045,7 @@ export const DETECTORS = [
|
|
|
885
1045
|
// every firing, so there's no wall-clock-backed tier left to gate on.
|
|
886
1046
|
if (stageDurationMs > 0) {
|
|
887
1047
|
const ratio = stage.taskDurationMax / stageDurationMs;
|
|
888
|
-
if (ratio >
|
|
1048
|
+
if (ratio > thresholds.skewWarn) {
|
|
889
1049
|
out.push({
|
|
890
1050
|
type: 'stageShape', stageId: stage.id, impactBand: 'info',
|
|
891
1051
|
rule: 'taskStageSkew', metric: 'taskStageSkew', value: Math.round(ratio * 10) / 10,
|
|
@@ -897,22 +1057,47 @@ export const DETECTORS = [
|
|
|
897
1057
|
}
|
|
898
1058
|
return out;
|
|
899
1059
|
},
|
|
900
|
-
|
|
901
|
-
|
|
902
|
-
|
|
1060
|
+
estimate(finding, ctx) {
|
|
1061
|
+
const stage = ctx.stages.get(finding.stageId );
|
|
1062
|
+
if (!stage) return null;
|
|
1063
|
+
if (finding.rule === 'lowParallelism') {
|
|
1064
|
+
const stageDurationMs = (stage.completedAt ?? 0) - (stage.submittedAt ?? 0);
|
|
1065
|
+
const idleCoreMs =
|
|
1066
|
+
Math.max(0, ((finding.totalCores ) ?? 0) - (stage.taskCount ?? 0)) * stageDurationMs;
|
|
1067
|
+
// Real per-stage data (cores, task count, duration), no assumed constant.
|
|
1068
|
+
return costOnly('measured', { value: idleCoreMs, unit: 'coreMs' });
|
|
1069
|
+
}
|
|
1070
|
+
if (finding.rule === 'dataExplosion') {
|
|
1071
|
+
const excessBytes = Math.max(0, (stage.outputBytes ?? 0) - (stage.inputBytes ?? 0));
|
|
1072
|
+
// Measured input/output byte counts, no assumed constant.
|
|
1073
|
+
return costOnly('measured', { value: excessBytes, unit: 'bytes' });
|
|
1074
|
+
}
|
|
1075
|
+
if (finding.rule === 'taskStageSkew') {
|
|
1076
|
+
const totalCores = (finding.totalCores ) ?? 0;
|
|
1077
|
+
const taskCount = stage.taskCount ?? 0;
|
|
1078
|
+
// Cores idle during the straggler's tail, at achieved concurrency (not full cluster
|
|
1079
|
+
// capacity, which is lowParallelism's territory): this rule's trigger forces the
|
|
1080
|
+
// occupancy-clipped estimate to zero on every firing, so it's resourceOnly, not a wall-clock claim.
|
|
1081
|
+
const idleCoreMs =
|
|
1082
|
+
Math.max(0, Math.min(totalCores, taskCount) - 1) *
|
|
1083
|
+
Math.max(0, (stage.taskDurationMax ?? 0) - (stage.taskDurationP50 ?? 0));
|
|
1084
|
+
return costOnly('measured', { value: idleCoreMs, unit: 'coreMs' });
|
|
1085
|
+
}
|
|
1086
|
+
return null;
|
|
1087
|
+
},
|
|
1088
|
+
}),
|
|
1089
|
+
defineStageDetector({
|
|
1090
|
+
type: 'shuffle', order: 20, fixEffort: 'config', version: 1,
|
|
1091
|
+
emits: ['shuffle'],
|
|
903
1092
|
docAnchor: '#bottleneck-shuffle',
|
|
904
1093
|
// stageFloorPct: the 0.5% runtime floor. The shuffle claim is clipped to the stage, so on a
|
|
905
1094
|
// shorter stage it graded info: 182 of 284 shuffle findings on the 14 real logs. The shuffle
|
|
906
1095
|
// is still there on those stages; the floor is why they're dropped.
|
|
907
1096
|
thresholds: { minBytes: 50 * MB, stageFloorPct: 0.005 },
|
|
908
|
-
detect(
|
|
909
|
-
|
|
910
|
-
stage ,
|
|
911
|
-
ctx ,
|
|
912
|
-
) {
|
|
1097
|
+
detect(stage, ctx, thresholds) {
|
|
913
1098
|
const bytes = stage.shuffleReadBytes;
|
|
914
|
-
if (bytes <=
|
|
915
|
-
if (stageBelowRuntimeFloor(stage, ctx,
|
|
1099
|
+
if (bytes <= thresholds.minBytes) return null;
|
|
1100
|
+
if (stageBelowRuntimeFloor(stage, ctx, thresholds.stageFloorPct)) return null;
|
|
916
1101
|
return {
|
|
917
1102
|
type: 'shuffle', stageId: stage.id,
|
|
918
1103
|
impactBand: 'info',
|
|
@@ -920,23 +1105,30 @@ export const DETECTORS = [
|
|
|
920
1105
|
recommendation: `${formatBytes(bytes)} shuffled in this stage: consider increasing spark.sql.shuffle.partitions or adding a broadcast join.`,
|
|
921
1106
|
};
|
|
922
1107
|
},
|
|
923
|
-
|
|
924
|
-
|
|
925
|
-
|
|
1108
|
+
estimate(finding, ctx) {
|
|
1109
|
+
if (finding.stageId == null) return null;
|
|
1110
|
+
const stage = ctx.stages.get(finding.stageId);
|
|
1111
|
+
if (!stage) return null;
|
|
1112
|
+
const shuffleReadBytes = stage.shuffleReadBytes ?? 0;
|
|
1113
|
+
const modeledMs = (shuffleReadBytes / (SHUFFLE_THROUGHPUT_BPS * stageIoParallelism(stage))) * 1000;
|
|
1114
|
+
// The link model can't see whether the reads stalled the tasks: capped at the fetch wait the
|
|
1115
|
+
// tasks measured, the claim never exceeds what the stage spent blocked on the network.
|
|
1116
|
+
const measuredMs = fetchWaitWallClockMs(stage);
|
|
1117
|
+
const wasteMs = measuredMs == null ? modeledMs : Math.min(modeledMs, measuredMs);
|
|
1118
|
+
// rawWaste: the measured byte volume behind the modeled figure.
|
|
1119
|
+
return singleStageImpact(wasteMs, finding.stageId, ctx,
|
|
1120
|
+
measuredMs != null && measuredMs < modeledMs ? 'measured' : 'modeled', { value: shuffleReadBytes, unit: 'bytes' });
|
|
1121
|
+
},
|
|
1122
|
+
}),
|
|
1123
|
+
defineStageDetector({
|
|
1124
|
+
type: 'partitionSizing', order: 22, fixEffort: 'config', version: 1,
|
|
1125
|
+
emits: ['partitionSizing'],
|
|
926
1126
|
docAnchor: '#bottleneck-partition-sizing',
|
|
927
1127
|
thresholds: { skewRatio: 5, skewFloorBytes: 256 * MB, lowParTotalBytes: GB, lowParMaxTasks: 7, maxPartBytes: 5 * GB },
|
|
928
|
-
detect(
|
|
929
|
-
|
|
930
|
-
|
|
931
|
-
|
|
932
|
-
|
|
933
|
-
|
|
934
|
-
|
|
935
|
-
stage ,
|
|
936
|
-
) {
|
|
1128
|
+
detect(stage, _ctx, thresholds) {
|
|
937
1129
|
const out = [];
|
|
938
1130
|
const { shuffleReadP50: p50, shuffleReadMax: max, shuffleReadBytes: total, taskCount } = stage;
|
|
939
|
-
if (max >
|
|
1131
|
+
if (max > thresholds.skewRatio * p50 && max > thresholds.skewFloorBytes) {
|
|
940
1132
|
// p50 can be 0 (over half the shuffle partitions empty): a ratio against zero renders
|
|
941
1133
|
// "Infinity×", so fall back to median-free phrasing.
|
|
942
1134
|
const ratioText = p50 > 0
|
|
@@ -948,14 +1140,14 @@ export const DETECTORS = [
|
|
|
948
1140
|
recommendation: `The largest shuffle partition (${formatBytes(max)}) is ${ratioText}: for join skew, enable AQE skew-join handling (spark.sql.adaptive.skewJoin.enabled); otherwise salt the key or repartition on a better key.`,
|
|
949
1141
|
});
|
|
950
1142
|
}
|
|
951
|
-
if (total >=
|
|
1143
|
+
if (total >= thresholds.lowParTotalBytes && taskCount <= thresholds.lowParMaxTasks) {
|
|
952
1144
|
out.push({
|
|
953
1145
|
type: 'partitionSizing', stageId: stage.id, impactBand: 'warning',
|
|
954
1146
|
rule: 'lowShuffleParallelism', metric: 'taskCount', value: taskCount,
|
|
955
1147
|
recommendation: `${Math.round(total / GB * 10) / 10} GB of shuffle spread over only ${taskCount} tasks: raise spark.sql.shuffle.partitions so each partition is smaller.`,
|
|
956
1148
|
});
|
|
957
1149
|
}
|
|
958
|
-
if (max >=
|
|
1150
|
+
if (max >= thresholds.maxPartBytes) {
|
|
959
1151
|
// Fixed 'critical': an OOM/crash-risk safety signal, not a time-waste one. Exempted in
|
|
960
1152
|
// impact-band.ts's deriveImpactBand from the wall-clock-based overwrite every other
|
|
961
1153
|
// finding here gets, so a long-running job can't demote an active crash risk to 'info'
|
|
@@ -968,19 +1160,46 @@ export const DETECTORS = [
|
|
|
968
1160
|
}
|
|
969
1161
|
return out;
|
|
970
1162
|
},
|
|
971
|
-
|
|
972
|
-
|
|
973
|
-
|
|
1163
|
+
estimate(finding, ctx) {
|
|
1164
|
+
if (finding.stageId == null) return null;
|
|
1165
|
+
const stage = ctx.stages.get(finding.stageId);
|
|
1166
|
+
if (!stage) return null;
|
|
1167
|
+
let wasteMs = 0;
|
|
1168
|
+
if (finding.rule === 'maxPartitionTooBig') {
|
|
1169
|
+
wasteMs = ((stage.shuffleReadMax ?? 0) / SHUFFLE_THROUGHPUT_BPS) * 1000;
|
|
1170
|
+
} else if (finding.rule === 'shufflePartitionSkew') {
|
|
1171
|
+
const delta = Math.max(0, (stage.shuffleReadMax ?? 0) - (stage.shuffleReadP50 ?? 0));
|
|
1172
|
+
wasteMs = (delta / SHUFFLE_THROUGHPUT_BPS) * 1000;
|
|
1173
|
+
} else if (finding.rule === 'lowShuffleParallelism') {
|
|
1174
|
+
const targetTaskCount = Math.ceil((stage.shuffleReadBytes ?? 0) / IDEAL_BYTES_PER_PARTITION_TASK);
|
|
1175
|
+
const taskCount = stage.taskCount ?? 0;
|
|
1176
|
+
if (targetTaskCount > taskCount && taskCount > 0) {
|
|
1177
|
+
const stageDurationMs = Math.max(0, (stage.completedAt ?? 0) - (stage.submittedAt ?? 0));
|
|
1178
|
+
// Too few shuffle partitions means each task processes more than the ideal bytes,
|
|
1179
|
+
// serializing work more partitions would run concurrently: the waste is that serialized
|
|
1180
|
+
// work, not the scheduling cost of tasks you'd add (adding tasks incurs overhead, recovers
|
|
1181
|
+
// nothing). Model the achievable duration at target parallelism by scaling down proportionally.
|
|
1182
|
+
wasteMs = stageDurationMs * (1 - taskCount / targetTaskCount);
|
|
1183
|
+
}
|
|
1184
|
+
} else {
|
|
1185
|
+
return null;
|
|
1186
|
+
}
|
|
1187
|
+
return singleStageImpact(wasteMs, finding.stageId, ctx, 'modeled', { value: wasteMs, unit: 'ms' });
|
|
1188
|
+
},
|
|
1189
|
+
}),
|
|
1190
|
+
defineStageDetector({
|
|
1191
|
+
type: 'spill', order: 10, fixEffort: 'code', version: 1,
|
|
1192
|
+
emits: ['spill'],
|
|
974
1193
|
docAnchor: '#bottleneck-spill',
|
|
975
1194
|
// stageFloorPct: the 0.5% runtime floor, as for shuffle (12 of 38 spill findings on the 14 real
|
|
976
1195
|
// logs, all info). The spill is still there on those stages; the floor is why they're dropped.
|
|
977
1196
|
thresholds: { singleTaskDiskGiB: 1, singleTaskMemGiB: 4, highDiskGiB: 1, highTaskDiskMB: 512, highMemGiB: 4, medDiskMB: 256, medMemGiB: 1, skewRatio: 5, skewDiskFloorMB: 128, skewMemFloorMB: 256, skewMinTasks: 10, stageFloorPct: 0.005 },
|
|
978
|
-
detect(
|
|
1197
|
+
detect(stage, ctx, thresholds) {
|
|
979
1198
|
if (stage.memoryBytesSpilled === 0) return null;
|
|
980
|
-
if (stageBelowRuntimeFloor(stage, ctx,
|
|
1199
|
+
if (stageBelowRuntimeFloor(stage, ctx, thresholds.stageFloorPct)) return null;
|
|
981
1200
|
const cls = stage.spillClassification;
|
|
982
1201
|
const classified = cls === 'skew' || cls === 'volume';
|
|
983
|
-
const mag = computeSpillMagnitude(stage,
|
|
1202
|
+
const mag = computeSpillMagnitude(stage, thresholds);
|
|
984
1203
|
const impactBand = 'warning';
|
|
985
1204
|
return {
|
|
986
1205
|
type: 'spill', stageId: stage.id, impactBand,
|
|
@@ -995,11 +1214,20 @@ export const DETECTORS = [
|
|
|
995
1214
|
: `${formatBytes(stage.memoryBytesSpilled)} spilled: raise spark.sql.shuffle.partitions or increase executor memory.`,
|
|
996
1215
|
};
|
|
997
1216
|
},
|
|
998
|
-
|
|
999
|
-
|
|
1000
|
-
|
|
1217
|
+
estimate(finding, ctx) {
|
|
1218
|
+
if (finding.stageId == null) return null;
|
|
1219
|
+
const stage = ctx.stages.get(finding.stageId);
|
|
1220
|
+
if (!stage) return null;
|
|
1221
|
+
const diskBytesSpilled = stage.diskBytesSpilled ?? 0;
|
|
1222
|
+
const wasteMs = (diskBytesSpilled / (SPILL_IO_THROUGHPUT_BPS * stageIoParallelism(stage))) * 1000;
|
|
1223
|
+
// Surfaces the number the formula uses: the displayed metric is memoryBytesSpilled, but disk spill costs the I/O time.
|
|
1224
|
+
return singleStageImpact(wasteMs, finding.stageId, ctx, 'modeled', { value: diskBytesSpilled, unit: 'bytes' });
|
|
1225
|
+
},
|
|
1226
|
+
}),
|
|
1227
|
+
defineStageDetector({
|
|
1228
|
+
type: 'gc', order: 50, fixEffort: 'config', version: 1,
|
|
1229
|
+
emits: ['gc'],
|
|
1001
1230
|
docAnchor: '#bottleneck-gc',
|
|
1002
|
-
validationRequired: 'This finding is gated by a 10-second minimum-runtime floor, our own noise floor for this metric.',
|
|
1003
1231
|
thresholds: {
|
|
1004
1232
|
warnPct100: 10,
|
|
1005
1233
|
// Descending tier: ExecutorGcHeuristic, ported as-is.
|
|
@@ -1012,44 +1240,60 @@ export const DETECTORS = [
|
|
|
1012
1240
|
// findings); the low-GC pattern is still true on those stages, the floor is why they're dropped.
|
|
1013
1241
|
lowInfoFloorPct: 0.005,
|
|
1014
1242
|
},
|
|
1015
|
-
detect(
|
|
1016
|
-
|
|
1017
|
-
|
|
1018
|
-
|
|
1019
|
-
|
|
1020
|
-
stage ,
|
|
1021
|
-
ctx ,
|
|
1022
|
-
) {
|
|
1243
|
+
detect(stage, ctx, thresholds) {
|
|
1023
1244
|
const pct = stage.gcPct;
|
|
1024
|
-
if ((stage.executorRunTime ?? 0) >=
|
|
1025
|
-
&& pct >
|
|
1245
|
+
if ((stage.executorRunTime ?? 0) >= thresholds.minRunTimeMs
|
|
1246
|
+
&& pct > thresholds.warnPct100) {
|
|
1026
1247
|
const value = Math.round(pct * 10) / 10;
|
|
1027
1248
|
return {
|
|
1028
1249
|
type: 'gc', stageId: stage.id,
|
|
1029
1250
|
impactBand: 'warning',
|
|
1030
1251
|
metric: 'gcPct', value,
|
|
1031
|
-
confidence: gcConfidence(pct,
|
|
1252
|
+
confidence: gcConfidence(pct, thresholds, 'high'), validationRequired: gcValidation(thresholds.minRunTimeMs),
|
|
1032
1253
|
recommendation: `GC consumed ${value}% of executor run time: reduce object creation, use primitive types, avoid UDFs, increase executor memory.`,
|
|
1033
1254
|
};
|
|
1034
1255
|
}
|
|
1035
1256
|
// Low-GC (cost) branch: only for stages that ran long enough to be meaningful.
|
|
1036
|
-
if ((stage.executorRunTime ?? 0) >=
|
|
1037
|
-
&& pct <
|
|
1038
|
-
&& !stageBelowRuntimeFloor(stage, ctx,
|
|
1257
|
+
if ((stage.executorRunTime ?? 0) >= thresholds.minRunTimeMs
|
|
1258
|
+
&& pct < thresholds.lowInfoPct100
|
|
1259
|
+
&& !stageBelowRuntimeFloor(stage, ctx, thresholds.lowInfoFloorPct)) {
|
|
1039
1260
|
const value = Math.round(pct * 10) / 10;
|
|
1040
1261
|
return {
|
|
1041
1262
|
type: 'gc', stageId: stage.id, direction: 'low',
|
|
1042
1263
|
impactBand: 'info',
|
|
1043
1264
|
metric: 'gcPct', value,
|
|
1044
|
-
confidence: gcConfidence(pct,
|
|
1265
|
+
confidence: gcConfidence(pct, thresholds, 'low'), validationRequired: gcValidation(thresholds.minRunTimeMs),
|
|
1045
1266
|
recommendation: `GC consumed only ${value}% of executor run time: memory may be over-provisioned; consider reducing spark.executor.memory for cost savings.`,
|
|
1046
1267
|
};
|
|
1047
1268
|
}
|
|
1048
1269
|
return null;
|
|
1049
1270
|
},
|
|
1050
|
-
|
|
1051
|
-
|
|
1052
|
-
|
|
1271
|
+
estimate(finding, ctx) {
|
|
1272
|
+
// The low-GC branch is an over-provisioning signal whose fix (less executor memory) raises
|
|
1273
|
+
// GC rather than recovering it: the stage's GC time is no saving there, so no waste model.
|
|
1274
|
+
if (finding.direction === 'low') return costOnly('none');
|
|
1275
|
+
if (finding.stageId == null) return null;
|
|
1276
|
+
const stage = ctx.stages.get(finding.stageId);
|
|
1277
|
+
if (!stage) return null;
|
|
1278
|
+
const stageDurationMs = (stage.completedAt ?? 0) - (stage.submittedAt ?? 0);
|
|
1279
|
+
const executorRunTime = stage.executorRunTime ?? 0;
|
|
1280
|
+
const jvmGCTime = stage.jvmGCTime ?? 0;
|
|
1281
|
+
// The raw cross-task core-time sum, before any conversion: the one figure here
|
|
1282
|
+
// that is straight from the log rather than modeled.
|
|
1283
|
+
const rawWaste = { value: jvmGCTime, unit: 'coreMs' } ;
|
|
1284
|
+
if (executorRunTime <= 0 || stageDurationMs <= 0) {
|
|
1285
|
+
return costOnly('modeled', rawWaste);
|
|
1286
|
+
}
|
|
1287
|
+
const avgConcurrency = executorRunTime / stageDurationMs;
|
|
1288
|
+
// jvmGCTime is a cross-task core-time sum (same shape as executorRunTime); dividing by the
|
|
1289
|
+
// stage's average concurrency converts it to an approximate wall-clock figure. Modeled, not exact.
|
|
1290
|
+
const wasteMs = jvmGCTime / avgConcurrency;
|
|
1291
|
+
return singleStageImpact(wasteMs, finding.stageId, ctx, 'modeled', rawWaste);
|
|
1292
|
+
},
|
|
1293
|
+
}),
|
|
1294
|
+
defineStageDetector({
|
|
1295
|
+
type: 'slowHost', order: 60, fixEffort: 'config', version: 1,
|
|
1296
|
+
emits: ['slowHost'],
|
|
1053
1297
|
docAnchor: '#bottleneck-slow-host',
|
|
1054
1298
|
thresholds: {
|
|
1055
1299
|
minHosts: 3, minTasks: 15, ratioWarn: 2.0, minShare: 0.20, shareWarn: 0.75, taskShareWarn: 0.50, ratioTiers: [1.33, 1.78, 3.16, 10],
|
|
@@ -1065,23 +1309,13 @@ export const DETECTORS = [
|
|
|
1065
1309
|
// floor is why they're dropped.
|
|
1066
1310
|
stageFloorPct: 0.005,
|
|
1067
1311
|
},
|
|
1068
|
-
detect(
|
|
1069
|
-
|
|
1070
|
-
|
|
1071
|
-
|
|
1072
|
-
|
|
1073
|
-
|
|
1074
|
-
|
|
1075
|
-
|
|
1076
|
-
stage ,
|
|
1077
|
-
ctx ,
|
|
1078
|
-
) {
|
|
1312
|
+
detect(stage, ctx, thresholds) {
|
|
1079
1313
|
const hosts = stage.hostStats ?? [];
|
|
1080
1314
|
const execs0 = stage.executorStats ?? [];
|
|
1081
|
-
if ((hosts.length <
|
|
1082
|
-
if (stageBelowRuntimeFloor(stage, ctx,
|
|
1315
|
+
if ((hosts.length < thresholds.minHosts && execs0.length < thresholds.minHosts) || stage.taskCount < thresholds.minTasks) return null;
|
|
1316
|
+
if (stageBelowRuntimeFloor(stage, ctx, thresholds.stageFloorPct)) return null;
|
|
1083
1317
|
const out = [];
|
|
1084
|
-
if (hosts.length >=
|
|
1318
|
+
if (hosts.length >= thresholds.minHosts) {
|
|
1085
1319
|
const means = hosts.map(h => ({ host: h.host, taskCount: h.taskCount, mean: h.totalDuration / h.taskCount }));
|
|
1086
1320
|
const sorted = [...means].map(h => h.mean).sort((a, b) => a - b);
|
|
1087
1321
|
const overallMedian = sorted[Math.floor(sorted.length / 2)];
|
|
@@ -1089,7 +1323,7 @@ export const DETECTORS = [
|
|
|
1089
1323
|
for (const h of means) {
|
|
1090
1324
|
const ratio = h.mean / overallMedian;
|
|
1091
1325
|
const share = h.taskCount / stage.taskCount;
|
|
1092
|
-
if (ratio <
|
|
1326
|
+
if (ratio < thresholds.ratioWarn || share < thresholds.minShare || h.mean < thresholds.floorMs) continue;
|
|
1093
1327
|
out.push({
|
|
1094
1328
|
type: 'slowHost', stageId: stage.id,
|
|
1095
1329
|
impactBand: 'warning',
|
|
@@ -1106,7 +1340,7 @@ export const DETECTORS = [
|
|
|
1106
1340
|
for (const h of hosts) {
|
|
1107
1341
|
const durationShare = h.totalDuration / totalDuration;
|
|
1108
1342
|
const taskShare = h.taskCount / stage.taskCount;
|
|
1109
|
-
if (durationShare >=
|
|
1343
|
+
if (durationShare >= thresholds.shareWarn && taskShare >= thresholds.taskShareWarn) {
|
|
1110
1344
|
out.push({
|
|
1111
1345
|
type: 'slowHost', stageId: stage.id, impactBand: 'warning',
|
|
1112
1346
|
variant: 'durationShare',
|
|
@@ -1121,11 +1355,11 @@ export const DETECTORS = [
|
|
|
1121
1355
|
}
|
|
1122
1356
|
}
|
|
1123
1357
|
const execs = stage.executorStats ?? [];
|
|
1124
|
-
const tiers =
|
|
1358
|
+
const tiers = thresholds.ratioTiers;
|
|
1125
1359
|
const impactBandFor = (r ) =>
|
|
1126
1360
|
r >= tiers[3] ? 'critical' : (r >= tiers[1] ? 'warning' : (r >= tiers[0] ? 'info' : null));
|
|
1127
|
-
const floorMs =
|
|
1128
|
-
const dims
|
|
1361
|
+
const floorMs = thresholds.floorMs, floorBytes = thresholds.floorBytes;
|
|
1362
|
+
const dims = [
|
|
1129
1363
|
{ dimension: 'taskTime', floor: floorMs, samples: execs.filter(e => e.taskCount > 0).map(e => ({ key: e.executorId, value: e.totalDuration / e.taskCount })) },
|
|
1130
1364
|
{ dimension: 'inputBytes', floor: floorBytes, samples: execs.map(e => ({ key: e.executorId, value: e.inputBytes ?? 0 })) },
|
|
1131
1365
|
{ dimension: 'shuffleBytes', floor: floorBytes, samples: execs.map(e => ({ key: e.executorId, value: (e.shuffleReadBytes ?? 0) + (e.shuffleWriteBytes ?? 0) })) },
|
|
@@ -1158,26 +1392,41 @@ export const DETECTORS = [
|
|
|
1158
1392
|
}
|
|
1159
1393
|
return out;
|
|
1160
1394
|
},
|
|
1161
|
-
|
|
1162
|
-
|
|
1163
|
-
|
|
1395
|
+
estimate(finding, ctx) {
|
|
1396
|
+
// Three duration-based shapes, each carrying its absolute-ms figure under a different field
|
|
1397
|
+
// (`value` is always a ratio/share, never ms): the per-host mean branch (discriminated by
|
|
1398
|
+
// `metric`), the duration-share branch (`variant`), and the multiDim taskTime dimension. Every
|
|
1399
|
+
// byte-based multiDim dimension has no absolute figure today, so it stays informational.
|
|
1400
|
+
const absoluteMs =
|
|
1401
|
+
finding.metric === 'hostMeanRatio' || finding.variant === 'durationShare'
|
|
1402
|
+
? (finding.hostMeanMs )
|
|
1403
|
+
: finding.variant === 'multiDim' && finding.dimension === 'taskTime'
|
|
1404
|
+
? (finding.execMaxValue )
|
|
1405
|
+
: null;
|
|
1406
|
+
if (absoluteMs == null) {
|
|
1407
|
+
return costOnly('none'); // byte-based multiDim dims: no absolute figure today, no model applied
|
|
1408
|
+
}
|
|
1409
|
+
if (finding.stageId == null) return null;
|
|
1410
|
+
const stage = ctx.stages.get(finding.stageId);
|
|
1411
|
+
if (!stage) return null;
|
|
1412
|
+
const wasteMs = Math.max(0, absoluteMs - (stage.taskDurationP50 ?? 0));
|
|
1413
|
+
return singleStageImpact(wasteMs, finding.stageId, ctx, 'measured', { value: wasteMs, unit: 'ms' });
|
|
1414
|
+
},
|
|
1415
|
+
}),
|
|
1416
|
+
defineStageDetector({
|
|
1417
|
+
type: 'stageSlowness', order: 65, fixEffort: 'code', version: 2,
|
|
1418
|
+
emits: ['stageSlowness'],
|
|
1164
1419
|
docAnchor: '#bottleneck-stage-slowness',
|
|
1165
1420
|
thresholds: { infoMin: 15 },
|
|
1166
|
-
//
|
|
1167
|
-
|
|
1168
|
-
|
|
1169
|
-
return out.some(o => o.type === 'slowHost' && o.stageId === finding.stageId);
|
|
1170
|
-
},
|
|
1171
|
-
detect(
|
|
1172
|
-
|
|
1173
|
-
stage ,
|
|
1174
|
-
) {
|
|
1421
|
+
// A stage slowHost already explains needs no generic "this stage is slow" finding on top.
|
|
1422
|
+
suppressedBy: 'slowHost',
|
|
1423
|
+
detect(stage, _ctx, thresholds) {
|
|
1175
1424
|
// Basis is real wall-clock stage duration, not per-executor average; the impact-estimator
|
|
1176
1425
|
// formula reuses this exact stageDurationMs computation.
|
|
1177
1426
|
const stageDurationMs = (stage.completedAt ?? 0) - (stage.submittedAt ?? 0);
|
|
1178
1427
|
if (!(stageDurationMs > 0)) return null;
|
|
1179
1428
|
const durationMinutes = stageDurationMs / 60000;
|
|
1180
|
-
const t =
|
|
1429
|
+
const t = thresholds;
|
|
1181
1430
|
const impactBand = durationMinutes >= t.infoMin ? 'info' : null;
|
|
1182
1431
|
if (!impactBand) return null;
|
|
1183
1432
|
const value = Math.round(durationMinutes * 10) / 10;
|
|
@@ -1187,53 +1436,88 @@ export const DETECTORS = [
|
|
|
1187
1436
|
recommendation: `This stage ran ${value} minutes with no more specific cause flagged: often a partition-count problem, raise parallelism via spark.sql.shuffle.partitions or spark.default.parallelism, or check for a large per-task data volume driving heavy shuffle and spill.`,
|
|
1188
1437
|
};
|
|
1189
1438
|
},
|
|
1190
|
-
|
|
1191
|
-
|
|
1192
|
-
|
|
1439
|
+
estimate(finding, ctx) {
|
|
1440
|
+
if (finding.stageId == null) return null;
|
|
1441
|
+
const stage = ctx.stages.get(finding.stageId);
|
|
1442
|
+
if (!stage) return null;
|
|
1443
|
+
// The recommended fix is more partitions, which only helps a stage that ran fewer tasks than
|
|
1444
|
+
// the cluster has cores: the time its tasks were running could then spread over up to
|
|
1445
|
+
// totalCores (lowShuffleParallelism's shape). Time the stage sat open with no task running
|
|
1446
|
+
// is queueing no partition count recovers. Splitting partitions splits the longest task
|
|
1447
|
+
// too, hence TAIL_CLAIM's post-fix floor. Unknown cluster size: no defensible figure.
|
|
1448
|
+
if (ctx.totalCores <= 0) return costOnly('modeled');
|
|
1449
|
+
// A stage that read no input and no shuffle, its tasks idle waiting on an external system,
|
|
1450
|
+
// gains nothing from more partitions (a 1-task JDBC count stage open 27 minutes on 5s of CPU
|
|
1451
|
+
// was claimed 99% recoverable): claim 0. Stages that read bytes keep their claim.
|
|
1452
|
+
const readBytes = (stage.inputBytes ?? 0) + (stage.shuffleReadBytes ?? 0);
|
|
1453
|
+
const activeMs = typeof stage.taskActiveMs === 'number'
|
|
1454
|
+
? stage.taskActiveMs
|
|
1455
|
+
: Math.max(0, (stage.completedAt ?? 0) - (stage.submittedAt ?? 0));
|
|
1456
|
+
const taskCount = stage.taskCount ?? 0;
|
|
1457
|
+
const wasteMs = readBytes <= 0 && tasksMostlyIdle(stage) ? 0 : activeMs * Math.max(0, 1 - taskCount / ctx.totalCores);
|
|
1458
|
+
return singleStageImpact(wasteMs, finding.stageId, ctx, 'modeled', { value: wasteMs, unit: 'ms' }, TAIL_CLAIM);
|
|
1459
|
+
},
|
|
1460
|
+
}),
|
|
1461
|
+
defineStageDetector({
|
|
1462
|
+
type: 'stageFailed', order: 42, fixEffort: 'code', version: 1,
|
|
1463
|
+
emits: ['stageFailed'],
|
|
1193
1464
|
docAnchor: '#bottleneck-failures',
|
|
1194
1465
|
thresholds: {},
|
|
1195
|
-
detect(stage
|
|
1466
|
+
detect(stage) {
|
|
1196
1467
|
if (stage.stageFailureReason == null) return null;
|
|
1197
1468
|
return {
|
|
1198
1469
|
type: 'stageFailed', stageId: stage.id, impactBand: 'critical',
|
|
1199
1470
|
variant: 'stageFailure',
|
|
1200
|
-
metric: 'stageFailureReason',
|
|
1471
|
+
metric: 'stageFailureReason', valueText: stage.stageFailureReason,
|
|
1201
1472
|
numTasks: stage.taskCount,
|
|
1202
1473
|
memoryBytesSpilled: stage.memoryBytesSpilled,
|
|
1203
1474
|
failedTaskDetails: stage.failedTaskSamples ?? [],
|
|
1204
1475
|
recommendation: `This stage attempt failed outright. Inspect the driver log for the failure reason and the job that triggered it.`,
|
|
1205
1476
|
};
|
|
1206
1477
|
},
|
|
1207
|
-
|
|
1208
|
-
|
|
1209
|
-
|
|
1478
|
+
estimate: noWasteModel,
|
|
1479
|
+
}),
|
|
1480
|
+
defineStageDetector({
|
|
1481
|
+
type: 'failures', order: 40, fixEffort: 'code', version: 2,
|
|
1482
|
+
emits: ['failures'],
|
|
1210
1483
|
docAnchor: '#bottleneck-failures',
|
|
1211
1484
|
thresholds: { minTasks: 10, warnRate: 0.05, critRate: 0.20 },
|
|
1212
|
-
detect(
|
|
1213
|
-
|
|
1214
|
-
stage ,
|
|
1215
|
-
) {
|
|
1216
|
-
if (stage.taskCount < this.thresholds.minTasks) return null;
|
|
1485
|
+
detect(stage, _ctx, thresholds) {
|
|
1486
|
+
if (stage.taskCount < thresholds.minTasks) return null;
|
|
1217
1487
|
if (!stage.failedTasks) return null;
|
|
1218
1488
|
const failureRate = stage.failedTasks / stage.taskCount;
|
|
1219
|
-
if (failureRate <=
|
|
1489
|
+
if (failureRate <= thresholds.warnRate) return null;
|
|
1220
1490
|
const value = Math.round(failureRate * 1000) / 10;
|
|
1221
1491
|
const dominantReason = pickDominantReason(stage.failureReasons);
|
|
1492
|
+
// Groups arrive most frequent first. Name the dominant error from the largest group under the
|
|
1493
|
+
// dominant tag, so the error and the tag agree even when one tag splits into many messages.
|
|
1494
|
+
const allGroups = stage.failureGroups ?? [];
|
|
1495
|
+
const dominantGroup = allGroups.find((g) => g.reason === dominantReason);
|
|
1496
|
+
const dominantError = (dominantGroup ? describeTaskFailure(dominantGroup) : null) ?? dominantReason;
|
|
1497
|
+
const failureGroups = allGroups.slice(0, MAX_FAILURE_GROUPS);
|
|
1498
|
+
const groupedTasks = failureGroups.reduce((sum, g) => sum + g.count, 0);
|
|
1222
1499
|
return {
|
|
1223
1500
|
type: 'failures', stageId: stage.id,
|
|
1224
|
-
impactBand: failureRate >
|
|
1501
|
+
impactBand: failureRate > thresholds.critRate ? 'critical' : 'warning',
|
|
1225
1502
|
metric: 'failureRate', value,
|
|
1226
1503
|
failedTasks: stage.failedTasks,
|
|
1227
1504
|
dominantReason,
|
|
1228
|
-
|
|
1505
|
+
dominantError,
|
|
1506
|
+
// One entry per distinct error (tag, class, message, loss reason), each with one bounded
|
|
1507
|
+
// stack excerpt; otherFailedTasks counts the failed tasks no shown group covers.
|
|
1508
|
+
failureGroups,
|
|
1509
|
+
otherFailedTasks: Math.max(0, stage.failedTasks - groupedTasks),
|
|
1510
|
+
recommendation: `${value}% of tasks failed${dominantError ? ` (dominant error: ${dominantError})` : ''}: investigate driver logs for executor instability or data-driven errors.`,
|
|
1229
1511
|
};
|
|
1230
1512
|
},
|
|
1231
|
-
|
|
1232
|
-
|
|
1233
|
-
|
|
1513
|
+
estimate: noWasteModel,
|
|
1514
|
+
}),
|
|
1515
|
+
defineStageDetector({
|
|
1516
|
+
type: 'straggler', order: 70, fixEffort: 'code', version: 1,
|
|
1517
|
+
emits: ['straggler'],
|
|
1234
1518
|
docAnchor: '#bottleneck-straggler',
|
|
1235
|
-
// floorPctWarn/floorPctCrit
|
|
1236
|
-
//
|
|
1519
|
+
// floorPctWarn/floorPctCrit default to impact-band.ts's run-wide noise floor, so a tail this
|
|
1520
|
+
// gate admits at its warn floor grades at least warning there too.
|
|
1237
1521
|
// shareWarnAtFloor: in a large stage, the few stragglers that gate it for tens of seconds can be
|
|
1238
1522
|
// only 2.5-5% of its tasks. Scored against a task-level replay of every stage on 14 real logs
|
|
1239
1523
|
// (recoverable = replay with each task over 4x P50 capped at P50), admitting 2.5-5% shares
|
|
@@ -1243,40 +1527,27 @@ export const DETECTORS = [
|
|
|
1243
1527
|
// than the stage's own duration, so every finding there graded info. On the 14 real logs that
|
|
1244
1528
|
// was 671 of 753 straggler findings, none above info; the slow tail is still real on those
|
|
1245
1529
|
// stages, the floor is why they're dropped.
|
|
1246
|
-
thresholds: { minTasks: 10, shareWarn: 0.05, shareWarnAtFloor: 0.025, warnPct: 0.10, critPct: 0.20, floorPctWarn:
|
|
1247
|
-
detect(
|
|
1248
|
-
|
|
1249
|
-
|
|
1250
|
-
|
|
1251
|
-
|
|
1252
|
-
|
|
1253
|
-
|
|
1254
|
-
stage ,
|
|
1255
|
-
ctx ,
|
|
1256
|
-
) {
|
|
1257
|
-
if (stage.taskCount < this.thresholds.minTasks) return null;
|
|
1258
|
-
const appDurationMs = computeAppDurationMs(ctx);
|
|
1259
|
-
if (stageBelowRuntimeFloor(stage, ctx, this.thresholds.floorPctWarn)) return null;
|
|
1530
|
+
thresholds: { minTasks: 10, shareWarn: 0.05, shareWarnAtFloor: 0.025, warnPct: 0.10, critPct: 0.20, floorPctWarn: IMPACT_FLOOR_PCT_WARN, floorPctCrit: IMPACT_FLOOR_PCT_CRIT },
|
|
1531
|
+
detect(stage, ctx, thresholds) {
|
|
1532
|
+
if (stage.taskCount < thresholds.minTasks) return null;
|
|
1533
|
+
const runMs = appDurationMs(ctx.app);
|
|
1534
|
+
if (stageBelowRuntimeFloor(stage, ctx, thresholds.floorPctWarn)) return null;
|
|
1260
1535
|
const stragglerShare = (stage.stragglerCount ?? 0) / stage.taskCount;
|
|
1261
1536
|
const useSpeculative = (stage.speculativeTasks ?? 0) > 0;
|
|
1262
|
-
if (!useSpeculative && stragglerShare <=
|
|
1537
|
+
if (!useSpeculative && stragglerShare <= thresholds.shareWarnAtFloor) return null;
|
|
1263
1538
|
const speculativeShare = useSpeculative ? stage.speculativeTasks / stage.taskCount : 0;
|
|
1264
|
-
//
|
|
1265
|
-
//
|
|
1266
|
-
|
|
1267
|
-
const
|
|
1268
|
-
const
|
|
1269
|
-
const wasteMs = tailRecoveryMs(stage, singleDelta);
|
|
1270
|
-
const floorWasteMs = clippedWasteMs(wasteMs, stage.id, ctx, tailRemovedWorkMs(stage, singleDelta), longestTaskAfterFixMs);
|
|
1271
|
-
const meetsWarnFloor = meetsRuntimeFloor(floorWasteMs, appDurationMs, this.thresholds.floorPctWarn);
|
|
1272
|
-
const meetsCritFloor = meetsRuntimeFloor(floorWasteMs, appDurationMs, this.thresholds.floorPctCrit);
|
|
1539
|
+
// The claim estimate() reports as savings: a high straggler/speculative share on a stage
|
|
1540
|
+
// whose tasks barely vary models near-zero savings, so it must not outrank 'info'.
|
|
1541
|
+
const floorWasteMs = tailClaimFloorMs(stragglerTailClaim(stage), stage.id, ctx);
|
|
1542
|
+
const meetsWarnFloor = meetsRuntimeFloor(floorWasteMs, runMs, thresholds.floorPctWarn);
|
|
1543
|
+
const meetsCritFloor = meetsRuntimeFloor(floorWasteMs, runMs, thresholds.floorPctCrit);
|
|
1273
1544
|
// The lower gate needs positive evidence the tail matters: meetsRuntimeFloor passes by
|
|
1274
1545
|
// default when the app's duration is unknown (an incomplete run), which isn't that.
|
|
1275
|
-
const stragglerShareFires = stragglerShare >
|
|
1276
|
-
|| (stragglerShare >
|
|
1546
|
+
const stragglerShareFires = stragglerShare > thresholds.shareWarn
|
|
1547
|
+
|| (stragglerShare > thresholds.shareWarnAtFloor && runMs != null && meetsWarnFloor);
|
|
1277
1548
|
if (!useSpeculative && !stragglerShareFires) return null;
|
|
1278
|
-
const speculativeTier = speculativeShare >=
|
|
1279
|
-
: speculativeShare >=
|
|
1549
|
+
const speculativeTier = speculativeShare >= thresholds.critPct && meetsCritFloor ? 'critical'
|
|
1550
|
+
: speculativeShare >= thresholds.warnPct && meetsWarnFloor ? 'warning' : 'info';
|
|
1280
1551
|
// Straggler share has no dedicated critical tier per detector-contract.md; only warning.
|
|
1281
1552
|
const stragglerTier = stragglerShareFires && meetsWarnFloor ? 'warning' : 'info';
|
|
1282
1553
|
// Fixed fallback: overwritten by deriveImpactBand when this finding gets a real wallClock
|
|
@@ -1298,46 +1569,53 @@ export const DETECTORS = [
|
|
|
1298
1569
|
speculativeTasks: stage.speculativeTasks ?? 0,
|
|
1299
1570
|
stragglerCount: stage.stragglerCount ?? 0,
|
|
1300
1571
|
confidence: useSpeculativeMetric
|
|
1301
|
-
? stragglerConfidence(speculativeShare,
|
|
1302
|
-
: stragglerConfidence(stragglerShare,
|
|
1303
|
-
validationRequired:
|
|
1572
|
+
? stragglerConfidence(speculativeShare, thresholds.warnPct, thresholds.critPct)
|
|
1573
|
+
: stragglerConfidence(stragglerShare, thresholds.shareWarn, thresholds.critPct),
|
|
1574
|
+
validationRequired: `This finding is gated by ${shareLabel(thresholds.floorPctWarn)}/${shareLabel(thresholds.floorPctCrit)} runtime-floor thresholds, our own noise floor for this metric.`,
|
|
1304
1575
|
recommendation: `${detail}: rule out a GC pause or a slow shuffle fetch before assuming a hardware issue; if a skewed key is the real cause, that's a candidate for AQE's skew-join handling.`,
|
|
1305
1576
|
};
|
|
1306
1577
|
},
|
|
1307
|
-
|
|
1308
|
-
|
|
1309
|
-
|
|
1578
|
+
estimate(finding, ctx) {
|
|
1579
|
+
if (finding.stageId == null) return null;
|
|
1580
|
+
const stage = ctx.stages.get(finding.stageId);
|
|
1581
|
+
if (!stage) return null;
|
|
1582
|
+
return tailClaimImpact(stragglerTailClaim(stage), finding.stageId, ctx);
|
|
1583
|
+
},
|
|
1584
|
+
}),
|
|
1585
|
+
defineStageDetector({
|
|
1586
|
+
type: 'speculationWaste', order: 71, fixEffort: 'config', version: 1,
|
|
1587
|
+
emits: ['speculationWaste'],
|
|
1310
1588
|
docAnchor: '#bottleneck-speculation-waste',
|
|
1311
1589
|
thresholds: { minWasted: 5, minWasteMs: 60000 },
|
|
1312
|
-
detect(
|
|
1313
|
-
|
|
1314
|
-
|
|
1315
|
-
|
|
1316
|
-
stage ,
|
|
1317
|
-
) {
|
|
1590
|
+
detect(stage, _ctx, thresholds) {
|
|
1318
1591
|
const wasted = stage.speculationWastedAttempts ?? 0;
|
|
1319
1592
|
const wastedMs = stage.speculationWasteMs ?? 0;
|
|
1320
|
-
if (wasted <
|
|
1593
|
+
if (wasted < thresholds.minWasted || wastedMs < thresholds.minWasteMs) return null;
|
|
1321
1594
|
return {
|
|
1322
1595
|
type: 'speculationWaste', stageId: stage.id,
|
|
1323
1596
|
impactBand: 'warning',
|
|
1324
1597
|
metric: 'speculationWasteMs', value: wastedMs,
|
|
1325
|
-
confidence: speculationWasteConfidence(wastedMs,
|
|
1598
|
+
confidence: speculationWasteConfidence(wastedMs, thresholds.minWasteMs),
|
|
1326
1599
|
recommendation: `Speculative execution discarded ${Math.round(wastedMs / 1000)}s of executor time in this stage; if task durations are naturally variable rather than genuine stragglers, consider tuning spark.speculation.multiplier/quantile.`,
|
|
1327
1600
|
};
|
|
1328
1601
|
},
|
|
1329
|
-
|
|
1330
|
-
|
|
1331
|
-
|
|
1602
|
+
estimate(finding, ctx) {
|
|
1603
|
+
if (finding.stageId == null) return null;
|
|
1604
|
+
const stage = ctx.stages.get(finding.stageId);
|
|
1605
|
+
if (!stage) return null;
|
|
1606
|
+
const wasteMs = (stage.speculationWasteMs ) ?? 0;
|
|
1607
|
+
return singleStageImpact(wasteMs, finding.stageId, ctx, 'measured', { value: wasteMs, unit: 'ms' });
|
|
1608
|
+
},
|
|
1609
|
+
}),
|
|
1610
|
+
defineStageDetector({
|
|
1611
|
+
type: 'retryWaste', order: 45, fixEffort: 'code', version: 1,
|
|
1612
|
+
emits: ['retryWaste'],
|
|
1332
1613
|
docAnchor: '#bottleneck-retry-waste',
|
|
1333
1614
|
thresholds: { minWasted: 3, minWasteMs: 30000 },
|
|
1334
|
-
detect(
|
|
1335
|
-
|
|
1336
|
-
stage ,
|
|
1337
|
-
) {
|
|
1615
|
+
detect(stage, _ctx, thresholds) {
|
|
1338
1616
|
const wasted = stage.wastedAttempts ?? 0;
|
|
1339
1617
|
const wastedMs = stage.retryWasteMs ?? 0;
|
|
1340
|
-
if (wasted <
|
|
1618
|
+
if (wasted < thresholds.minWasted || wastedMs < thresholds.minWasteMs) return null;
|
|
1341
1619
|
return {
|
|
1342
1620
|
type: 'retryWaste', stageId: stage.id,
|
|
1343
1621
|
impactBand: 'warning',
|
|
@@ -1349,22 +1627,29 @@ export const DETECTORS = [
|
|
|
1349
1627
|
extended: `${wasted} task attempts were superseded by a later retry, wasting ${Math.round(wastedMs / 1000)}s of executor time. Common causes: executor loss (OOM-kill, node death) or shuffle FetchFailed forcing a stage-map recompute. Check driver logs for the dominant reason (see the Failures widget) even if the final failure rate looks low; retries hide the true cost.`,
|
|
1350
1628
|
};
|
|
1351
1629
|
},
|
|
1352
|
-
|
|
1353
|
-
|
|
1354
|
-
|
|
1630
|
+
estimate(finding, ctx) {
|
|
1631
|
+
// The waste figure lives on the Stage, not the Finding: detect() only re-publishes it as metric/value.
|
|
1632
|
+
if (finding.stageId == null) return null;
|
|
1633
|
+
const stage = ctx.stages.get(finding.stageId);
|
|
1634
|
+
if (!stage) return null;
|
|
1635
|
+
const wasteMs = (stage.retryWasteMs ) ?? 0;
|
|
1636
|
+
const wallClockMs = retryWallClockMs(stage);
|
|
1637
|
+
return singleStageImpact(wallClockMs, finding.stageId, ctx,
|
|
1638
|
+
wallClockMs === wasteMs ? 'measured' : 'modeled', { value: wasteMs, unit: 'ms' });
|
|
1639
|
+
},
|
|
1640
|
+
}),
|
|
1641
|
+
defineStageDetector({
|
|
1642
|
+
type: 'tinyTask', order: 80, fixEffort: 'code', version: 1,
|
|
1643
|
+
emits: ['tinyTask'],
|
|
1355
1644
|
docAnchor: '#bottleneck-tiny-tasks',
|
|
1356
1645
|
// stageFloorPct: the tiered detectors' 0.5% runtime floor. Coalescing can't save more than the
|
|
1357
1646
|
// stage's own duration, so on a shorter stage every finding graded info: on the 14 real logs
|
|
1358
1647
|
// 132 of 151 tinyTask findings. The tasks are still tiny there; the floor is why they're dropped.
|
|
1359
1648
|
thresholds: { minTasks: 100, maxP50: 500, maxP95: 1000, stageFloorPct: 0.005 },
|
|
1360
|
-
detect(
|
|
1361
|
-
|
|
1362
|
-
stage
|
|
1363
|
-
|
|
1364
|
-
) {
|
|
1365
|
-
if (stage.taskCount < this.thresholds.minTasks) return null;
|
|
1366
|
-
if (stageBelowRuntimeFloor(stage, ctx, this.thresholds.stageFloorPct)) return null;
|
|
1367
|
-
if (stage.taskDurationP50 > this.thresholds.maxP50 || stage.taskDurationP95 > this.thresholds.maxP95) return null;
|
|
1649
|
+
detect(stage, ctx, thresholds) {
|
|
1650
|
+
if (stage.taskCount < thresholds.minTasks) return null;
|
|
1651
|
+
if (stageBelowRuntimeFloor(stage, ctx, thresholds.stageFloorPct)) return null;
|
|
1652
|
+
if (stage.taskDurationP50 > thresholds.maxP50 || stage.taskDurationP95 > thresholds.maxP95) return null;
|
|
1368
1653
|
const coalesceTo = Math.max(1, Math.round(stage.taskCount / 10));
|
|
1369
1654
|
const fix = stage.shuffleReadBytes > 0
|
|
1370
1655
|
? `lower spark.sql.shuffle.partitions or .coalesce(${coalesceTo})`
|
|
@@ -1375,30 +1660,46 @@ export const DETECTORS = [
|
|
|
1375
1660
|
recommendation: `Many small tasks (${stage.taskCount}, P50 ${Math.round(stage.taskDurationP50)}ms): scheduler overhead may dominate. Try ${fix}.`,
|
|
1376
1661
|
};
|
|
1377
1662
|
},
|
|
1378
|
-
|
|
1379
|
-
|
|
1663
|
+
estimate(finding, ctx) {
|
|
1664
|
+
if (finding.stageId == null) return null;
|
|
1665
|
+
const stage = ctx.stages.get(finding.stageId);
|
|
1666
|
+
if (!stage) return null;
|
|
1667
|
+
const taskCount = stage.taskCount ?? 0;
|
|
1668
|
+
const excessTaskCount = Math.max(0, taskCount - Math.round(taskCount / 10));
|
|
1669
|
+
const measured = measuredTaskOverhead(stage);
|
|
1670
|
+
if (measured) {
|
|
1671
|
+
// Coalescing to a tenth of the tasks removes the excess tasks' per-task overhead: core
|
|
1672
|
+
// time spent in parallel, so wall-clock at the stage's achieved concurrency (floored at
|
|
1673
|
+
// 1: a mostly-idle stage can't save more wall-clock than the task time it removes).
|
|
1674
|
+
const wasteMs = (excessTaskCount * measured.perTaskMs) / Math.max(1, measured.concurrency);
|
|
1675
|
+
return singleStageImpact(wasteMs, finding.stageId, ctx, 'measured', { value: wasteMs, unit: 'ms' });
|
|
1676
|
+
}
|
|
1677
|
+
const wasteMs = excessTaskCount * TASK_SCHEDULING_OVERHEAD_MS;
|
|
1678
|
+
return singleStageImpact(wasteMs, finding.stageId, ctx, 'modeled', { value: wasteMs, unit: 'ms' });
|
|
1679
|
+
},
|
|
1680
|
+
}),
|
|
1681
|
+
defineAppDetector({
|
|
1380
1682
|
// No docAnchor: the upstream spark-tuning-reference docs have no section for this
|
|
1381
1683
|
// tool-specific "capture stopped early" signal.
|
|
1382
|
-
type: 'incompleteRun',
|
|
1684
|
+
type: 'incompleteRun', order: 5, fixEffort: 'code', version: 1,
|
|
1685
|
+
emits: ['incompleteRun'],
|
|
1383
1686
|
thresholds: {},
|
|
1384
|
-
|
|
1385
|
-
detect( ctx ) {
|
|
1687
|
+
detect(ctx) {
|
|
1386
1688
|
if (!ctx.app || ctx.app.startTime == null || ctx.app.endTime != null) return null;
|
|
1387
1689
|
return {
|
|
1388
1690
|
type: 'incompleteRun', stageId: null, impactBand: 'warning',
|
|
1389
|
-
metric: 'applicationEnd',
|
|
1390
|
-
recommendation:
|
|
1691
|
+
metric: 'applicationEnd', valueText: 'missing',
|
|
1692
|
+
recommendation: INCOMPLETE_RUN_RECOMMENDATION,
|
|
1391
1693
|
};
|
|
1392
1694
|
},
|
|
1393
|
-
|
|
1394
|
-
|
|
1395
|
-
|
|
1695
|
+
estimate: noWasteModel,
|
|
1696
|
+
}),
|
|
1697
|
+
defineAppDetector({
|
|
1698
|
+
type: 'coldStart', order: 90, fixEffort: 'code', version: 1,
|
|
1699
|
+
emits: ['coldStart'],
|
|
1396
1700
|
docAnchor: '#bottleneck-cold-start',
|
|
1397
1701
|
thresholds: { gapSeconds: 30 },
|
|
1398
|
-
detect(
|
|
1399
|
-
|
|
1400
|
-
ctx ,
|
|
1401
|
-
) {
|
|
1702
|
+
detect(ctx, thresholds) {
|
|
1402
1703
|
const { app, stages, executorsAdded, executorsRemoved } = ctx;
|
|
1403
1704
|
// Nullish (not falsy) check: a literal startTime:0 must not be treated as "missing".
|
|
1404
1705
|
if (!app || app.startTime == null || stages.size === 0) return null;
|
|
@@ -1429,7 +1730,7 @@ export const DETECTORS = [
|
|
|
1429
1730
|
}
|
|
1430
1731
|
if (!Number.isFinite(firstExecutorAdded)) return null;
|
|
1431
1732
|
const gapSeconds = (firstExecutorAdded - firstStageSubmitted) / 1000;
|
|
1432
|
-
if (gapSeconds <=
|
|
1733
|
+
if (gapSeconds <= thresholds.gapSeconds) return null;
|
|
1433
1734
|
const value = Math.round(gapSeconds);
|
|
1434
1735
|
return {
|
|
1435
1736
|
type: 'coldStart', stageId: null, impactBand: 'warning',
|
|
@@ -1437,15 +1738,21 @@ export const DETECTORS = [
|
|
|
1437
1738
|
recommendation: `The first stage waited ${value}s for an executor to become available: keep a warm pool of idle executors, or if using dynamic allocation, raise the minimum/initial executor count so it doesn't scale up from zero.`,
|
|
1438
1739
|
};
|
|
1439
1740
|
},
|
|
1440
|
-
|
|
1441
|
-
|
|
1442
|
-
|
|
1741
|
+
estimate(finding) {
|
|
1742
|
+
// detect() reports the gap as `metric: 'startupGapSeconds', value: <seconds>`.
|
|
1743
|
+
if (typeof finding.value !== 'number') return null;
|
|
1744
|
+
const wasteMs = finding.value * 1000;
|
|
1745
|
+
// Time before any task starts can never overlap any stage; a genuine unclipped point estimate,
|
|
1746
|
+
// not tied to any stage's gate (coldStart is app-scoped, stageId: null).
|
|
1747
|
+
return { basis: 'serial', wallClock: { low: wasteMs, high: wasteMs }, estimateMethod: 'measured' };
|
|
1748
|
+
},
|
|
1749
|
+
}),
|
|
1750
|
+
defineAppDetector({
|
|
1751
|
+
type: 'utilization', order: 100, fixEffort: 'config', version: 1,
|
|
1752
|
+
emits: ['utilization'],
|
|
1443
1753
|
docAnchor: '#bottleneck-utilization',
|
|
1444
1754
|
thresholds: { minUtil: 0.60 },
|
|
1445
|
-
detect(
|
|
1446
|
-
|
|
1447
|
-
ctx ,
|
|
1448
|
-
) {
|
|
1755
|
+
detect(ctx, thresholds) {
|
|
1449
1756
|
const { app, executorsAdded, executorsRemoved, runAggregates } = ctx;
|
|
1450
1757
|
// Nullish (not falsy) check: a literal startTime:0 must not be treated as "missing".
|
|
1451
1758
|
if (!app || executorsAdded.length === 0 || app.startTime == null || app.endTime == null) return null;
|
|
@@ -1464,7 +1771,7 @@ export const DETECTORS = [
|
|
|
1464
1771
|
// lifetime-based measure this replaces.
|
|
1465
1772
|
const busyCoreMs = runAggregates?.busyCoreMs ?? 0;
|
|
1466
1773
|
const utilization = busyCoreMs / capacityCoreMs;
|
|
1467
|
-
if (utilization >=
|
|
1774
|
+
if (utilization >= thresholds.minUtil) return null;
|
|
1468
1775
|
|
|
1469
1776
|
// CPU-time-based utilization (sparkMeasure): metric only, no threshold.
|
|
1470
1777
|
let cpuUtilizationPct = null;
|
|
@@ -1486,9 +1793,20 @@ export const DETECTORS = [
|
|
|
1486
1793
|
recommendation: `Average executor utilization was only ${value}%: consider reducing cluster size or enabling dynamic allocation.`,
|
|
1487
1794
|
};
|
|
1488
1795
|
},
|
|
1489
|
-
|
|
1490
|
-
|
|
1491
|
-
|
|
1796
|
+
estimate(finding) {
|
|
1797
|
+
const fraction = finding.utilizationFraction ;
|
|
1798
|
+
const appDurationMs = finding.appDurationMs ;
|
|
1799
|
+
const totalCores = finding.totalCores ;
|
|
1800
|
+
if (fraction == null || appDurationMs == null || totalCores == null) {
|
|
1801
|
+
return costOnly('measured');
|
|
1802
|
+
}
|
|
1803
|
+
const idleCoreHours = (1 - fraction) * appDurationMs * totalCores / 3.6e6;
|
|
1804
|
+
return costOnly('measured', { value: idleCoreHours, unit: 'coreHours' });
|
|
1805
|
+
},
|
|
1806
|
+
}),
|
|
1807
|
+
defineAppDetector({
|
|
1808
|
+
type: 'memoryUtilization', order: 102, fixEffort: 'config', version: 1,
|
|
1809
|
+
emits: ['memoryUtilization'],
|
|
1492
1810
|
docAnchor: '#bottleneck-memory-utilization',
|
|
1493
1811
|
thresholds: {
|
|
1494
1812
|
idleCoreWarn: 0.50, // WastedCoresAlertsReducer
|
|
@@ -1496,14 +1814,7 @@ export const DETECTORS = [
|
|
|
1496
1814
|
bandTooHigh: 0.70, // below this => over-provisioned (cost signal)
|
|
1497
1815
|
wasteBufferMultiplier: 1.5, // UNVERIFIED
|
|
1498
1816
|
},
|
|
1499
|
-
detect(
|
|
1500
|
-
|
|
1501
|
-
|
|
1502
|
-
|
|
1503
|
-
|
|
1504
|
-
|
|
1505
|
-
ctx ,
|
|
1506
|
-
) {
|
|
1817
|
+
detect(ctx, thresholds) {
|
|
1507
1818
|
const { app, executorsAdded, executorsRemoved, runAggregates, stages } = ctx;
|
|
1508
1819
|
const out = [];
|
|
1509
1820
|
// Nullish (not falsy) check: a literal startTime:0 must not be treated as "missing".
|
|
@@ -1521,10 +1832,11 @@ export const DETECTORS = [
|
|
|
1521
1832
|
const allocatedMB = app.resources?.executor?.memoryMB ?? null;
|
|
1522
1833
|
|
|
1523
1834
|
// ── 1a idle-cores rate ────────────────────────────────────────────────
|
|
1524
|
-
|
|
1835
|
+
const busyCoreMs = runAggregates?.busyCoreMs;
|
|
1836
|
+
if (busyCoreMs != null && totalCores > 0) {
|
|
1525
1837
|
const capacityCoreMs = totalCores * appDurationMs;
|
|
1526
|
-
const idleRate = capacityCoreMs > 0 ? 1 - (
|
|
1527
|
-
if (idleRate >
|
|
1838
|
+
const idleRate = capacityCoreMs > 0 ? 1 - (busyCoreMs / capacityCoreMs) : 0;
|
|
1839
|
+
if (idleRate > thresholds.idleCoreWarn) {
|
|
1528
1840
|
const value = Math.round(idleRate * 100);
|
|
1529
1841
|
out.push({
|
|
1530
1842
|
type: 'memoryUtilization', variant: 'idleCores', stageId: null,
|
|
@@ -1559,14 +1871,14 @@ export const DETECTORS = [
|
|
|
1559
1871
|
const ratio = heap / allocatedBytes;
|
|
1560
1872
|
// The two bands are opposite signals: an explicit `rule` discriminator lets consumers
|
|
1561
1873
|
// tell OOM-risk from over-provisioning without re-deriving the ratio.
|
|
1562
|
-
if (ratio >
|
|
1874
|
+
if (ratio > thresholds.bandTooSmall) {
|
|
1563
1875
|
out.push({
|
|
1564
1876
|
type: 'memoryUtilization', variant: 'memoryBand', rule: 'heapNearCapacity',
|
|
1565
1877
|
stageId: null, executorId: execId,
|
|
1566
1878
|
impactBand: 'warning', metric: 'heapUsedRatio', value: Math.round(ratio * 100),
|
|
1567
1879
|
recommendation: `Executor ${execId} peaked at ${Math.round(ratio * 100)}% of allocated heap: memory may be too small; raise spark.executor.memory to avoid OOM/spill.`,
|
|
1568
1880
|
});
|
|
1569
|
-
} else if (ratio <
|
|
1881
|
+
} else if (ratio < thresholds.bandTooHigh) {
|
|
1570
1882
|
out.push({
|
|
1571
1883
|
type: 'memoryUtilization', variant: 'memoryBand', rule: 'heapOverProvisioned',
|
|
1572
1884
|
stageId: null, executorId: execId,
|
|
@@ -1587,13 +1899,13 @@ export const DETECTORS = [
|
|
|
1587
1899
|
for (const s of stages.values()) usedRunTimeMs += s.executorRunTime ?? 0;
|
|
1588
1900
|
const usedMBSeconds = allocatedMB * (usedRunTimeMs / 1000);
|
|
1589
1901
|
const wastedMBSeconds = allocatedMBSeconds - usedMBSeconds;
|
|
1590
|
-
if (wastedMBSeconds >
|
|
1902
|
+
if (wastedMBSeconds > thresholds.wasteBufferMultiplier * usedMBSeconds) {
|
|
1591
1903
|
const value = Math.round(wastedMBSeconds);
|
|
1592
1904
|
out.push({
|
|
1593
1905
|
type: 'memoryUtilization', variant: 'wasteModel', stageId: null,
|
|
1594
1906
|
impactBand: 'info', metric: 'wastedMBSeconds', value,
|
|
1595
|
-
confidence: memoryWasteConfidence(wastedMBSeconds, usedMBSeconds,
|
|
1596
|
-
validationRequired:
|
|
1907
|
+
confidence: memoryWasteConfidence(wastedMBSeconds, usedMBSeconds, thresholds.wasteBufferMultiplier),
|
|
1908
|
+
validationRequired: `Memory-waste estimate uses allocated-vs-used memory-time and a ${thresholds.wasteBufferMultiplier}x buffer: confirm against the Spark UI before acting.`,
|
|
1597
1909
|
recommendation: `Allocated executor memory sat largely idle over the run (~${value.toLocaleString('en-US')} MB-seconds wasted): review spark.executor.memory and executor count.`,
|
|
1598
1910
|
});
|
|
1599
1911
|
}
|
|
@@ -1601,97 +1913,153 @@ export const DETECTORS = [
|
|
|
1601
1913
|
|
|
1602
1914
|
return out;
|
|
1603
1915
|
},
|
|
1604
|
-
|
|
1605
|
-
|
|
1916
|
+
estimate(finding) {
|
|
1917
|
+
// The wasteModel variant reports metric: 'wastedMBSeconds', value: <MB-seconds>.
|
|
1918
|
+
if (finding.variant === 'wasteModel' && typeof finding.value === 'number') {
|
|
1919
|
+
return costOnly('measured', { value: finding.value, unit: 'mbSeconds' });
|
|
1920
|
+
}
|
|
1921
|
+
if (finding.variant === 'idleCores') {
|
|
1922
|
+
// Idle core-time priced as memory held but unused: the same MB-seconds unit as wasteModel, so comparable.
|
|
1923
|
+
const idleRateFraction = finding.idleRateFraction ;
|
|
1924
|
+
const allocatedMB = finding.allocatedMB ;
|
|
1925
|
+
const peakExecutors = finding.peakExecutors ;
|
|
1926
|
+
const appDurationMs = finding.appDurationMs ;
|
|
1927
|
+
if (idleRateFraction != null && allocatedMB != null && peakExecutors != null && appDurationMs != null) {
|
|
1928
|
+
const wastedMBSeconds = idleRateFraction * allocatedMB * peakExecutors * (appDurationMs / 1000);
|
|
1929
|
+
return costOnly('modeled', { value: wastedMBSeconds, unit: 'mbSeconds' });
|
|
1930
|
+
}
|
|
1931
|
+
return costOnly('modeled');
|
|
1932
|
+
}
|
|
1933
|
+
// Only the over-provisioned band is a waste; the near-capacity band is an OOM-risk signal with
|
|
1934
|
+
// no magnitude, and the dataUnavailable shape has no inputs: both stay informational.
|
|
1935
|
+
if (finding.variant === 'memoryBand' && finding.rule === 'heapOverProvisioned') {
|
|
1936
|
+
const allocatedBytes = finding.allocatedBytes ;
|
|
1937
|
+
const heap = finding.heap ;
|
|
1938
|
+
const appDurationMs = finding.appDurationMs ;
|
|
1939
|
+
if (allocatedBytes != null && heap != null && appDurationMs != null) {
|
|
1940
|
+
const unusedMB = (allocatedBytes - heap) / (1024 * 1024);
|
|
1941
|
+
const wastedMBSeconds = unusedMB * (appDurationMs / 1000);
|
|
1942
|
+
return costOnly('modeled', { value: wastedMBSeconds, unit: 'mbSeconds' });
|
|
1943
|
+
}
|
|
1944
|
+
}
|
|
1945
|
+
return costOnly('modeled');
|
|
1946
|
+
},
|
|
1947
|
+
}),
|
|
1948
|
+
defineAppDetector({
|
|
1606
1949
|
// Per-RDD cache-utilization proxies (this repo's own design: Spark event logs carry no
|
|
1607
1950
|
// block-access events, so a literal cache hit rate isn't derivable). Two per-RDD tiered
|
|
1608
|
-
// checks over rddInfo
|
|
1609
|
-
|
|
1951
|
+
// checks over rddInfo: partial caching and disk spillover. An RDD can produce both. rddInfo's
|
|
1952
|
+
// sizes come from SparkListenerBlockUpdated when the log has it, else from StageSubmitted's
|
|
1953
|
+
// RDD Info (0 since Spark 2.3; Spark 1.x fills it only on StageCompleted); with neither, on
|
|
1954
|
+
// Spark 2.3+ with logBlockUpdates off, a storageUnobserved caveat replaces them.
|
|
1955
|
+
type: 'cacheUtilization', order: 103, fixEffort: 'code', version: 2,
|
|
1956
|
+
emits: ['cacheUtilization'],
|
|
1610
1957
|
docAnchor: '#bottleneck-cache-utilization',
|
|
1611
1958
|
thresholds: {
|
|
1612
1959
|
cachedRatioWarn: 0.50, cachedRatioInfo: 0.90,
|
|
1613
1960
|
diskRatioWarn: 0.40, diskRatioInfo: 0.15,
|
|
1614
1961
|
},
|
|
1615
|
-
detect(
|
|
1616
|
-
|
|
1617
|
-
|
|
1618
|
-
|
|
1619
|
-
|
|
1620
|
-
|
|
1621
|
-
ctx ,
|
|
1622
|
-
) {
|
|
1962
|
+
detect(ctx, thresholds) {
|
|
1623
1963
|
const rddInfo = ctx.app?.rddInfo;
|
|
1624
1964
|
if (!(rddInfo instanceof Map)) return null;
|
|
1625
1965
|
const out = [];
|
|
1966
|
+
let persistedRddCount = 0;
|
|
1967
|
+
// With block-update logging on, zero rdd_* updates means nothing was ever cached, not a gap.
|
|
1968
|
+
const blockUpdatesLogged = String(ctx.app?.config?.['spark.eventLog.logBlockUpdates.enabled']).toLowerCase() === 'true';
|
|
1969
|
+
// Spark before 2.3 has no block-update logging and writes RDD Info's cache figures only on
|
|
1970
|
+
// StageCompleted, which isn't read: the caveat's advice doesn't apply there. A log with no
|
|
1971
|
+
// version is pre-1.3 (no SparkListenerLogStart); every 2.3+ log records one.
|
|
1972
|
+
const version = /^(\d+)\.(\d+)/.exec(ctx.app?.sparkVersion ?? '');
|
|
1973
|
+
const preBlockUpdates = version == null || Number(version[1]) < 2 || (Number(version[1]) === 2 && Number(version[2]) < 3);
|
|
1974
|
+
let anyStorageEvidence = blockUpdatesLogged || preBlockUpdates || (ctx.app?.rddBlockUpdates ?? 0) > 0;
|
|
1626
1975
|
for (const rdd of rddInfo.values()) {
|
|
1627
1976
|
const sl = rdd.storageLevel ?? {};
|
|
1628
1977
|
if (!(sl.useMemory || sl.useDisk)) continue;
|
|
1978
|
+
persistedRddCount++;
|
|
1629
1979
|
if (!((rdd.numCachedPartitions ?? 0) > 0)) continue;
|
|
1980
|
+
anyStorageEvidence = true;
|
|
1630
1981
|
|
|
1631
1982
|
if ((rdd.numPartitions ?? 0) > 0) {
|
|
1632
1983
|
const cachedRatio = rdd.numCachedPartitions / rdd.numPartitions;
|
|
1633
|
-
if (cachedRatio <
|
|
1634
|
-
else if (cachedRatio <
|
|
1984
|
+
if (cachedRatio < thresholds.cachedRatioWarn) out.push(partialCacheFinding(rdd, cachedRatio, 'warning'));
|
|
1985
|
+
else if (cachedRatio < thresholds.cachedRatioInfo) out.push(partialCacheFinding(rdd, cachedRatio, 'info'));
|
|
1635
1986
|
}
|
|
1636
1987
|
|
|
1637
1988
|
if (sl.useMemory && sl.useDisk) {
|
|
1638
1989
|
const total = (rdd.memorySize ?? 0) + (rdd.diskSize ?? 0);
|
|
1639
1990
|
if (total > 0) {
|
|
1640
1991
|
const diskRatio = (rdd.diskSize ?? 0) / total;
|
|
1641
|
-
if (diskRatio >
|
|
1642
|
-
else if (diskRatio >
|
|
1992
|
+
if (diskRatio > thresholds.diskRatioWarn) out.push(diskSpilloverFinding(rdd, diskRatio, 'warning'));
|
|
1993
|
+
else if (diskRatio > thresholds.diskRatioInfo) out.push(diskSpilloverFinding(rdd, diskRatio, 'info'));
|
|
1643
1994
|
}
|
|
1644
1995
|
}
|
|
1645
1996
|
}
|
|
1997
|
+
if (persistedRddCount > 0 && !anyStorageEvidence) out.push(storageUnobservedFinding(persistedRddCount));
|
|
1646
1998
|
return out;
|
|
1647
1999
|
},
|
|
1648
|
-
|
|
1649
|
-
|
|
2000
|
+
estimate(finding) {
|
|
2001
|
+
// storageUnobserved reports missing evidence: no sizes, so nothing to model.
|
|
2002
|
+
if (finding.dataUnavailable) return costOnly('none');
|
|
2003
|
+
const memorySize = (finding.memorySize ) ?? 0;
|
|
2004
|
+
const diskSize = (finding.diskSize ) ?? 0;
|
|
2005
|
+
const numCachedPartitions = (finding.numCachedPartitions ) ?? 0;
|
|
2006
|
+
const numPartitions = (finding.numPartitions ) ?? 0;
|
|
2007
|
+
const numUncachedPartitions = Math.max(0, numPartitions - numCachedPartitions);
|
|
2008
|
+
const cachedBytes = memorySize + diskSize;
|
|
2009
|
+
// Extrapolate never-cached partitions' size from the CACHED partitions' average (uncached/
|
|
2010
|
+
// cached, not uncached/total: numCachedPartitions produced cachedBytes). diskSize is added
|
|
2011
|
+
// once more: those bytes are cached but on disk, so re-reading them still costs I/O like an uncached partition.
|
|
2012
|
+
const uncachedBytes = numCachedPartitions > 0 ? (cachedBytes / numCachedPartitions) * numUncachedPartitions : 0;
|
|
2013
|
+
const uncachedOrSpilledBytes = uncachedBytes + diskSize;
|
|
2014
|
+
const wasteMs = (uncachedOrSpilledBytes / RE_READ_THROUGHPUT_BPS) * 1000;
|
|
2015
|
+
return costOnly('modeled', { value: wasteMs, unit: 'ms' });
|
|
2016
|
+
},
|
|
2017
|
+
}),
|
|
2018
|
+
defineAppDetector({
|
|
1650
2019
|
// Non-local task ratio across stage.localityStats (RACK_LOCAL + ANY vs all tasks), the other
|
|
1651
2020
|
// half of the "Wasted Cores Ratio" (idle-core half is memoryUtilization's idleCores).
|
|
1652
2021
|
// NO_PREF stays in the denominator only: shuffle-read stages legitimately report it.
|
|
1653
|
-
type: 'coreLocality',
|
|
2022
|
+
type: 'coreLocality', order: 103, fixEffort: 'config', version: 1,
|
|
2023
|
+
emits: ['coreLocality'],
|
|
1654
2024
|
docAnchor: '#bottleneck-core-locality',
|
|
1655
2025
|
thresholds: { minTasks: 50, warnRatio: 0.15, critRatio: 0.35 },
|
|
1656
|
-
detect(
|
|
1657
|
-
|
|
1658
|
-
ctx ,
|
|
1659
|
-
) {
|
|
2026
|
+
detect(ctx, thresholds) {
|
|
1660
2027
|
const { totalTasks, nonLocalTasks, ratio } = computeCoreLocalityRatio([...ctx.stages.values()]);
|
|
1661
|
-
if (totalTasks == null || totalTasks <
|
|
2028
|
+
if (totalTasks == null || totalTasks < thresholds.minTasks) return null;
|
|
1662
2029
|
// computeCoreLocalityRatio only returns ratio:null together with totalTasks:null (shared
|
|
1663
2030
|
// EMPTY sentinel); the totalTasks guard above rules that out, so ratio is non-null here.
|
|
1664
|
-
if (ratio <
|
|
2031
|
+
if (ratio < thresholds.warnRatio) return null;
|
|
1665
2032
|
|
|
1666
2033
|
const value = Math.round(ratio * 100);
|
|
1667
2034
|
return {
|
|
1668
2035
|
type: 'coreLocality', stageId: null,
|
|
1669
|
-
impactBand: ratio >=
|
|
2036
|
+
impactBand: ratio >= thresholds.critRatio ? 'critical' : 'warning',
|
|
1670
2037
|
metric: 'nonLocalRatio', value,
|
|
1671
2038
|
// Raw count behind the ratio, for the impact estimator. Non-null whenever totalTasks is.
|
|
1672
2039
|
nonLocalTaskCount: nonLocalTasks ,
|
|
1673
|
-
confidence: coreLocalityConfidence(ratio , totalTasks,
|
|
1674
|
-
validationRequired:
|
|
2040
|
+
confidence: coreLocalityConfidence(ratio , totalTasks, thresholds),
|
|
2041
|
+
validationRequired: `This finding is gated by ${shareLabel(thresholds.warnRatio)}/${shareLabel(thresholds.critRatio)} non-local-ratio thresholds (and a ${thresholds.minTasks}-task minimum), our own noise floor for this metric.`,
|
|
1675
2042
|
recommendation: `${value}% of tasks (${nonLocalTasks }) ran without process- or node-local data placement: check spark.locality.wait settings and executor/data colocation.`,
|
|
1676
2043
|
};
|
|
1677
2044
|
},
|
|
1678
|
-
|
|
1679
|
-
|
|
2045
|
+
estimate(finding) {
|
|
2046
|
+
const nonLocal = (finding.nonLocalTaskCount ) ?? 0;
|
|
2047
|
+
const coreMs = nonLocal * NETWORK_FETCH_PENALTY_MS;
|
|
2048
|
+
return costOnly('modeled', { value: coreMs, unit: 'coreMs' });
|
|
2049
|
+
},
|
|
2050
|
+
}),
|
|
2051
|
+
defineAppDetector({
|
|
1680
2052
|
// Short-lived executors: stood up and torn down before doing useful work (wasteful
|
|
1681
2053
|
// re-provisioning, not normal scale-down). Reuses utilization's add/remove matching, but
|
|
1682
2054
|
// measures lifetime against a threshold instead of aggregate active-time.
|
|
1683
|
-
type: 'autoscalingChurn',
|
|
2055
|
+
type: 'autoscalingChurn', order: 103, fixEffort: 'config', version: 1,
|
|
2056
|
+
emits: ['autoscalingChurn'],
|
|
1684
2057
|
docAnchor: '#bottleneck-autoscaling-churn',
|
|
1685
2058
|
thresholds: { shortLivedMs: 120_000, warningPct: 0.30, criticalPct: 0.60, minExecutors: 5 },
|
|
1686
|
-
detect(
|
|
1687
|
-
|
|
1688
|
-
|
|
1689
|
-
|
|
1690
|
-
ctx ,
|
|
1691
|
-
) {
|
|
2059
|
+
detect(ctx, thresholds) {
|
|
1692
2060
|
const { app, executorsAdded, executorsRemoved } = ctx;
|
|
1693
2061
|
if (!app || executorsAdded.length === 0 || app.endTime == null) return null;
|
|
1694
|
-
if (executorsAdded.length <
|
|
2062
|
+
if (executorsAdded.length < thresholds.minExecutors) return null;
|
|
1695
2063
|
|
|
1696
2064
|
const removedAt = new Map ();
|
|
1697
2065
|
for (const ev of executorsRemoved) removedAt.set(ev.executorId, ev.timestamp);
|
|
@@ -1700,12 +2068,12 @@ export const DETECTORS = [
|
|
|
1700
2068
|
for (const ev of executorsAdded) {
|
|
1701
2069
|
const endedAt = removedAt.has(ev.executorId) ? removedAt.get(ev.executorId) : app.endTime;
|
|
1702
2070
|
const lifetime = endedAt - ev.timestamp;
|
|
1703
|
-
if (lifetime <
|
|
2071
|
+
if (lifetime < thresholds.shortLivedMs) shortLivedCount++;
|
|
1704
2072
|
}
|
|
1705
2073
|
|
|
1706
2074
|
const shortLivedPct = shortLivedCount / executorsAdded.length;
|
|
1707
|
-
const impactBand = shortLivedPct >
|
|
1708
|
-
: shortLivedPct >
|
|
2075
|
+
const impactBand = shortLivedPct > thresholds.criticalPct ? 'critical'
|
|
2076
|
+
: shortLivedPct > thresholds.warningPct ? 'warning' : null;
|
|
1709
2077
|
if (!impactBand) return null;
|
|
1710
2078
|
|
|
1711
2079
|
const pct = Math.round(shortLivedPct * 100);
|
|
@@ -1714,18 +2082,24 @@ export const DETECTORS = [
|
|
|
1714
2082
|
metric: 'shortLivedExecutorPct', value: pct,
|
|
1715
2083
|
// Raw count behind the percentage, for the impact estimator's startup-overhead figure.
|
|
1716
2084
|
shortLivedExecutorCount: shortLivedCount,
|
|
1717
|
-
confidence: autoscalingChurnConfidence(shortLivedPct,
|
|
2085
|
+
confidence: autoscalingChurnConfidence(shortLivedPct, thresholds.warningPct, thresholds.criticalPct),
|
|
1718
2086
|
recommendation: `${pct}% of executors ran for under 2 minutes before being removed. This looks like wasteful re-provisioning rather than normal scale-down; consider raising spark.dynamicAllocation.executorIdleTimeout or widening the minExecutors/maxExecutors bounds to reduce flapping.`,
|
|
1719
2087
|
};
|
|
1720
2088
|
},
|
|
1721
|
-
|
|
1722
|
-
|
|
2089
|
+
estimate(finding) {
|
|
2090
|
+
const shortLived = (finding.shortLivedExecutorCount ) ?? 0;
|
|
2091
|
+
const executorHours = (shortLived * EXECUTOR_STARTUP_OVERHEAD_MS) / 3.6e6;
|
|
2092
|
+
return costOnly('modeled', { value: executorHours, unit: 'coreHours' });
|
|
2093
|
+
},
|
|
2094
|
+
}),
|
|
2095
|
+
defineAppDetector({
|
|
1723
2096
|
// Cross-execution relation reuse: flags an input relation scanned by two or more SQL
|
|
1724
2097
|
// executions in one run, firing on real relation names (parquet:..., jdbc:...).
|
|
1725
|
-
type: 'cachingOpportunity',
|
|
2098
|
+
type: 'cachingOpportunity', order: 105, fixEffort: 'code', version: 1,
|
|
2099
|
+
emits: ['cachingOpportunity'],
|
|
1726
2100
|
docAnchor: '#bottleneck-caching-opportunity',
|
|
1727
2101
|
thresholds: { minExecutions: 2 },
|
|
1728
|
-
detect(
|
|
2102
|
+
detect(ctx, thresholds) {
|
|
1729
2103
|
const sql = ctx.sql;
|
|
1730
2104
|
if (!(sql instanceof Map) || sql.size === 0) return null;
|
|
1731
2105
|
|
|
@@ -1813,7 +2187,7 @@ export const DETECTORS = [
|
|
|
1813
2187
|
// Qualifying = enough distinct executions on its own. Nested-dedupe: a qualifying composite
|
|
1814
2188
|
// with a qualifying ANCESTOR is subsumed, fully (equal sets) or partially (residual).
|
|
1815
2189
|
const isQualifying = (fp ) =>
|
|
1816
|
-
byComposite.has(fp) && byComposite.get(fp) .executionIds.size >=
|
|
2190
|
+
byComposite.has(fp) && byComposite.get(fp) .executionIds.size >= thresholds.minExecutions;
|
|
1817
2191
|
const compositeResolutions = new Map ();
|
|
1818
2192
|
for (const [fingerprint, agg] of byComposite) {
|
|
1819
2193
|
if (!isQualifying(fingerprint)) { compositeResolutions.set(fingerprint, { finalExecutionIds: agg.executionIds, suppressed: true }); continue; }
|
|
@@ -1822,7 +2196,7 @@ export const DETECTORS = [
|
|
|
1822
2196
|
const coveredByAncestors = new Set(qualifyingAncestors.flatMap(outer => [...outer.executionIds]));
|
|
1823
2197
|
const residual = new Set([...agg.executionIds].filter(id => !coveredByAncestors.has(id)));
|
|
1824
2198
|
if (residual.size === 0) compositeResolutions.set(fingerprint, { finalExecutionIds: residual, suppressed: true });
|
|
1825
|
-
else if (residual.size <
|
|
2199
|
+
else if (residual.size < thresholds.minExecutions) compositeResolutions.set(fingerprint, { finalExecutionIds: residual, suppressed: true });
|
|
1826
2200
|
else compositeResolutions.set(fingerprint, { finalExecutionIds: residual, suppressed: false });
|
|
1827
2201
|
}
|
|
1828
2202
|
|
|
@@ -1855,7 +2229,7 @@ export const DETECTORS = [
|
|
|
1855
2229
|
metric: 'executionReuse', value,
|
|
1856
2230
|
format: 'derived', relations, operator: agg.operator, relation: relationDisplay,
|
|
1857
2231
|
executionIds: finalExecutionIds, totalReadBytes,
|
|
1858
|
-
confidence: cachingReuseConfidence(value,
|
|
2232
|
+
confidence: cachingReuseConfidence(value, thresholds.minExecutions),
|
|
1859
2233
|
validationRequired:
|
|
1860
2234
|
'Composite reuse is inferred from a structural plan-shape match (operator + normalized ' +
|
|
1861
2235
|
'join/filter condition + child shapes) across SQL executions; confirm these executions ' +
|
|
@@ -1874,7 +2248,7 @@ export const DETECTORS = [
|
|
|
1874
2248
|
const residualExecutionIds = covered
|
|
1875
2249
|
? [...agg.executionIds].filter(id => !covered.has(id))
|
|
1876
2250
|
: [...agg.executionIds];
|
|
1877
|
-
if (residualExecutionIds.length <
|
|
2251
|
+
if (residualExecutionIds.length < thresholds.minExecutions) continue;
|
|
1878
2252
|
const value = residualExecutionIds.length;
|
|
1879
2253
|
const totalReadBytes = residualExecutionIds.reduce((sum, id) => sum + (agg.executionBytes.get(id) ?? 0), 0);
|
|
1880
2254
|
const recommendation = totalReadBytes >= 128 * MB
|
|
@@ -1886,7 +2260,7 @@ export const DETECTORS = [
|
|
|
1886
2260
|
relation: agg.relation, format: agg.format,
|
|
1887
2261
|
executionIds: residualExecutionIds.sort((a, b) => a - b),
|
|
1888
2262
|
totalReadBytes,
|
|
1889
|
-
confidence: cachingReuseConfidence(value,
|
|
2263
|
+
confidence: cachingReuseConfidence(value, thresholds.minExecutions),
|
|
1890
2264
|
validationRequired:
|
|
1891
2265
|
'Relation-reuse is inferred from the pre-AQE plan scan identity across SQL ' +
|
|
1892
2266
|
'executions; confirm the reads are the same data and cacheable within one ' +
|
|
@@ -1896,15 +2270,18 @@ export const DETECTORS = [
|
|
|
1896
2270
|
}
|
|
1897
2271
|
return out;
|
|
1898
2272
|
},
|
|
1899
|
-
|
|
1900
|
-
|
|
1901
|
-
|
|
2273
|
+
estimate(finding) {
|
|
2274
|
+
const totalReadBytes = (finding.totalReadBytes ) ?? 0;
|
|
2275
|
+
const wasteMs = (totalReadBytes / RE_READ_THROUGHPUT_BPS) * 1000;
|
|
2276
|
+
return costOnly('modeled', { value: wasteMs, unit: 'ms' });
|
|
2277
|
+
},
|
|
2278
|
+
}),
|
|
2279
|
+
defineAppDetector({
|
|
2280
|
+
type: 'jobFailureRate', order: 110, fixEffort: 'code', version: 1,
|
|
2281
|
+
emits: ['jobFailureRate'],
|
|
1902
2282
|
docAnchor: '#bottleneck-job-failure-rate',
|
|
1903
2283
|
thresholds: { infoRate: 0.10, warnRate: 0.30, critRate: 0.50 },
|
|
1904
|
-
detect(
|
|
1905
|
-
|
|
1906
|
-
ctx ,
|
|
1907
|
-
) {
|
|
2284
|
+
detect(ctx, thresholds) {
|
|
1908
2285
|
const { jobs, stages } = ctx;
|
|
1909
2286
|
const all = jobs ? [...jobs.values()] : [];
|
|
1910
2287
|
const completed = all.filter(j => j.result != null);
|
|
@@ -1912,7 +2289,7 @@ export const DETECTORS = [
|
|
|
1912
2289
|
const failedJobList = completed.filter(j => j.succeeded === false);
|
|
1913
2290
|
const failedJobs = failedJobList.length;
|
|
1914
2291
|
const rate = failedJobs / completed.length;
|
|
1915
|
-
if (rate <
|
|
2292
|
+
if (rate < thresholds.infoRate) return null;
|
|
1916
2293
|
let totalTasks = 0, failedTasks = 0;
|
|
1917
2294
|
for (const s of stages.values()) { totalTasks += s.taskCount ?? 0; failedTasks += s.failedTasks ?? 0; }
|
|
1918
2295
|
const taskFailureRate = totalTasks > 0 ? failedTasks / totalTasks : 0;
|
|
@@ -1925,113 +2302,121 @@ export const DETECTORS = [
|
|
|
1925
2302
|
const totalJobs = completed.length;
|
|
1926
2303
|
return {
|
|
1927
2304
|
type: 'jobFailureRate', stageId: null,
|
|
1928
|
-
impactBand: rate >=
|
|
2305
|
+
impactBand: rate >= thresholds.critRate ? 'critical' : rate >= thresholds.warnRate ? 'warning' : 'info',
|
|
1929
2306
|
metric: 'jobFailureRate', value: Math.round(rate * 1000) / 10,
|
|
1930
2307
|
failedJobs, totalJobs, failedTasks, totalTasks, avgJobDurationMs,
|
|
1931
2308
|
taskFailureRate: Math.round(taskFailureRate * 1000) / 10,
|
|
1932
2309
|
recommendation: `${failedJobs} of ${totalJobs} jobs never recovered: inspect the driver log for the failed job(s) and the stage failures that triggered them.`,
|
|
1933
2310
|
};
|
|
1934
2311
|
},
|
|
1935
|
-
|
|
2312
|
+
estimate(finding) {
|
|
2313
|
+
const failedJobs = (finding.failedJobs ) ?? 0;
|
|
2314
|
+
const avgJobDurationMs = (finding.avgJobDurationMs ) ?? 0;
|
|
2315
|
+
const coreHoursIsh = (failedJobs * avgJobDurationMs) / 3.6e6;
|
|
2316
|
+
return costOnly('modeled', { value: coreHoursIsh, unit: 'coreHours' });
|
|
2317
|
+
},
|
|
2318
|
+
}),
|
|
1936
2319
|
// ── Config-sanity entries (scope:'config', inScorecard:false) ────────────────
|
|
1937
|
-
{
|
|
1938
|
-
type: 'configAudit',
|
|
2320
|
+
defineConfigDetector({
|
|
2321
|
+
type: 'configAudit', order: 120, fixEffort: 'config', version: 1, inScorecard: false,
|
|
2322
|
+
emits: ['configAudit'],
|
|
1939
2323
|
docAnchor: '#config-shuffle-service', thresholds: {}, property: 'spark.shuffle.service.enabled',
|
|
1940
|
-
detect(
|
|
1941
|
-
const res =
|
|
2324
|
+
detect(target) {
|
|
2325
|
+
const res = target.app?.resources ?? null;
|
|
1942
2326
|
if (res?.dynamicAllocationEnabled === true && res?.shuffleServiceEnabled === false) {
|
|
1943
2327
|
return {
|
|
1944
2328
|
type: 'configAudit', property: 'spark.shuffle.service.enabled',
|
|
1945
|
-
impactBand: 'warning', metric: 'config',
|
|
2329
|
+
impactBand: 'warning', metric: 'config', valueText: 'false',
|
|
1946
2330
|
recommendation: 'Dynamic allocation is on but the external shuffle service is off: set spark.shuffle.service.enabled=true so shuffle data survives executor removal.',
|
|
1947
2331
|
};
|
|
1948
2332
|
}
|
|
1949
2333
|
return null;
|
|
1950
2334
|
},
|
|
1951
|
-
|
|
1952
|
-
|
|
1953
|
-
|
|
2335
|
+
estimate: noWasteModel,
|
|
2336
|
+
}),
|
|
2337
|
+
defineConfigDetector({
|
|
2338
|
+
type: 'configAudit', order: 121, fixEffort: 'config', version: 1, inScorecard: false,
|
|
2339
|
+
emits: ['configAudit'],
|
|
1954
2340
|
docAnchor: '#config-autoscale-bounds', thresholds: {}, property: 'spark.dynamicAllocation.maxExecutors',
|
|
1955
|
-
detect(
|
|
1956
|
-
const app =
|
|
2341
|
+
detect(target) {
|
|
2342
|
+
const app = target.app; const config = app?.config ?? {}; const res = app?.resources ?? null;
|
|
1957
2343
|
if (res?.dynamicAllocationEnabled !== true) return null;
|
|
1958
2344
|
const minN = config['spark.dynamicAllocation.minExecutors'] != null ? parseInt(config['spark.dynamicAllocation.minExecutors'], 10) : null;
|
|
1959
2345
|
const maxN = config['spark.dynamicAllocation.maxExecutors'] != null ? parseInt(config['spark.dynamicAllocation.maxExecutors'], 10) : null;
|
|
1960
2346
|
if (minN != null && maxN != null && minN > maxN) {
|
|
1961
2347
|
return {
|
|
1962
2348
|
type: 'configAudit', property: 'spark.dynamicAllocation.minExecutors',
|
|
1963
|
-
impactBand: 'critical', metric: 'config',
|
|
2349
|
+
impactBand: 'critical', metric: 'config', valueText: `${minN} > ${maxN}`,
|
|
1964
2350
|
recommendation: `Autoscaling bounds are inverted: spark.dynamicAllocation.minExecutors (${minN}) exceeds maxExecutors (${maxN}). Set min ≤ max.`,
|
|
1965
2351
|
};
|
|
1966
2352
|
}
|
|
1967
2353
|
if (maxN == null) {
|
|
1968
2354
|
return {
|
|
1969
2355
|
type: 'configAudit', property: 'spark.dynamicAllocation.maxExecutors',
|
|
1970
|
-
impactBand: 'info', metric: 'config',
|
|
2356
|
+
impactBand: 'info', metric: 'config', valueText: '(unset)',
|
|
1971
2357
|
recommendation: 'Dynamic allocation is on with no upper bound: set spark.dynamicAllocation.maxExecutors to cap cluster growth.',
|
|
1972
2358
|
};
|
|
1973
2359
|
}
|
|
1974
2360
|
return null;
|
|
1975
2361
|
},
|
|
1976
|
-
|
|
1977
|
-
|
|
1978
|
-
|
|
2362
|
+
estimate: noWasteModel,
|
|
2363
|
+
}),
|
|
2364
|
+
defineConfigDetector({
|
|
2365
|
+
type: 'configAudit', order: 122, fixEffort: 'config', version: 1, inScorecard: false,
|
|
2366
|
+
emits: ['configAudit'],
|
|
1979
2367
|
docAnchor: '#config-serializer', thresholds: {}, property: 'spark.serializer',
|
|
1980
|
-
detect(
|
|
1981
|
-
const app =
|
|
2368
|
+
detect(target) {
|
|
2369
|
+
const app = target.app; const config = app?.config ?? {}; const res = app?.resources ?? null;
|
|
1982
2370
|
if (Object.keys(config).length === 0) return null;
|
|
1983
2371
|
const ser = res?.serializer ?? config['spark.serializer'] ?? null;
|
|
1984
2372
|
const isKryo = typeof ser === 'string' && /kryo/i.test(ser);
|
|
1985
2373
|
if (isKryo) return null;
|
|
1986
2374
|
return {
|
|
1987
2375
|
type: 'configAudit', property: 'spark.serializer',
|
|
1988
|
-
impactBand: 'info', metric: 'config',
|
|
2376
|
+
impactBand: 'info', metric: 'config', valueText: ser ?? '(default JavaSerializer)',
|
|
1989
2377
|
recommendation: `Current serializer is ${ser ?? 'the default JavaSerializer'}: consider spark.serializer=org.apache.spark.serializer.KryoSerializer for faster, smaller buffers.`,
|
|
1990
2378
|
};
|
|
1991
2379
|
},
|
|
1992
|
-
|
|
1993
|
-
|
|
1994
|
-
|
|
2380
|
+
estimate: noWasteModel,
|
|
2381
|
+
}),
|
|
2382
|
+
defineConfigDetector({
|
|
2383
|
+
type: 'configAudit', order: 123, fixEffort: 'config', version: 1, inScorecard: false,
|
|
2384
|
+
emits: ['configAudit'],
|
|
1995
2385
|
docAnchor: '#config-memory-overhead', thresholds: { floorMB: 384, floorPct: 0.1 }, property: 'spark.executor.memoryOverhead',
|
|
1996
|
-
detect(
|
|
1997
|
-
|
|
1998
|
-
ctx ,
|
|
1999
|
-
) {
|
|
2000
|
-
const res = ctx.app?.resources ?? null;
|
|
2386
|
+
detect(target, thresholds) {
|
|
2387
|
+
const res = target.app?.resources ?? null;
|
|
2001
2388
|
const memMB = res?.executor?.memoryMB ?? null;
|
|
2002
2389
|
const ovMB = res?.executor?.memoryOverheadMB ?? null;
|
|
2003
2390
|
if (memMB == null || ovMB == null) return null;
|
|
2004
|
-
const floor = Math.max(
|
|
2391
|
+
const floor = Math.max(thresholds.floorMB, Math.round(memMB * thresholds.floorPct));
|
|
2005
2392
|
if (ovMB >= floor) return null;
|
|
2006
2393
|
return {
|
|
2007
2394
|
type: 'configAudit', property: 'spark.executor.memoryOverhead',
|
|
2008
|
-
impactBand: 'info', metric: 'config',
|
|
2395
|
+
impactBand: 'info', metric: 'config', valueText: `${ovMB} MiB`,
|
|
2009
2396
|
recommendation: `Executor memoryOverhead (${ovMB} MiB) is below Spark's default floor of ${floor} MiB (max of 384 MiB or 10% of executor memory): raise it to avoid off-heap OOM-kills.`,
|
|
2010
2397
|
};
|
|
2011
2398
|
},
|
|
2012
|
-
|
|
2399
|
+
estimate: noWasteModel,
|
|
2400
|
+
}),
|
|
2013
2401
|
// ── Plan-metric entries (scope:'sql') ────────────────────────────────────
|
|
2014
|
-
{
|
|
2015
|
-
type: 'duplicatePlanSubtree',
|
|
2402
|
+
defineSqlDetector({
|
|
2403
|
+
type: 'duplicatePlanSubtree', order: 130, fixEffort: 'code', version: 2,
|
|
2404
|
+
emits: ['duplicatePlanSubtree'],
|
|
2016
2405
|
docAnchor: '#bottleneck-duplicate-plan-subtree',
|
|
2017
2406
|
// stageFloorPct: the 0.5% runtime floor. The claim counts at most each linked stage's own
|
|
2018
2407
|
// task-active time, so a repeat whose stages together lasted less than this share of the run
|
|
2019
2408
|
// graded info: 340 of 545 findings on the 14 real logs. The repeat is still in the plan; the
|
|
2020
2409
|
// floor is why they're dropped. A repeat with no linked stage time is kept.
|
|
2021
2410
|
thresholds: { minSubtreeSize: 3, minOccurrences: 2, stageFloorPct: 0.005 },
|
|
2022
|
-
detect(
|
|
2023
|
-
|
|
2024
|
-
sqlExec ,
|
|
2025
|
-
ctx ,
|
|
2026
|
-
) {
|
|
2411
|
+
detect(sqlExec, ctx, thresholds) {
|
|
2027
2412
|
if (!sqlExec.planTree) return null;
|
|
2028
|
-
const groups = findDuplicateSubtrees(sqlExec.planTree,
|
|
2413
|
+
const groups = findDuplicateSubtrees(sqlExec.planTree, thresholds);
|
|
2029
2414
|
if (groups.length === 0) return null;
|
|
2030
2415
|
const fallbackStageIds = stageIdsForSqlExec(sqlExec.id, ctx.stages);
|
|
2031
2416
|
const executionNodes = [];
|
|
2032
2417
|
walkPlanTree(sqlExec.planTree, (node) => executionNodes.push(node));
|
|
2033
2418
|
const operatorsByStage = operatorCountByStage(executionNodes);
|
|
2034
|
-
const
|
|
2419
|
+
const runMs = appDurationMs(ctx.app);
|
|
2035
2420
|
const findings = groups.map((g) => {
|
|
2036
2421
|
const nodes = [];
|
|
2037
2422
|
for (const n of g.nodes) walkPlanTree(n, (node) => nodes.push(node));
|
|
@@ -2041,7 +2426,7 @@ export const DETECTORS = [
|
|
|
2041
2426
|
const stage = ctx.stages.get(id);
|
|
2042
2427
|
if (stage) stagesMs += Math.max(0, (stage.completedAt ?? 0) - (stage.submittedAt ?? 0));
|
|
2043
2428
|
}
|
|
2044
|
-
if (
|
|
2429
|
+
if (runMs != null && stagesMs > 0 && stagesMs < runMs * thresholds.stageFloorPct) return null;
|
|
2045
2430
|
const occurrencesIdentical = occurrencesHaveIdenticalDetails(g.nodes);
|
|
2046
2431
|
const stageShares = stageOperatorShares(nodes, operatorsByStage);
|
|
2047
2432
|
// resolvePlanTree always sets id; safe downstream of it.
|
|
@@ -2058,7 +2443,7 @@ export const DETECTORS = [
|
|
|
2058
2443
|
metric: 'subtreeOccurrences', value: g.occurrences,
|
|
2059
2444
|
rootName: g.rootName, subtreeSize: g.subtreeSize, sampleRelation: g.sampleRelation,
|
|
2060
2445
|
groupIndex: g.groupIndex,
|
|
2061
|
-
confidence: occurrencesIdentical ? duplicateSubtreeConfidence(g.subtreeSize, g.occurrences,
|
|
2446
|
+
confidence: occurrencesIdentical ? duplicateSubtreeConfidence(g.subtreeSize, g.occurrences, thresholds) : 'low',
|
|
2062
2447
|
validationRequired: 'Duplicate-subtree matching compares operator names and metric names only, not literal values or expr IDs: confirm the repeated work is real in the Spark SQL plan tab before acting.',
|
|
2063
2448
|
recommendation: (g.isExchangeRoot
|
|
2064
2449
|
? `A ${g.subtreeSize}-node subtree rooted at ${pathBasename(g.rootName)} repeats ${g.occurrences}x in this plan${touching}: this looks like a possible missed exchange reuse; check whether the same shuffle could be computed once and reused.`
|
|
@@ -2067,18 +2452,50 @@ export const DETECTORS = [
|
|
|
2067
2452
|
}).filter((f) => f !== null);
|
|
2068
2453
|
return findings.length > 0 ? findings : null;
|
|
2069
2454
|
},
|
|
2070
|
-
|
|
2071
|
-
|
|
2072
|
-
|
|
2455
|
+
estimate(finding, ctx) {
|
|
2456
|
+
const stageIds = finding.stageIds ;
|
|
2457
|
+
if (!stageIds || stageIds.length === 0) return null;
|
|
2458
|
+
// detect() reports subtreeOccurrences >= 2. Only repeats past the first are redundant:
|
|
2459
|
+
// computing the subtree once is real work, so waste is (occurrences-1)/occurrences of the stages' time.
|
|
2460
|
+
const occurrences = typeof finding.value === 'number' ? finding.value : 0;
|
|
2461
|
+
// Defensive only: minOccurrences guarantees occurrences >= 2 on real data; a malformed-value fallback.
|
|
2462
|
+
if (occurrences < 2) return costOnly('none');
|
|
2463
|
+
// Same-shaped repeats whose details differ compute different data: nothing is known to be
|
|
2464
|
+
// recomputed, so there is no time to claim.
|
|
2465
|
+
if (finding.occurrencesIdentical === false) return costOnly('none');
|
|
2466
|
+
// Each stage contributes the share of its operators inside the repeated subtree: a stage it
|
|
2467
|
+
// shares with other operators (the consuming join, the join's other side) isn't all its
|
|
2468
|
+
// time, and claiming whole stages let sibling groups claim the same stage twice. Findings
|
|
2469
|
+
// built without the field (hand-made fixtures) count every linked stage whole.
|
|
2470
|
+
const shares = (finding.stageShares ?? null) ;
|
|
2471
|
+
const redundantFraction = (occurrences - 1) / occurrences;
|
|
2472
|
+
const wasteMsByStage = new Map ();
|
|
2473
|
+
for (const id of stageIds) {
|
|
2474
|
+
const s = ctx.stages.get(id);
|
|
2475
|
+
const share = shares ? (shares[id] ?? 0) : 1;
|
|
2476
|
+
if (s && share > 0) {
|
|
2477
|
+
// Time with tasks running, not submit-to-complete: a stage left waiting for cores
|
|
2478
|
+
// (2491s open, 60s of tasks on a real log) isn't recomputing anything while it waits.
|
|
2479
|
+
const activeMs = s.taskActiveMs ?? Math.max(0, (s.completedAt ?? 0) - (s.submittedAt ?? 0));
|
|
2480
|
+
wasteMsByStage.set(id, activeMs * share * redundantFraction);
|
|
2481
|
+
}
|
|
2482
|
+
}
|
|
2483
|
+
// No operator of the subtree ran in a known stage: no time to attribute.
|
|
2484
|
+
if (wasteMsByStage.size === 0) return costOnly('none');
|
|
2485
|
+
const totalWasteMs = [...wasteMsByStage.values()].reduce((sum, ms) => sum + ms, 0);
|
|
2486
|
+
const rawWaste = { value: totalWasteMs, unit: 'ms' };
|
|
2487
|
+
return multiStageImpact([...wasteMsByStage.keys()], wasteMsByStage, ctx, 'measured', rawWaste)
|
|
2488
|
+
?? costOnly('measured', rawWaste);
|
|
2489
|
+
},
|
|
2490
|
+
}),
|
|
2491
|
+
defineSqlDetector({
|
|
2492
|
+
type: 'smallFiles', order: 131, fixEffort: 'config', version: 2,
|
|
2493
|
+
emits: ['smallFiles'],
|
|
2073
2494
|
docAnchor: '#bottleneck-small-files',
|
|
2074
2495
|
thresholds: { minFiles: 100, maxAvgFileSizeMB: 3 },
|
|
2075
|
-
detect(
|
|
2076
|
-
|
|
2077
|
-
sqlExec ,
|
|
2078
|
-
ctx ,
|
|
2079
|
-
) {
|
|
2496
|
+
detect(sqlExec, ctx, thresholds) {
|
|
2080
2497
|
if (!sqlExec.planTree) return null;
|
|
2081
|
-
const { minFiles, maxAvgFileSizeMB } =
|
|
2498
|
+
const { minFiles, maxAvgFileSizeMB } = thresholds;
|
|
2082
2499
|
|
|
2083
2500
|
const hits = [];
|
|
2084
2501
|
walkPlanTree(sqlExec.planTree, (node) => {
|
|
@@ -2113,27 +2530,36 @@ export const DETECTORS = [
|
|
|
2113
2530
|
};
|
|
2114
2531
|
});
|
|
2115
2532
|
},
|
|
2116
|
-
|
|
2117
|
-
|
|
2533
|
+
estimate(finding, ctx) {
|
|
2534
|
+
const fileMs = ((finding.fileCount ) ?? 0) * FILE_OPEN_OVERHEAD_MS;
|
|
2535
|
+
// A read's files are opened by the scan's tasks, in parallel: spread the per-file cost over
|
|
2536
|
+
// the most tasks its stages ran at once (91344 files x 10ms is 913s, claimed against a 117s
|
|
2537
|
+
// stage that ran 314 tasks at once). A write keeps the serial sum: the job commit moves each
|
|
2538
|
+
// output file on the driver, one after another.
|
|
2539
|
+
const stageIds = finding.stageIds ;
|
|
2540
|
+
let slots = 1;
|
|
2541
|
+
if (finding.direction === 'read') {
|
|
2542
|
+
for (const id of stageIds ?? []) slots = Math.max(slots, ctx.stages.get(id)?.peakConcurrentTasks ?? 1);
|
|
2543
|
+
}
|
|
2544
|
+
return stageMappableWasteOrCostOnly(fileMs / slots, stageIds, ctx);
|
|
2545
|
+
},
|
|
2546
|
+
}),
|
|
2547
|
+
defineSqlDetector({
|
|
2118
2548
|
// Entry-level type is an identifier only; it never appears on an emitted finding. Findings
|
|
2119
2549
|
// carry 'underBroadcast'/'overBroadcast' since one shared plan-walk covers both
|
|
2120
2550
|
// opposite-direction rules (JoinToBroadcastAlert / BroadcastTooLargeAlert).
|
|
2121
|
-
type: 'broadcastSizing',
|
|
2551
|
+
type: 'broadcastSizing', order: 132, fixEffort: 'config', version: 2,
|
|
2552
|
+
// Listed over-first: the two share order 132, and this list order is their display tie-break.
|
|
2553
|
+
emits: ['overBroadcast', 'underBroadcast'],
|
|
2122
2554
|
docAnchor: '#bottleneck-broadcast-sizing',
|
|
2123
2555
|
thresholds: {
|
|
2124
2556
|
broadcastTiers: [10 * MB, 100 * MB, GB, 5 * GB],
|
|
2125
2557
|
comparisonTiers: [10 * GB, 300 * GB, TB],
|
|
2126
2558
|
overBroadcastBytes: GB,
|
|
2127
2559
|
},
|
|
2128
|
-
detect(
|
|
2129
|
-
|
|
2130
|
-
|
|
2131
|
-
|
|
2132
|
-
sqlExec ,
|
|
2133
|
-
ctx ,
|
|
2134
|
-
) {
|
|
2560
|
+
detect(sqlExec, ctx, thresholds) {
|
|
2135
2561
|
if (!sqlExec.planTree) return null;
|
|
2136
|
-
const { broadcastTiers, comparisonTiers, overBroadcastBytes } =
|
|
2562
|
+
const { broadcastTiers, comparisonTiers, overBroadcastBytes } = thresholds;
|
|
2137
2563
|
const fallbackStageIds = stageIdsForSqlExec(sqlExec.id, ctx.stages);
|
|
2138
2564
|
const out = [];
|
|
2139
2565
|
walkPlanTree(sqlExec.planTree, (node) => {
|
|
@@ -2174,12 +2600,60 @@ export const DETECTORS = [
|
|
|
2174
2600
|
// resolvePlanTree always sets id; safe downstream of it.
|
|
2175
2601
|
planNodeIds: [node.id ].filter(Boolean),
|
|
2176
2602
|
impactBand: 'warning', metric: 'broadcastBytes', value: m.value,
|
|
2177
|
-
recommendation: `This broadcast (${formatBytes(m.value)}) exceeds the
|
|
2603
|
+
recommendation: `This broadcast (${formatBytes(m.value)}) exceeds the ${binaryThresholdLabel(overBroadcastBytes)} threshold: check for a misapplied broadcast hint or a misconfigured spark.sql.autoBroadcastJoinThreshold.`,
|
|
2178
2604
|
});
|
|
2179
2605
|
}
|
|
2180
2606
|
}
|
|
2181
2607
|
});
|
|
2182
2608
|
return out.length ? out : null;
|
|
2183
2609
|
},
|
|
2184
|
-
|
|
2185
|
-
|
|
2610
|
+
estimate(finding, ctx) {
|
|
2611
|
+
// Both finding types carry bytes as `value`: overBroadcast's broadcastBytes, underBroadcast's
|
|
2612
|
+
// smallerSideBytes (the smaller join side), each priced as one broadcast transfer.
|
|
2613
|
+
const wasteMs = (((finding.value ) ?? 0) / BROADCAST_BANDWIDTH_BPS) * 1000;
|
|
2614
|
+
return stageMappableWasteOrCostOnly(wasteMs, finding.stageIds , ctx);
|
|
2615
|
+
},
|
|
2616
|
+
}),
|
|
2617
|
+
] ;
|
|
2618
|
+
|
|
2619
|
+
/** The entry that emits each finding type: the one whose estimate prices it and whose thresholds
|
|
2620
|
+
* and order describe it. Several entries can emit one type (the four configAudit audits), and the
|
|
2621
|
+
* first declared wins. */
|
|
2622
|
+
export const ENTRY_BY_TYPE = (() => {
|
|
2623
|
+
const byType = new Map ();
|
|
2624
|
+
for (const entry of DETECTORS ) {
|
|
2625
|
+
for (const type of entry.emits) if (!byType.has(type)) byType.set(type, entry);
|
|
2626
|
+
}
|
|
2627
|
+
return byType;
|
|
2628
|
+
})();
|
|
2629
|
+
|
|
2630
|
+
/** A `DETECTORS` entry's own `type`: every emitted finding type, plus broadcastSizing. */
|
|
2631
|
+
|
|
2632
|
+
|
|
2633
|
+
/** Every finding `type` a detector can emit, from the entries' `emits` lists. */
|
|
2634
|
+
|
|
2635
|
+
|
|
2636
|
+
|
|
2637
|
+
|
|
2638
|
+
|
|
2639
|
+
|
|
2640
|
+
/** The `thresholds` of the entry (or entries, for configAudit) that emit finding type `T`. */
|
|
2641
|
+
|
|
2642
|
+
|
|
2643
|
+
// `emits` already only names Finding members (Detector.emits); this makes the reverse hold too, so a
|
|
2644
|
+
// Finding member no detector emits, or a detector whose type has no Finding member, fails to compile.
|
|
2645
|
+
|
|
2646
|
+
|
|
2647
|
+
|
|
2648
|
+
|
|
2649
|
+
// A `suppressedBy` that names no entry would never suppress anything; this makes it a compile error.
|
|
2650
|
+
// An entry without one infers the bare `string` constraint, which contributes nothing here.
|
|
2651
|
+
|
|
2652
|
+
|
|
2653
|
+
|
|
2654
|
+
|
|
2655
|
+
/** Per-detector threshold overrides, keyed by entry `type`: each value a partial of that entry's
|
|
2656
|
+
* own thresholds. Built by threshold-overrides.ts from a user's config file. */
|
|
2657
|
+
|
|
2658
|
+
|
|
2659
|
+
|