sparkforensics-mcp 0.3.0 → 0.5.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/bin/sparkforensics-mcp.mjs +41 -10
- package/package.json +1 -1
- package/vendor-core/allocation.js +106 -0
- package/vendor-core/analyzer.js +168 -60
- package/vendor-core/check-coverage.js +88 -0
- package/vendor-core/cli/budgets.js +54 -27
- package/vendor-core/cli/collect-run.js +84 -32
- package/vendor-core/cli/regression-budgets.js +83 -0
- package/vendor-core/cli/threshold-config.js +28 -0
- package/vendor-core/comparison-verdict.js +177 -0
- package/vendor-core/core-source-hash.txt +1 -0
- package/vendor-core/core-usage-locality.js +56 -2
- package/vendor-core/detector-docs.js +58 -0
- package/vendor-core/detectors.js +1094 -500
- package/vendor-core/docs-config.js +0 -36
- package/vendor-core/docs-content/chapters/nav-index.json +31 -0
- package/vendor-core/docs-content/detection/cache.md +3 -2
- package/vendor-core/docs-content/detection/cfg.md +9 -8
- package/vendor-core/docs-content/detection/chrn.md +1 -2
- package/vendor-core/docs-content/detection/cold.md +4 -2
- package/vendor-core/docs-content/detection/cstor.md +9 -0
- package/vendor-core/docs-content/detection/fail.md +3 -2
- package/vendor-core/docs-content/detection/gc.md +3 -2
- package/vendor-core/docs-content/detection/host.md +2 -1
- package/vendor-core/docs-content/detection/local.md +1 -1
- package/vendor-core/docs-content/detection/mem.md +5 -2
- package/vendor-core/docs-content/detection/plan.md +2 -1
- package/vendor-core/docs-content/detection/sfail.md +2 -1
- package/vendor-core/docs-content/detection/shape.md +5 -4
- package/vendor-core/docs-content/detection/skew.md +3 -1
- package/vendor-core/docs-content/detection/slow.md +2 -2
- package/vendor-core/docs-content/detection/spec.md +2 -3
- package/vendor-core/docs-content/detection/spill.md +1 -1
- package/vendor-core/docs-site-config.js +3 -0
- package/vendor-core/effective-conf.js +107 -0
- package/vendor-core/efficiency-model.js +8 -6
- package/vendor-core/event-handlers.js +321 -44
- package/vendor-core/event-schemas.js +23 -0
- package/vendor-core/evidence-report.js +432 -115
- package/vendor-core/export-data.js +79 -6
- package/vendor-core/finding-action-label.js +9 -88
- package/vendor-core/finding-filter-predicate.js +9 -0
- package/vendor-core/finding-generic-recommendation.js +26 -105
- package/vendor-core/finding-names.js +28 -45
- package/vendor-core/finding-presentation.js +368 -0
- package/vendor-core/finding-tag-help.js +110 -0
- package/vendor-core/finding-types.js +373 -0
- package/vendor-core/findings-of-type.js +11 -0
- package/vendor-core/format-utils.js +96 -30
- package/vendor-core/html-export.js +51 -0
- package/vendor-core/impact-band.js +21 -8
- package/vendor-core/impact-estimator.js +25 -520
- package/vendor-core/impact-format.js +115 -0
- package/vendor-core/impact-model.js +197 -0
- package/vendor-core/ingest.js +6 -2
- package/vendor-core/intervals.js +13 -0
- package/vendor-core/list-runs.js +7 -5
- package/vendor-core/load-vendored.js +70 -5
- package/vendor-core/mcp-server-factory.js +14 -10
- package/vendor-core/mcp-tools.js +105 -45
- package/vendor-core/model-assembler.js +35 -1
- package/vendor-core/occupancy.js +1 -1
- package/vendor-core/parser-worker.js +2 -2
- package/vendor-core/plan-graph-model.js +3 -2
- package/vendor-core/plan-node-detail.js +1 -1
- package/vendor-core/proxy.js +3 -1
- package/vendor-core/python-stage.js +25 -0
- package/vendor-core/recommendation-rollup.js +70 -3
- package/vendor-core/redact.js +96 -37
- package/vendor-core/remediation.js +20 -0
- package/vendor-core/run-comparison.js +73 -29
- package/vendor-core/run-interpretation.js +291 -0
- package/vendor-core/run-metrics.js +198 -0
- package/vendor-core/run-outcome.js +74 -0
- package/vendor-core/run-payload.js +17 -0
- package/vendor-core/run-shape.js +40 -0
- package/vendor-core/run-totals.js +24 -0
- package/vendor-core/run-verdict.js +352 -0
- package/vendor-core/scaling-sim.js +4 -5
- package/vendor-core/scorecard-estimates.js +63 -0
- package/vendor-core/session-snapshot.js +7 -0
- package/vendor-core/shs-schemas.js +2 -2
- package/vendor-core/spark-memory.js +17 -0
- package/vendor-core/sql-stages.js +11 -0
- package/vendor-core/stage-plan-nodes.js +18 -0
- package/vendor-core/stage-quantiles.js +6 -0
- package/vendor-core/threshold-overrides.js +160 -0
- package/vendor-core/threshold-summary.js +11 -33
- package/vendor-core/types.js +54 -42
- package/vendor-core/wall-clock.js +1 -12
- package/vendor-core/wasted-core-hours.js +12 -9
- package/vendor-core/write-targets.js +312 -0
package/vendor-core/detectors.js
CHANGED
|
@@ -1,18 +1,38 @@
|
|
|
1
|
-
import { pathBasename, formatBytes,
|
|
1
|
+
import { pathBasename, formatBytes, IMPACT_BAND_ORDER } from './format-utils.js';
|
|
2
|
+
import { shareLabel } from './finding-presentation.js';
|
|
2
3
|
import { scanRelationId } from './plan-summary.js';
|
|
3
4
|
import { computePeakConcurrentCores, computePeakConcurrentExecutorCount } from './core-count.js';
|
|
4
5
|
import { walkPlanTree } from './plan-tree-walk.js';
|
|
5
6
|
import { computeCoreLocalityRatio } from './core-locality-ratio.js';
|
|
6
|
-
import {
|
|
7
|
+
import { tailRecoveryMs, tailRemovedWorkMs, stragglerFixLongestTaskMs, } from './occupancy.js';
|
|
8
|
+
import { IMPACT_FLOOR_PCT_WARN, IMPACT_FLOOR_PCT_CRIT, appDurationMs } from './impact-band.js';
|
|
9
|
+
import {
|
|
10
|
+
BROADCAST_BANDWIDTH_BPS, EXECUTOR_STARTUP_OVERHEAD_MS, FILE_OPEN_OVERHEAD_MS, IDEAL_BYTES_PER_PARTITION_TASK,
|
|
11
|
+
NETWORK_FETCH_PENALTY_MS, RE_READ_THROUGHPUT_BPS, SHUFFLE_THROUGHPUT_BPS, SPILL_IO_THROUGHPUT_BPS, TAIL_CLAIM,
|
|
12
|
+
TASK_SCHEDULING_OVERHEAD_MS, costOnly, fetchWaitWallClockMs, measuredTaskOverhead, multiStageImpact, noWasteModel,
|
|
13
|
+
retryWallClockMs, singleStageImpact, stageIoParallelism, stageMappableWasteOrCostOnly, tasksMostlyIdle,
|
|
14
|
+
|
|
15
|
+
} from './impact-model.js';
|
|
7
16
|
import { isExchangeNode, isBroadcastExchangeNode } from './plan-node-detail.js';
|
|
17
|
+
import { totalExecutorCpuMs } from './run-totals.js';
|
|
18
|
+
import { DUPLICATE_SUBTREE_DIFFERING_NOTE, duplicateSubtreeDetail, SLOW_HOST_DIMENSION_LABEL } from './finding-generic-recommendation.js';
|
|
19
|
+
import { stageIdsForSqlExec } from './sql-stages.js';
|
|
8
20
|
import { cyrb53 } from './string-hash.js';
|
|
21
|
+
import { decreaseConf, increaseConf, setConf } from './remediation.js';
|
|
9
22
|
import { MAX_FAILURE_GROUPS, describeTaskFailure, } from './task-failure.js';
|
|
10
|
-
|
|
23
|
+
|
|
24
|
+
|
|
11
25
|
|
|
12
26
|
const MB = 1024 * 1024;
|
|
13
27
|
const GB = 1024 * MB;
|
|
14
28
|
const TB = 1024 * GB;
|
|
15
29
|
|
|
30
|
+
// Labels a byte threshold in this file's binary units, so a 1 GiB default reads '1 GB'.
|
|
31
|
+
function binaryThresholdLabel(bytes ) {
|
|
32
|
+
const [unit, size] = ([['GB', GB], ['MB', MB], ['KB', 1024]] ).find(([, u]) => bytes >= u) ?? ['bytes', 1];
|
|
33
|
+
return `${Math.round(bytes / size * 10) / 10} ${unit}`;
|
|
34
|
+
}
|
|
35
|
+
|
|
16
36
|
// Local runtime shapes.
|
|
17
37
|
//
|
|
18
38
|
// types.ts's Stage/SqlExecution/SparkAppInfo describe the posted AppModel surface for the view
|
|
@@ -27,16 +47,7 @@ const TB = 1024 * GB;
|
|
|
27
47
|
|
|
28
48
|
|
|
29
49
|
|
|
30
|
-
|
|
31
|
-
|
|
32
|
-
|
|
33
|
-
|
|
34
|
-
|
|
35
|
-
|
|
36
|
-
|
|
37
|
-
|
|
38
|
-
|
|
39
|
-
|
|
50
|
+
|
|
40
51
|
// Per-executor snapshot from a StageExecutorMetrics event: a loose bag of Spark's
|
|
41
52
|
// ExecutorMetrics field names, only a few of which any detector reads.
|
|
42
53
|
|
|
@@ -103,6 +114,7 @@ const TB = 1024 * GB;
|
|
|
103
114
|
|
|
104
115
|
|
|
105
116
|
|
|
117
|
+
|
|
106
118
|
|
|
107
119
|
|
|
108
120
|
|
|
@@ -115,7 +127,10 @@ const TB = 1024 * GB;
|
|
|
115
127
|
|
|
116
128
|
|
|
117
129
|
|
|
130
|
+
|
|
118
131
|
|
|
132
|
+
|
|
133
|
+
|
|
119
134
|
|
|
120
135
|
|
|
121
136
|
|
|
@@ -128,25 +143,22 @@ const TB = 1024 * GB;
|
|
|
128
143
|
|
|
129
144
|
|
|
130
145
|
|
|
131
|
-
// The full context analyze() passes as every stage/sql detect()'s second arg, and as the
|
|
132
|
-
// arg for 'app' scope. 'config' scope gets a narrower `{ app }` (auditConfig)
|
|
146
|
+
// The full context analyze() passes as every stage/sql detect()'s second arg, and as the first
|
|
147
|
+
// arg for 'app' scope. 'config' scope gets a narrower `{ app }` (auditConfig) instead.
|
|
133
148
|
//
|
|
134
|
-
// `app` is
|
|
135
|
-
//
|
|
136
|
-
// malformed/incomplete logs. Despite the type annotation, app may be null in edge cases, and
|
|
137
|
-
// these detectors handle it gracefully. auditConfig's config-scope target keeps `app` nullable
|
|
138
|
-
// instead.
|
|
149
|
+
// `app` is nullable as on AppModel.app: a malformed or cut-short log may have no app record, so
|
|
150
|
+
// every app-reading detector guards it. `runAggregates.busyCoreMs` is optional for the same reason.
|
|
139
151
|
|
|
140
|
-
|
|
152
|
+
|
|
141
153
|
|
|
142
154
|
|
|
143
155
|
|
|
144
156
|
|
|
145
157
|
|
|
146
|
-
|
|
147
|
-
|
|
158
|
+
|
|
148
159
|
|
|
149
|
-
|
|
160
|
+
|
|
161
|
+
|
|
150
162
|
|
|
151
163
|
|
|
152
164
|
// auditConfig(app) calls every config-scope detect({ app }) with whatever appModel.app is
|
|
@@ -190,18 +202,6 @@ function pickDominantReason(reasons )
|
|
|
190
202
|
return [...reasons].sort((a, b) => b.count - a.count)[0].reason;
|
|
191
203
|
}
|
|
192
204
|
|
|
193
|
-
// Shared by every scope:'sql' detector. sql.get(id).stageIds is always empty (parser-worker
|
|
194
|
-
// never populates it), so stage linkage is derived from each stage's own sqlExecutionId.
|
|
195
|
-
// Minimal param shape (not DetectorStage): external callers pass Map<StageId, Stage>.
|
|
196
|
-
export function stageIdsForSqlExec(
|
|
197
|
-
executionId ,
|
|
198
|
-
stages ,
|
|
199
|
-
) {
|
|
200
|
-
const out = [];
|
|
201
|
-
for (const s of stages.values()) if (s.sqlExecutionId === executionId) out.push(s.id);
|
|
202
|
-
return out;
|
|
203
|
-
}
|
|
204
|
-
|
|
205
205
|
// Shared by the three Plan Advisor detectors: union the given nodes' own stageIds, or fall back
|
|
206
206
|
// to the whole execution's stage set when none have coverage (never a partial blend). `fallback`
|
|
207
207
|
// is a param so callers compute stageIdsForSqlExec once per detect() and reuse it per finding.
|
|
@@ -550,15 +550,19 @@ function maxMedianRatio(
|
|
|
550
550
|
}
|
|
551
551
|
|
|
552
552
|
|
|
553
|
-
|
|
553
|
+
|
|
554
554
|
|
|
555
|
-
|
|
556
|
-
|
|
555
|
+
|
|
556
|
+
|
|
557
|
+
|
|
557
558
|
|
|
559
|
+
|
|
560
|
+
|
|
558
561
|
|
|
559
562
|
|
|
560
563
|
// Machine-readable detector metadata for the evidence report (no `detect` closure), so a
|
|
561
|
-
// portable report records which detector + thresholds produced each finding.
|
|
564
|
+
// portable report records which detector + thresholds produced each finding. These are the
|
|
565
|
+
// defaults; tunedDetectorCatalog() (threshold-overrides.ts) is the same rows under overrides.
|
|
562
566
|
export function detectorCatalog() {
|
|
563
567
|
return DETECTORS.map((d) => ({
|
|
564
568
|
type: d.type,
|
|
@@ -587,49 +591,146 @@ export function computeSkewRatio(
|
|
|
587
591
|
|
|
588
592
|
// Absolute-magnitude floor as a % of app runtime, not a fixed ms constant: a skew/straggler
|
|
589
593
|
// ratio on a few ms is noise in an hours-long run but real in a seconds-long one; a fixed-ms
|
|
590
|
-
// floor can't scale. Used by skew/straggler, gated
|
|
591
|
-
//
|
|
594
|
+
// floor can't scale. Used by skew/straggler, gated on the same occupancy-clipped tail claim their
|
|
595
|
+
// estimate() displays as savings (tailClaimImpact).
|
|
592
596
|
// NOT SOURCED: floor percentages are our own noise floor, unvalidated.
|
|
593
|
-
function computeAppDurationMs(ctx ) {
|
|
594
|
-
const app = ctx?.app;
|
|
595
|
-
if (app?.startTime == null || app?.endTime == null) return null;
|
|
596
|
-
const durationMs = app.endTime - app.startTime;
|
|
597
|
-
return durationMs > 0 ? durationMs : null;
|
|
598
|
-
}
|
|
599
|
-
|
|
600
597
|
// Unknown app timing never suppresses a finding; it just skips the floor gate.
|
|
601
|
-
function meetsRuntimeFloor(wasteMs ,
|
|
602
|
-
return
|
|
598
|
+
function meetsRuntimeFloor(wasteMs , runMs , floorPct ) {
|
|
599
|
+
return runMs == null || wasteMs >= runMs * floorPct;
|
|
603
600
|
}
|
|
604
601
|
|
|
605
602
|
// A stage that ran for less than floorPct of the run (a known duration): an estimate clipped to
|
|
606
603
|
// the stage can't reach floorPct, so every finding there grades info. A zero-length stage (no
|
|
607
604
|
// submission time on older Spark) is not skipped: it gets no estimate and keeps its own band.
|
|
608
|
-
function stageBelowRuntimeFloor(stage , ctx
|
|
605
|
+
function stageBelowRuntimeFloor(stage , ctx , floorPct ) {
|
|
609
606
|
const stageDurationMs = (stage.completedAt ?? 0) - (stage.submittedAt ?? 0);
|
|
610
|
-
const
|
|
611
|
-
return
|
|
607
|
+
const runMs = appDurationMs(ctx.app);
|
|
608
|
+
return runMs != null && stageDurationMs > 0 && stageDurationMs < runMs * floorPct;
|
|
609
|
+
}
|
|
610
|
+
|
|
611
|
+
// What a skew or straggler fix claims off its stage: the wall-clock its slow tail costs, the task
|
|
612
|
+
// time the fix removes, and the longest task it leaves. One figure, read twice: detect() gates its
|
|
613
|
+
// runtime floor on the claim's clipped estimate and estimate() reports that same estimate as the
|
|
614
|
+
// savings, so the firing floor and the displayed figure can't disagree.
|
|
615
|
+
|
|
616
|
+
|
|
617
|
+
|
|
618
|
+
|
|
619
|
+
|
|
620
|
+
|
|
621
|
+
// singleDelta: the slowest task's own excess, the fallback when the stage has no task replay.
|
|
622
|
+
function tailClaim(stage , singleDelta , longestTaskAfterFixMs ) {
|
|
623
|
+
return {
|
|
624
|
+
wasteMs: tailRecoveryMs(stage, singleDelta),
|
|
625
|
+
removedCoreWorkMs: tailRemovedWorkMs(stage, singleDelta),
|
|
626
|
+
longestTaskAfterFixMs,
|
|
627
|
+
};
|
|
628
|
+
}
|
|
629
|
+
|
|
630
|
+
// skew's delta is the task computeSkewRatio's metric sampled (P95 or max) over the median. Fixing
|
|
631
|
+
// the skew still waits on the longest task it leaves, as for straggler.
|
|
632
|
+
function skewTailClaim(stage , usesP95Branch ) {
|
|
633
|
+
const p50 = stage.taskDurationP50 ?? 0;
|
|
634
|
+
const singleDelta = Math.max(0, usesP95Branch ? (stage.taskDurationP95 ?? 0) - p50 : (stage.taskDurationMax ?? 0) - p50);
|
|
635
|
+
return tailClaim(stage, singleDelta, stragglerFixLongestTaskMs(stage));
|
|
636
|
+
}
|
|
637
|
+
|
|
638
|
+
// straggler's delta is the slowest task over the longest one the fix leaves.
|
|
639
|
+
function stragglerTailClaim(stage ) {
|
|
640
|
+
const longestTaskAfterFixMs = stragglerFixLongestTaskMs(stage);
|
|
641
|
+
return tailClaim(stage, Math.max(0, (stage.taskDurationMax ?? 0) - longestTaskAfterFixMs), longestTaskAfterFixMs);
|
|
642
|
+
}
|
|
643
|
+
|
|
644
|
+
// A tail claim shortens the stage's longest task, hence TAIL_CLAIM (see occupancy.ts). Its core
|
|
645
|
+
// time is the measured task time the fix removes (removedCoreWorkMs).
|
|
646
|
+
function tailClaimImpact(claim , stageId , ctx ) {
|
|
647
|
+
const estimate = singleStageImpact(claim.wasteMs, stageId, ctx, 'measured', { value: claim.wasteMs, unit: 'ms' },
|
|
648
|
+
{ ...TAIL_CLAIM, removedCoreWorkMs: claim.removedCoreWorkMs, longestTaskAfterFixMs: claim.longestTaskAfterFixMs });
|
|
649
|
+
return { ...estimate, coreTimeMs: { low: claim.removedCoreWorkMs, high: claim.removedCoreWorkMs } };
|
|
650
|
+
}
|
|
651
|
+
|
|
652
|
+
// The figure a runtime floor checks: the claim's recoverable wall-clock, not a delta a physical
|
|
653
|
+
// floor leaves unrecoverable. Falls back to the raw claim when occupancy data is unavailable.
|
|
654
|
+
function tailClaimFloorMs(claim , stageId , ctx ) {
|
|
655
|
+
return tailClaimImpact(claim, stageId, ctx.impact).wallClock?.high ?? claim.wasteMs;
|
|
656
|
+
}
|
|
657
|
+
|
|
658
|
+
// A setting is only a fix when the run's logged conf doesn't already have it: a run that set it
|
|
659
|
+
// needs another remedy. Only explicitly logged properties count; Spark's unlogged version
|
|
660
|
+
// defaults are not modeled. Booleans compare case-insensitively, as Spark parses them.
|
|
661
|
+
function loggedAs(app , key , suggested ) {
|
|
662
|
+
const logged = app?.config?.[key]?.trim();
|
|
663
|
+
return typeof suggested === 'boolean'
|
|
664
|
+
? logged?.toLowerCase() === String(suggested)
|
|
665
|
+
: logged === suggested;
|
|
666
|
+
}
|
|
667
|
+
|
|
668
|
+
function setConfUnlessLogged(app , key , suggested ) {
|
|
669
|
+
return loggedAs(app, key, suggested) ? [] : [setConf(key, suggested)];
|
|
670
|
+
}
|
|
671
|
+
|
|
672
|
+
// A switch's fix worded for the run's logged conf: `recommend` names the property while the run
|
|
673
|
+
// doesn't have it, `alreadyOn` points at the remedy left once it does, with no remediation.
|
|
674
|
+
function switchFix(on , key , suggested , recommend , alreadyOn ) {
|
|
675
|
+
return on ? { text: alreadyOn, remediation: [] } : { text: recommend, remediation: [setConf(key, suggested)] };
|
|
676
|
+
}
|
|
677
|
+
|
|
678
|
+
function skewJoinFix(app ) {
|
|
679
|
+
const key = 'spark.sql.adaptive.skewJoin.enabled';
|
|
680
|
+
const remedy = 'salt the key or repartition on a better key';
|
|
681
|
+
if (loggedAs(app, 'spark.sql.adaptive.enabled', false)) {
|
|
682
|
+
return {
|
|
683
|
+
text: `AQE is off, so enable it (spark.sql.adaptive.enabled) for skew-join handling to apply; otherwise ${remedy}`,
|
|
684
|
+
remediation: [setConf('spark.sql.adaptive.enabled', true), ...setConfUnlessLogged(app, key, true)],
|
|
685
|
+
};
|
|
686
|
+
}
|
|
687
|
+
return switchFix(loggedAs(app, key, true), key, true,
|
|
688
|
+
`for join-driven skew, enable AQE skew-join handling (${key}); otherwise ${remedy}`,
|
|
689
|
+
`AQE skew-join handling is already on, so ${remedy}`);
|
|
690
|
+
}
|
|
691
|
+
|
|
692
|
+
// The resources flag is read from the same property as the logged conf.
|
|
693
|
+
function dynamicAllocationFix(app , recommend , alreadyOn ) {
|
|
694
|
+
const key = 'spark.dynamicAllocation.enabled';
|
|
695
|
+
return switchFix(app.resources?.dynamicAllocationEnabled === true || loggedAs(app, key, true), key, true, recommend, alreadyOn);
|
|
696
|
+
}
|
|
697
|
+
|
|
698
|
+
// The run's logged spark.sql.shuffle.partitions as a count, or null when unlogged or not a count.
|
|
699
|
+
function loggedShufflePartitions(app ) {
|
|
700
|
+
const logged = app?.config?.['spark.sql.shuffle.partitions']?.trim();
|
|
701
|
+
return logged != null && /^\d+$/.test(logged) ? Number(logged) : null;
|
|
702
|
+
}
|
|
703
|
+
|
|
704
|
+
// lowShuffleParallelism's fix. The partition count that brings each shuffle partition down to the
|
|
705
|
+
// ideal size, as the estimate models it, is per stage; the property is job-wide. A logged value at
|
|
706
|
+
// or above it means the property is not what limits that stage (a repartition(n) or an RDD
|
|
707
|
+
// shuffle is), so the text points at the stage's own partitioning and no property is suggested.
|
|
708
|
+
// Unlogged, the count is not a safe value: null.
|
|
709
|
+
function lowShuffleParallelismFix(app , needed ) {
|
|
710
|
+
const logged = loggedShufflePartitions(app);
|
|
711
|
+
if (logged != null && logged >= needed) {
|
|
712
|
+
return {
|
|
713
|
+
text: `spark.sql.shuffle.partitions is already ${logged}, so raise this stage's own partition count (its repartition(n) or RDD parallelism) so each partition is smaller`,
|
|
714
|
+
remediation: [],
|
|
715
|
+
};
|
|
716
|
+
}
|
|
717
|
+
return {
|
|
718
|
+
text: 'raise spark.sql.shuffle.partitions so each partition is smaller',
|
|
719
|
+
remediation: [increaseConf('spark.sql.shuffle.partitions', logged == null ? null : needed)],
|
|
720
|
+
};
|
|
612
721
|
}
|
|
613
722
|
|
|
614
|
-
//
|
|
615
|
-
|
|
616
|
-
|
|
617
|
-
// skew/straggler claims shorten the stage's longest task, hence shortensLongestTask (see occupancy.ts).
|
|
618
|
-
function clippedWasteMs(
|
|
619
|
-
wasteMs , stageId , ctx , removedCoreWorkMs , longestTaskAfterFixMs = 0,
|
|
620
|
-
) {
|
|
621
|
-
if (!ctx) return wasteMs;
|
|
622
|
-
const est = estimateSingleStage(
|
|
623
|
-
wasteMs, stageId, ctx.stages , ctx.occupancy,
|
|
624
|
-
{ shortensLongestTask: true, removedCoreWorkMs, longestTaskAfterFixMs },
|
|
625
|
-
);
|
|
626
|
-
return est ? est.wallClock.high : wasteMs;
|
|
723
|
+
// No dynamic-allocation property has an effect on a run whose logged conf turns it off.
|
|
724
|
+
function dynamicAllocationOff(app ) {
|
|
725
|
+
return app?.resources?.dynamicAllocationEnabled === false || app?.config?.['spark.dynamicAllocation.enabled']?.trim().toLowerCase() === 'false';
|
|
627
726
|
}
|
|
628
727
|
|
|
629
|
-
// Shared by cacheUtilization's two variants
|
|
630
|
-
//
|
|
631
|
-
const CACHE_UTILIZATION_VALIDATION =
|
|
632
|
-
"This ratio is a point-in-time storage snapshot from stage-submission events, not a runtime read-count. Confirm against the Spark UI's Storage tab before acting."
|
|
728
|
+
// Shared by cacheUtilization's two variants, worded per storage source: neither is a runtime
|
|
729
|
+
// block-access read-count.
|
|
730
|
+
const CACHE_UTILIZATION_VALIDATION = {
|
|
731
|
+
rddInfo: "This ratio is a point-in-time storage snapshot from stage-submission events, not a runtime read-count. Confirm against the Spark UI's Storage tab before acting.",
|
|
732
|
+
blockUpdates: "This ratio is the RDD's peak cache residency rebuilt from block-update events, not a runtime read-count. Confirm against the Spark UI's Storage tab before acting.",
|
|
733
|
+
} ;
|
|
633
734
|
|
|
634
735
|
// Both cachedRatio and diskRatio are percentages of an RDD's partition/byte count: with few
|
|
635
736
|
// partitions, one partition flipping cached/evicted (or disk-resident/memory-resident) swings the
|
|
@@ -644,59 +745,195 @@ function cacheSampleConfidence(numPartitions )
|
|
|
644
745
|
return 'medium';
|
|
645
746
|
}
|
|
646
747
|
|
|
748
|
+
/** A sentence names a cached RDD by its first 40 characters, since RDD names are often a whole
|
|
749
|
+
* plan string (the Cache Storage card shows it in full); an unnamed RDD reads "RDD <id>". */
|
|
750
|
+
function rddLabel({ id, name } ) {
|
|
751
|
+
return `RDD ${!name ? id : name.length > 40 ? `${name.slice(0, 40)}...` : name}`;
|
|
752
|
+
}
|
|
753
|
+
|
|
647
754
|
function partialCacheFinding(rdd , cachedRatio , impactBand ) {
|
|
648
|
-
const rddName = rdd.name || `RDD ${rdd.id}`;
|
|
649
755
|
const cachedPct = Math.round(cachedRatio * 100);
|
|
650
756
|
const evictedPct = 100 - cachedPct;
|
|
651
757
|
return {
|
|
652
758
|
type: 'cacheUtilization', variant: 'partialCache', stageId: null,
|
|
653
|
-
rddId: rdd.id, rddName
|
|
759
|
+
rddId: rdd.id, rddName: rdd.name || `RDD ${rdd.id}`,
|
|
654
760
|
impactBand, metric: 'cachedRatio', value: cachedPct,
|
|
655
761
|
confidence: cacheSampleConfidence(rdd.numPartitions),
|
|
656
|
-
validationRequired: CACHE_UTILIZATION_VALIDATION,
|
|
762
|
+
validationRequired: CACHE_UTILIZATION_VALIDATION[rdd.storageSource ?? 'rddInfo'],
|
|
657
763
|
memorySize: rdd.memorySize, diskSize: rdd.diskSize,
|
|
658
764
|
numCachedPartitions: rdd.numCachedPartitions, numPartitions: rdd.numPartitions,
|
|
659
|
-
recommendation:
|
|
765
|
+
recommendation: `${rddLabel(rdd)} is ${evictedPct}% evicted from cache (${cachedPct}% of partitions cached): increase executor memory or reduce the cached dataset size.`,
|
|
766
|
+
remediation: [increaseConf('spark.executor.memory')],
|
|
660
767
|
};
|
|
661
768
|
}
|
|
662
769
|
|
|
663
770
|
function diskSpilloverFinding(rdd , diskRatio , impactBand ) {
|
|
664
|
-
const rddName = rdd.name || `RDD ${rdd.id}`;
|
|
665
771
|
const diskPct = Math.round(diskRatio * 100);
|
|
666
772
|
return {
|
|
667
773
|
type: 'cacheUtilization', variant: 'diskSpillover', stageId: null,
|
|
668
|
-
rddId: rdd.id, rddName
|
|
774
|
+
rddId: rdd.id, rddName: rdd.name || `RDD ${rdd.id}`,
|
|
669
775
|
impactBand, metric: 'diskRatio', value: diskPct,
|
|
670
776
|
confidence: cacheSampleConfidence(rdd.numPartitions),
|
|
671
|
-
validationRequired: CACHE_UTILIZATION_VALIDATION,
|
|
777
|
+
validationRequired: CACHE_UTILIZATION_VALIDATION[rdd.storageSource ?? 'rddInfo'],
|
|
672
778
|
memorySize: rdd.memorySize, diskSize: rdd.diskSize,
|
|
673
779
|
numCachedPartitions: rdd.numCachedPartitions, numPartitions: rdd.numPartitions,
|
|
674
|
-
recommendation:
|
|
780
|
+
recommendation: `${rddLabel(rdd)} is ${diskPct}% spilled to disk despite requesting MEMORY_AND_DISK: executor memory may be too small for it.`,
|
|
781
|
+
remediation: [increaseConf('spark.executor.memory')],
|
|
675
782
|
};
|
|
676
783
|
}
|
|
677
784
|
|
|
678
|
-
//
|
|
679
|
-
//
|
|
680
|
-
//
|
|
681
|
-
|
|
682
|
-
|
|
683
|
-
|
|
684
|
-
|
|
685
|
-
|
|
785
|
+
// Persisted RDDs with no storage evidence at all: no block updates in the log, and RDD Info's
|
|
786
|
+
// sizes are the 0 that Spark 2.3+ always writes. Reports the gap instead of a clean result, the
|
|
787
|
+
// same missing-evidence shape as memoryUtilization's dataUnavailable caveat.
|
|
788
|
+
function storageUnobservedFinding(persistedRddCount , app ) {
|
|
789
|
+
const rdds = persistedRddCount === 1 ? '1 persisted RDD has' : `${persistedRddCount} persisted RDDs have`;
|
|
790
|
+
return {
|
|
791
|
+
type: 'cacheUtilization', variant: 'storageUnobserved', stageId: null,
|
|
792
|
+
impactBand: 'info', metric: 'persistedRdds', value: persistedRddCount, dataUnavailable: true,
|
|
793
|
+
recommendation: `${rdds} no cache-storage evidence in this log, so eviction and disk spillover can't be checked: Spark 2.3+ records cached sizes only as block updates, which need spark.eventLog.logBlockUpdates.enabled=true.`,
|
|
794
|
+
remediation: setConfUnlessLogged(app, 'spark.eventLog.logBlockUpdates.enabled', true),
|
|
795
|
+
};
|
|
796
|
+
}
|
|
797
|
+
|
|
798
|
+
/** A detector entry's threshold set. number[] too: slowHost's ratioTiers and broadcastSizing's
|
|
799
|
+
* tiers are tier tables its detect() indexes by position. */
|
|
800
|
+
|
|
801
|
+
|
|
802
|
+
|
|
803
|
+
|
|
804
|
+
// What each scope's detect() is handed. `thresholds` is the entry's own set, or the caller's
|
|
805
|
+
// overrides merged over it (analyze()'s `thresholds` option). 'config' has no DetectorCtx:
|
|
806
|
+
// auditConfig() runs those entries on the app alone.
|
|
807
|
+
|
|
808
|
+
|
|
809
|
+
|
|
810
|
+
|
|
811
|
+
|
|
812
|
+
|
|
813
|
+
|
|
814
|
+
|
|
815
|
+
|
|
816
|
+
// The same calls with the thresholds already bound: what a runner holds after withThresholds().
|
|
817
|
+
|
|
818
|
+
|
|
819
|
+
|
|
820
|
+
|
|
821
|
+
|
|
822
|
+
|
|
823
|
+
|
|
824
|
+
// The fields a define*Detector() call spells out. `detect` is a property, not a method, so its
|
|
825
|
+
// parameters are checked contravariantly: a detect() that reads a threshold the entry doesn't
|
|
826
|
+
// declare, or expects a different target, fails to compile.
|
|
827
|
+
|
|
828
|
+
|
|
829
|
+
|
|
830
|
+
|
|
831
|
+
|
|
832
|
+
|
|
833
|
+
|
|
834
|
+
|
|
835
|
+
|
|
836
|
+
|
|
686
837
|
|
|
687
838
|
|
|
688
839
|
|
|
689
840
|
|
|
690
|
-
|
|
691
|
-
|
|
692
|
-
|
|
693
|
-
|
|
694
|
-
|
|
841
|
+
|
|
695
842
|
|
|
696
843
|
|
|
697
|
-
|
|
844
|
+
|
|
698
845
|
|
|
846
|
+
|
|
847
|
+
|
|
848
|
+
|
|
849
|
+
|
|
850
|
+
|
|
851
|
+
|
|
699
852
|
|
|
853
|
+
|
|
854
|
+
|
|
855
|
+
|
|
856
|
+
|
|
857
|
+
|
|
858
|
+
|
|
859
|
+
|
|
860
|
+
|
|
861
|
+
|
|
862
|
+
|
|
863
|
+
|
|
864
|
+
|
|
865
|
+
/** How runners (analyze(), auditConfig(), the catalog helpers) see any DETECTORS entry: its
|
|
866
|
+
* thresholds type erased and detect() left off, so they reach it only through withThresholds().
|
|
867
|
+
* estimate() takes any Finding here: estimateImpact() only hands an entry the types it emits. */
|
|
868
|
+
|
|
869
|
+
|
|
870
|
+
|
|
871
|
+
|
|
872
|
+
|
|
873
|
+
|
|
874
|
+
// Same-shape overrides merged over the defaults. analyze()'s callers validate overrides at their
|
|
875
|
+
// own boundary (threshold-overrides.ts); this re-checks so a programmatic caller can't hand a
|
|
876
|
+
// detector a threshold of the wrong shape, which is what makes the cast below sound.
|
|
877
|
+
function mergeThresholds (type , defaults , overrides ) {
|
|
878
|
+
for (const [name, value] of Object.entries(overrides)) {
|
|
879
|
+
// Own keys only: an inherited name such as `constructor` or `toString` is no threshold.
|
|
880
|
+
if (!Object.hasOwn(defaults, name)) throw new Error(`Detector ${type} has no threshold "${name}".`);
|
|
881
|
+
const fallback = defaults[name];
|
|
882
|
+
const sameShape = Array.isArray(fallback)
|
|
883
|
+
? Array.isArray(value) && value.length === fallback.length
|
|
884
|
+
: typeof value === 'number';
|
|
885
|
+
if (!sameShape) throw new Error(`Threshold ${type}.${name} must have the same shape as its default.`);
|
|
886
|
+
}
|
|
887
|
+
return Object.freeze({ ...defaults, ...overrides }) ;
|
|
888
|
+
}
|
|
889
|
+
|
|
890
|
+
function defineDetector
|
|
891
|
+
|
|
892
|
+
|
|
893
|
+
(
|
|
894
|
+
scope , spec ,
|
|
895
|
+
bind ,
|
|
896
|
+
) {
|
|
897
|
+
// Frozen: the defaults are the specification, never a knob to mutate in place.
|
|
898
|
+
const defaults = Object.freeze({ ...spec.thresholds }) ;
|
|
899
|
+
const boundDefaults = bind(defaults);
|
|
900
|
+
return {
|
|
901
|
+
...spec, scope, thresholds: defaults,
|
|
902
|
+
withThresholds: (overrides) => (overrides && Object.keys(overrides).length > 0
|
|
903
|
+
? bind(mergeThresholds(spec.type, defaults, overrides))
|
|
904
|
+
: boundDefaults),
|
|
905
|
+
};
|
|
906
|
+
}
|
|
907
|
+
|
|
908
|
+
// One helper per scope: each infers the entry's thresholds type from its `thresholds` literal
|
|
909
|
+
// and hands detect() exactly that type, with the scope's target and a required context.
|
|
910
|
+
export function defineStageDetector
|
|
911
|
+
|
|
912
|
+
|
|
913
|
+
(spec ) {
|
|
914
|
+
return defineDetector('stage', spec, (t) => (stage, ctx) => spec.detect(stage, ctx, t));
|
|
915
|
+
}
|
|
916
|
+
|
|
917
|
+
export function defineSqlDetector
|
|
918
|
+
|
|
919
|
+
|
|
920
|
+
(spec ) {
|
|
921
|
+
return defineDetector('sql', spec, (t) => (sqlExec, ctx) => spec.detect(sqlExec, ctx, t));
|
|
922
|
+
}
|
|
923
|
+
|
|
924
|
+
export function defineAppDetector
|
|
925
|
+
|
|
926
|
+
|
|
927
|
+
(spec ) {
|
|
928
|
+
return defineDetector('app', spec, (t) => (ctx) => spec.detect(ctx, t));
|
|
929
|
+
}
|
|
930
|
+
|
|
931
|
+
export function defineConfigDetector
|
|
932
|
+
|
|
933
|
+
|
|
934
|
+
(spec ) {
|
|
935
|
+
return defineDetector('config', spec, (t) => (target) => spec.detect(target, t));
|
|
936
|
+
}
|
|
700
937
|
|
|
701
938
|
// Threshold field naming convention:
|
|
702
939
|
// *Pct = 0–1 fraction (normalized)
|
|
@@ -704,11 +941,6 @@ function diskSpilloverFinding(rdd , diskRatio , impactBan
|
|
|
704
941
|
// *Ratio = multiplicative factor
|
|
705
942
|
// *Share/*Rate/*Util = 0–1 fraction (normalized)
|
|
706
943
|
|
|
707
|
-
// straggler's noise-floor thresholds (NOT SOURCED: unvalidated), exported so impact-band.ts
|
|
708
|
-
// reuses the same figures instead of hand-copying.
|
|
709
|
-
export const STRAGGLER_FLOOR_PCT_WARN = 0.005;
|
|
710
|
-
export const STRAGGLER_FLOOR_PCT_CRIT = 0.02;
|
|
711
|
-
|
|
712
944
|
// A ratio just past ratioWarn is the case most likely to be ordinary task-duration variance
|
|
713
945
|
// rather than real skew; a ratio many multiples past it (a 50x P95/median vs. a 3.1x one) is
|
|
714
946
|
// unambiguous. 1.5x/5x mirror this file's other "just past the floor vs. clearly past it" splits
|
|
@@ -806,75 +1038,88 @@ function cachingReuseConfidence(occurrences , minExecutions )
|
|
|
806
1038
|
return 'medium';
|
|
807
1039
|
}
|
|
808
1040
|
|
|
809
|
-
|
|
810
|
-
|
|
811
|
-
|
|
1041
|
+
// Caveats that name a threshold read it from the thresholds the detector ran with, so a tuned run
|
|
1042
|
+
// states the floor it actually used.
|
|
1043
|
+
function gcValidation(minRunTimeMs ) {
|
|
1044
|
+
return `Checked only on stages with at least ${minRunTimeMs / 1000}s of executor run time.`;
|
|
1045
|
+
}
|
|
1046
|
+
|
|
1047
|
+
const INCOMPLETE_RUN_RECOMMENDATION = 'This event log never recorded an ApplicationEnd event: the capture stopped before the run finished (a job still running, a rotated log, or a cut-short capture), so every figure on this board covers only what was captured.';
|
|
1048
|
+
|
|
1049
|
+
export const DETECTORS = [
|
|
1050
|
+
defineStageDetector({
|
|
1051
|
+
type: 'skew', order: 30, fixEffort: 'code', version: 1,
|
|
1052
|
+
emits: ['skew'],
|
|
812
1053
|
docAnchor: '#bottleneck-skew',
|
|
813
|
-
thresholds: { ratioWarn: 3, minTasksForP95: 20, floorPctWarn:
|
|
814
|
-
detect(
|
|
815
|
-
|
|
816
|
-
|
|
817
|
-
|
|
818
|
-
stage ,
|
|
819
|
-
ctx ,
|
|
820
|
-
) {
|
|
821
|
-
const result = computeSkewRatio(stage, this.thresholds.minTasksForP95);
|
|
1054
|
+
thresholds: { ratioWarn: 3, minTasksForP95: 20, floorPctWarn: IMPACT_FLOOR_PCT_WARN },
|
|
1055
|
+
detect(stage, ctx, thresholds) {
|
|
1056
|
+
const result = computeSkewRatio(stage, thresholds.minTasksForP95);
|
|
822
1057
|
if (result === null) return null;
|
|
823
1058
|
const { ratio, metric } = result;
|
|
824
|
-
if (ratio <=
|
|
825
|
-
//
|
|
826
|
-
|
|
827
|
-
|
|
828
|
-
const wasteMs = tailRecoveryMs(stage, singleDelta);
|
|
829
|
-
const appDurationMs = computeAppDurationMs(ctx);
|
|
830
|
-
const floorWasteMs = clippedWasteMs(wasteMs, stage.id, ctx, tailRemovedWorkMs(stage, singleDelta), stragglerFixLongestTaskMs(stage));
|
|
831
|
-
if (!meetsRuntimeFloor(floorWasteMs, appDurationMs, this.thresholds.floorPctWarn)) return null;
|
|
1059
|
+
if (ratio <= thresholds.ratioWarn) return null;
|
|
1060
|
+
// The claim estimate() reports as savings, clipped the same way, so the gate agrees with it.
|
|
1061
|
+
const floorWasteMs = tailClaimFloorMs(skewTailClaim(stage, metric === 'P95/median'), stage.id, ctx);
|
|
1062
|
+
if (!meetsRuntimeFloor(floorWasteMs, appDurationMs(ctx.app), thresholds.floorPctWarn)) return null;
|
|
832
1063
|
const value = Math.round(ratio * 10) / 10;
|
|
1064
|
+
const fix = skewJoinFix(ctx.app);
|
|
833
1065
|
return {
|
|
834
1066
|
type: 'skew', stageId: stage.id,
|
|
835
1067
|
impactBand: 'warning',
|
|
836
1068
|
metric, value,
|
|
837
|
-
confidence: skewConfidence(ratio,
|
|
838
|
-
validationRequired:
|
|
839
|
-
recommendation: `Task duration ratio (${metric}) is ${value}×:
|
|
1069
|
+
confidence: skewConfidence(ratio, thresholds.ratioWarn),
|
|
1070
|
+
validationRequired: `Flagged only when it costs at least ${shareLabel(thresholds.floorPctWarn)} of run time.`,
|
|
1071
|
+
recommendation: `Task duration ratio (${metric}) is ${value}×: ${fix.text}.`,
|
|
1072
|
+
remediation: fix.remediation,
|
|
840
1073
|
};
|
|
841
1074
|
},
|
|
842
|
-
|
|
843
|
-
|
|
844
|
-
|
|
1075
|
+
estimate(finding, ctx) {
|
|
1076
|
+
if (finding.stageId == null) return null;
|
|
1077
|
+
const stage = ctx.stages.get(finding.stageId);
|
|
1078
|
+
if (!stage) return null;
|
|
1079
|
+
// computeSkewRatio's own metric labels: 'P95/median' or 'max/median'.
|
|
1080
|
+
return tailClaimImpact(skewTailClaim(stage, finding.metric === 'P95/median'), finding.stageId, ctx);
|
|
1081
|
+
},
|
|
1082
|
+
}),
|
|
1083
|
+
defineStageDetector({
|
|
1084
|
+
type: 'stageShape', order: 35, fixEffort: 'code', version: 2,
|
|
1085
|
+
emits: ['stageShape'],
|
|
845
1086
|
docAnchor: '#bottleneck-stage-shape',
|
|
846
1087
|
// lowParallelismFloorPct: the same 0.5% runtime floor the tiered detectors use. Parallelizing a
|
|
847
1088
|
// stage can't save more than the stage's own duration, so a shorter stage can't clear it; on
|
|
848
1089
|
// the 14 real logs that was 2839 of 3005 lowParallelism findings (2168 on sub-second stages).
|
|
849
1090
|
// App-wide idle capacity stays covered by utilization.
|
|
850
|
-
|
|
851
|
-
|
|
852
|
-
|
|
853
|
-
|
|
854
|
-
|
|
855
|
-
|
|
1091
|
+
// taskStageSkew: stageShareMin is the share of the stage's wall-clock the longest task must
|
|
1092
|
+
// span, and skewWarn how far past the median task it must run (skew's own max/median 3×).
|
|
1093
|
+
// The share alone can't tell a straggler apart: on a single wave (tasks <= cores) the longest
|
|
1094
|
+
// task spans nearly the whole stage however even the tasks are. taskStageSkewFloorPct is the
|
|
1095
|
+
// same 0.5% runtime floor as lowParallelismFloorPct.
|
|
1096
|
+
thresholds: {
|
|
1097
|
+
pRatioMax: 0.5, oiRatioMax: 10, skewWarn: 3, stageShareMin: 0.5,
|
|
1098
|
+
lowParallelismFloorPct: 0.005, taskStageSkewFloorPct: 0.005,
|
|
1099
|
+
},
|
|
1100
|
+
detect(stage, ctx, thresholds) {
|
|
856
1101
|
const out = [];
|
|
857
1102
|
const execCount = (stage.executorStats ?? []).length;
|
|
858
|
-
const cores = ctx
|
|
1103
|
+
const cores = ctx.app?.resources?.executor?.cores ?? 1;
|
|
859
1104
|
const totalCores = execCount * cores;
|
|
860
1105
|
const stageDurationMs = (stage.completedAt ?? 0) - (stage.submittedAt ?? 0);
|
|
861
1106
|
// PRatio: under-parallelization.
|
|
862
|
-
if (totalCores > 0 && !stageBelowRuntimeFloor(stage, ctx,
|
|
1107
|
+
if (totalCores > 0 && !stageBelowRuntimeFloor(stage, ctx, thresholds.lowParallelismFloorPct)) {
|
|
863
1108
|
const pRatio = stage.taskCount / totalCores;
|
|
864
|
-
if (pRatio <
|
|
1109
|
+
if (pRatio < thresholds.pRatioMax) {
|
|
865
1110
|
out.push({
|
|
866
1111
|
type: 'stageShape', stageId: stage.id, impactBand: 'info',
|
|
867
1112
|
rule: 'lowParallelism', metric: 'pRatio', value: Math.round(pRatio * 100) / 100,
|
|
868
1113
|
// Absolute core count behind pRatio, for the impact estimator's idle-core-ms figure.
|
|
869
1114
|
totalCores,
|
|
870
|
-
recommendation: `This stage runs ${stage.taskCount} ${stage.taskCount === 1 ? 'task' : 'tasks'} across ~${totalCores} cores
|
|
1115
|
+
recommendation: `This stage runs ${stage.taskCount} ${stage.taskCount === 1 ? 'task' : 'tasks'} across ~${totalCores} cores: it is under-parallelized and leaves cluster capacity idle.`,
|
|
871
1116
|
});
|
|
872
1117
|
}
|
|
873
1118
|
}
|
|
874
1119
|
// OIRatio: data explosion. Skip when inputBytes is 0 (Infinity guard).
|
|
875
1120
|
if (stage.inputBytes > 0) {
|
|
876
1121
|
const oiRatio = stage.outputBytes / stage.inputBytes;
|
|
877
|
-
if (oiRatio >
|
|
1122
|
+
if (oiRatio > thresholds.oiRatioMax) {
|
|
878
1123
|
out.push({
|
|
879
1124
|
type: 'stageShape', stageId: stage.id, impactBand: 'info',
|
|
880
1125
|
rule: 'dataExplosion', metric: 'oiRatio', value: Math.round(oiRatio * 10) / 10,
|
|
@@ -882,82 +1127,122 @@ export const DETECTORS = [
|
|
|
882
1127
|
});
|
|
883
1128
|
}
|
|
884
1129
|
}
|
|
885
|
-
// TaskStageSkew: straggler
|
|
886
|
-
//
|
|
887
|
-
//
|
|
888
|
-
|
|
889
|
-
|
|
890
|
-
|
|
1130
|
+
// TaskStageSkew: one straggler sets when the stage ends. A task runs inside its stage's
|
|
1131
|
+
// window, so the longest task's share of the stage's wall-clock is at most 1. Skip a
|
|
1132
|
+
// zero-length or single-task stage. Always info like its siblings: skew and straggler
|
|
1133
|
+
// already make the wall-clock claim for the same tail, so this one reports idle core-time.
|
|
1134
|
+
if (stageDurationMs > 0 && stage.taskCount > 1 && stage.taskDurationP50 > 0
|
|
1135
|
+
&& !stageBelowRuntimeFloor(stage, ctx, thresholds.taskStageSkewFloorPct)) {
|
|
1136
|
+
const share = stage.taskDurationMax / stageDurationMs;
|
|
1137
|
+
const vsMedian = stage.taskDurationMax / stage.taskDurationP50;
|
|
1138
|
+
if (share > thresholds.stageShareMin && vsMedian > thresholds.skewWarn) {
|
|
891
1139
|
out.push({
|
|
892
1140
|
type: 'stageShape', stageId: stage.id, impactBand: 'info',
|
|
893
|
-
rule: 'taskStageSkew', metric: 'taskStageSkew', value: Math.round(
|
|
1141
|
+
rule: 'taskStageSkew', metric: 'taskStageSkew', value: Math.round(share * 100) / 100,
|
|
894
1142
|
// Absolute core count, for the impact estimator's idle-core-ms figure.
|
|
895
1143
|
totalCores,
|
|
896
|
-
recommendation: `
|
|
1144
|
+
recommendation: `The longest task ran for ${Math.round(share * 100)}% of this stage's wall-clock, ${Math.round(vsMedian * 10) / 10}× the median task: a single straggler is gating the whole stage.`,
|
|
897
1145
|
});
|
|
898
1146
|
}
|
|
899
1147
|
}
|
|
900
1148
|
return out;
|
|
901
1149
|
},
|
|
902
|
-
|
|
903
|
-
|
|
904
|
-
|
|
1150
|
+
estimate(finding, ctx) {
|
|
1151
|
+
const stage = ctx.stages.get(finding.stageId );
|
|
1152
|
+
if (!stage) return null;
|
|
1153
|
+
if (finding.rule === 'lowParallelism') {
|
|
1154
|
+
const stageDurationMs = (stage.completedAt ?? 0) - (stage.submittedAt ?? 0);
|
|
1155
|
+
const idleCoreMs =
|
|
1156
|
+
Math.max(0, ((finding.totalCores ) ?? 0) - (stage.taskCount ?? 0)) * stageDurationMs;
|
|
1157
|
+
// Real per-stage data (cores, task count, duration), no assumed constant.
|
|
1158
|
+
return costOnly('measured', { value: idleCoreMs, unit: 'coreMs', idle: true });
|
|
1159
|
+
}
|
|
1160
|
+
if (finding.rule === 'dataExplosion') {
|
|
1161
|
+
const excessBytes = Math.max(0, (stage.outputBytes ?? 0) - (stage.inputBytes ?? 0));
|
|
1162
|
+
// Measured input/output byte counts, no assumed constant.
|
|
1163
|
+
return costOnly('measured', { value: excessBytes, unit: 'bytes' });
|
|
1164
|
+
}
|
|
1165
|
+
if (finding.rule === 'taskStageSkew') {
|
|
1166
|
+
const totalCores = (finding.totalCores ) ?? 0;
|
|
1167
|
+
const taskCount = stage.taskCount ?? 0;
|
|
1168
|
+
// Cores idle during the straggler's tail, at achieved concurrency (not full cluster
|
|
1169
|
+
// capacity, which is lowParallelism's territory): resourceOnly, since skew and straggler
|
|
1170
|
+
// already claim that tail's wall-clock time.
|
|
1171
|
+
const idleCoreMs =
|
|
1172
|
+
Math.max(0, Math.min(totalCores, taskCount) - 1) *
|
|
1173
|
+
Math.max(0, (stage.taskDurationMax ?? 0) - (stage.taskDurationP50 ?? 0));
|
|
1174
|
+
return costOnly('measured', { value: idleCoreMs, unit: 'coreMs', idle: true });
|
|
1175
|
+
}
|
|
1176
|
+
return null;
|
|
1177
|
+
},
|
|
1178
|
+
}),
|
|
1179
|
+
defineStageDetector({
|
|
1180
|
+
type: 'shuffle', order: 20, fixEffort: 'config', version: 1,
|
|
1181
|
+
emits: ['shuffle'],
|
|
905
1182
|
docAnchor: '#bottleneck-shuffle',
|
|
906
1183
|
// stageFloorPct: the 0.5% runtime floor. The shuffle claim is clipped to the stage, so on a
|
|
907
1184
|
// shorter stage it graded info: 182 of 284 shuffle findings on the 14 real logs. The shuffle
|
|
908
1185
|
// is still there on those stages; the floor is why they're dropped.
|
|
909
1186
|
thresholds: { minBytes: 50 * MB, stageFloorPct: 0.005 },
|
|
910
|
-
detect(
|
|
911
|
-
|
|
912
|
-
stage ,
|
|
913
|
-
ctx ,
|
|
914
|
-
) {
|
|
1187
|
+
detect(stage, ctx, thresholds) {
|
|
915
1188
|
const bytes = stage.shuffleReadBytes;
|
|
916
|
-
if (bytes <=
|
|
917
|
-
if (stageBelowRuntimeFloor(stage, ctx,
|
|
1189
|
+
if (bytes <= thresholds.minBytes) return null;
|
|
1190
|
+
if (stageBelowRuntimeFloor(stage, ctx, thresholds.stageFloorPct)) return null;
|
|
918
1191
|
return {
|
|
919
1192
|
type: 'shuffle', stageId: stage.id,
|
|
920
1193
|
impactBand: 'info',
|
|
921
1194
|
metric: 'shuffleReadBytes', value: bytes,
|
|
922
1195
|
recommendation: `${formatBytes(bytes)} shuffled in this stage: consider increasing spark.sql.shuffle.partitions or adding a broadcast join.`,
|
|
1196
|
+
remediation: [increaseConf('spark.sql.shuffle.partitions')],
|
|
923
1197
|
};
|
|
924
1198
|
},
|
|
925
|
-
|
|
926
|
-
|
|
927
|
-
|
|
1199
|
+
estimate(finding, ctx) {
|
|
1200
|
+
if (finding.stageId == null) return null;
|
|
1201
|
+
const stage = ctx.stages.get(finding.stageId);
|
|
1202
|
+
if (!stage) return null;
|
|
1203
|
+
const shuffleReadBytes = stage.shuffleReadBytes ?? 0;
|
|
1204
|
+
const modeledMs = (shuffleReadBytes / (SHUFFLE_THROUGHPUT_BPS * stageIoParallelism(stage))) * 1000;
|
|
1205
|
+
// The link model can't see whether the reads stalled the tasks: capped at the fetch wait the
|
|
1206
|
+
// tasks measured, the claim never exceeds what the stage spent blocked on the network.
|
|
1207
|
+
const measuredMs = fetchWaitWallClockMs(stage);
|
|
1208
|
+
const wasteMs = measuredMs == null ? modeledMs : Math.min(modeledMs, measuredMs);
|
|
1209
|
+
// rawWaste: the measured byte volume behind the modeled figure.
|
|
1210
|
+
return singleStageImpact(wasteMs, finding.stageId, ctx,
|
|
1211
|
+
measuredMs != null && measuredMs < modeledMs ? 'measured' : 'modeled', { value: shuffleReadBytes, unit: 'bytes' });
|
|
1212
|
+
},
|
|
1213
|
+
}),
|
|
1214
|
+
defineStageDetector({
|
|
1215
|
+
type: 'partitionSizing', order: 22, fixEffort: 'config', version: 1,
|
|
1216
|
+
emits: ['partitionSizing'],
|
|
928
1217
|
docAnchor: '#bottleneck-partition-sizing',
|
|
929
1218
|
thresholds: { skewRatio: 5, skewFloorBytes: 256 * MB, lowParTotalBytes: GB, lowParMaxTasks: 7, maxPartBytes: 5 * GB },
|
|
930
|
-
detect(
|
|
931
|
-
|
|
932
|
-
|
|
933
|
-
|
|
934
|
-
|
|
935
|
-
|
|
936
|
-
|
|
937
|
-
stage ,
|
|
938
|
-
) {
|
|
1219
|
+
detect(stage, ctx, thresholds) {
|
|
939
1220
|
const out = [];
|
|
940
1221
|
const { shuffleReadP50: p50, shuffleReadMax: max, shuffleReadBytes: total, taskCount } = stage;
|
|
941
|
-
if (max >
|
|
1222
|
+
if (max > thresholds.skewRatio * p50 && max > thresholds.skewFloorBytes) {
|
|
942
1223
|
// p50 can be 0 (over half the shuffle partitions empty): a ratio against zero renders
|
|
943
1224
|
// "Infinity×", so fall back to median-free phrasing.
|
|
944
1225
|
const ratioText = p50 > 0
|
|
945
1226
|
? `${Math.round(max / p50 * 10) / 10}× the median (${formatBytes(p50)})`
|
|
946
|
-
:
|
|
1227
|
+
: 'far larger than the median, which is effectively empty';
|
|
1228
|
+
const fix = skewJoinFix(ctx.app);
|
|
947
1229
|
out.push({
|
|
948
1230
|
type: 'partitionSizing', stageId: stage.id, impactBand: 'warning',
|
|
949
1231
|
rule: 'shufflePartitionSkew', metric: 'shuffleReadMax', value: max,
|
|
950
|
-
recommendation: `The largest shuffle partition (${formatBytes(max)}) is ${ratioText}:
|
|
1232
|
+
recommendation: `The largest shuffle partition (${formatBytes(max)}) is ${ratioText}: ${fix.text}.`,
|
|
1233
|
+
remediation: fix.remediation,
|
|
951
1234
|
});
|
|
952
1235
|
}
|
|
953
|
-
if (total >=
|
|
1236
|
+
if (total >= thresholds.lowParTotalBytes && taskCount <= thresholds.lowParMaxTasks) {
|
|
1237
|
+
const fix = lowShuffleParallelismFix(ctx.app, Math.ceil(total / IDEAL_BYTES_PER_PARTITION_TASK));
|
|
954
1238
|
out.push({
|
|
955
1239
|
type: 'partitionSizing', stageId: stage.id, impactBand: 'warning',
|
|
956
1240
|
rule: 'lowShuffleParallelism', metric: 'taskCount', value: taskCount,
|
|
957
|
-
recommendation: `${Math.round(total / GB * 10) / 10} GB of shuffle spread over only ${taskCount} tasks:
|
|
1241
|
+
recommendation: `${Math.round(total / GB * 10) / 10} GB of shuffle spread over only ${taskCount} tasks: ${fix.text}.`,
|
|
1242
|
+
remediation: fix.remediation,
|
|
958
1243
|
});
|
|
959
1244
|
}
|
|
960
|
-
if (max >=
|
|
1245
|
+
if (max >= thresholds.maxPartBytes) {
|
|
961
1246
|
// Fixed 'critical': an OOM/crash-risk safety signal, not a time-waste one. Exempted in
|
|
962
1247
|
// impact-band.ts's deriveImpactBand from the wall-clock-based overwrite every other
|
|
963
1248
|
// finding here gets, so a long-running job can't demote an active crash risk to 'info'
|
|
@@ -970,19 +1255,46 @@ export const DETECTORS = [
|
|
|
970
1255
|
}
|
|
971
1256
|
return out;
|
|
972
1257
|
},
|
|
973
|
-
|
|
974
|
-
|
|
975
|
-
|
|
1258
|
+
estimate(finding, ctx) {
|
|
1259
|
+
if (finding.stageId == null) return null;
|
|
1260
|
+
const stage = ctx.stages.get(finding.stageId);
|
|
1261
|
+
if (!stage) return null;
|
|
1262
|
+
let wasteMs = 0;
|
|
1263
|
+
if (finding.rule === 'maxPartitionTooBig') {
|
|
1264
|
+
wasteMs = ((stage.shuffleReadMax ?? 0) / SHUFFLE_THROUGHPUT_BPS) * 1000;
|
|
1265
|
+
} else if (finding.rule === 'shufflePartitionSkew') {
|
|
1266
|
+
const delta = Math.max(0, (stage.shuffleReadMax ?? 0) - (stage.shuffleReadP50 ?? 0));
|
|
1267
|
+
wasteMs = (delta / SHUFFLE_THROUGHPUT_BPS) * 1000;
|
|
1268
|
+
} else if (finding.rule === 'lowShuffleParallelism') {
|
|
1269
|
+
const targetTaskCount = Math.ceil((stage.shuffleReadBytes ?? 0) / IDEAL_BYTES_PER_PARTITION_TASK);
|
|
1270
|
+
const taskCount = stage.taskCount ?? 0;
|
|
1271
|
+
if (targetTaskCount > taskCount && taskCount > 0) {
|
|
1272
|
+
const stageDurationMs = Math.max(0, (stage.completedAt ?? 0) - (stage.submittedAt ?? 0));
|
|
1273
|
+
// Too few shuffle partitions means each task processes more than the ideal bytes,
|
|
1274
|
+
// serializing work more partitions would run concurrently: the waste is that serialized
|
|
1275
|
+
// work, not the scheduling cost of tasks you'd add (adding tasks incurs overhead, recovers
|
|
1276
|
+
// nothing). Model the achievable duration at target parallelism by scaling down proportionally.
|
|
1277
|
+
wasteMs = stageDurationMs * (1 - taskCount / targetTaskCount);
|
|
1278
|
+
}
|
|
1279
|
+
} else {
|
|
1280
|
+
return null;
|
|
1281
|
+
}
|
|
1282
|
+
return singleStageImpact(wasteMs, finding.stageId, ctx, 'modeled', { value: wasteMs, unit: 'ms' });
|
|
1283
|
+
},
|
|
1284
|
+
}),
|
|
1285
|
+
defineStageDetector({
|
|
1286
|
+
type: 'spill', order: 10, fixEffort: 'code', version: 1,
|
|
1287
|
+
emits: ['spill'],
|
|
976
1288
|
docAnchor: '#bottleneck-spill',
|
|
977
1289
|
// stageFloorPct: the 0.5% runtime floor, as for shuffle (12 of 38 spill findings on the 14 real
|
|
978
1290
|
// logs, all info). The spill is still there on those stages; the floor is why they're dropped.
|
|
979
1291
|
thresholds: { singleTaskDiskGiB: 1, singleTaskMemGiB: 4, highDiskGiB: 1, highTaskDiskMB: 512, highMemGiB: 4, medDiskMB: 256, medMemGiB: 1, skewRatio: 5, skewDiskFloorMB: 128, skewMemFloorMB: 256, skewMinTasks: 10, stageFloorPct: 0.005 },
|
|
980
|
-
detect(
|
|
1292
|
+
detect(stage, ctx, thresholds) {
|
|
981
1293
|
if (stage.memoryBytesSpilled === 0) return null;
|
|
982
|
-
if (stageBelowRuntimeFloor(stage, ctx,
|
|
1294
|
+
if (stageBelowRuntimeFloor(stage, ctx, thresholds.stageFloorPct)) return null;
|
|
983
1295
|
const cls = stage.spillClassification;
|
|
984
1296
|
const classified = cls === 'skew' || cls === 'volume';
|
|
985
|
-
const mag = computeSpillMagnitude(stage,
|
|
1297
|
+
const mag = computeSpillMagnitude(stage, thresholds);
|
|
986
1298
|
const impactBand = 'warning';
|
|
987
1299
|
return {
|
|
988
1300
|
type: 'spill', stageId: stage.id, impactBand,
|
|
@@ -995,13 +1307,23 @@ export const DETECTORS = [
|
|
|
995
1307
|
recommendation: cls === 'skew'
|
|
996
1308
|
? `${formatBytes(stage.memoryBytesSpilled)} spilled, skew-driven: fix task skew first; adding memory will not help.`
|
|
997
1309
|
: `${formatBytes(stage.memoryBytesSpilled)} spilled: raise spark.sql.shuffle.partitions or increase executor memory.`,
|
|
1310
|
+
remediation: cls === 'skew' ? [] : [increaseConf('spark.sql.shuffle.partitions'), increaseConf('spark.executor.memory')],
|
|
998
1311
|
};
|
|
999
1312
|
},
|
|
1000
|
-
|
|
1001
|
-
|
|
1002
|
-
|
|
1313
|
+
estimate(finding, ctx) {
|
|
1314
|
+
if (finding.stageId == null) return null;
|
|
1315
|
+
const stage = ctx.stages.get(finding.stageId);
|
|
1316
|
+
if (!stage) return null;
|
|
1317
|
+
const diskBytesSpilled = stage.diskBytesSpilled ?? 0;
|
|
1318
|
+
const wasteMs = (diskBytesSpilled / (SPILL_IO_THROUGHPUT_BPS * stageIoParallelism(stage))) * 1000;
|
|
1319
|
+
// Surfaces the number the formula uses: the displayed metric is memoryBytesSpilled, but disk spill costs the I/O time.
|
|
1320
|
+
return singleStageImpact(wasteMs, finding.stageId, ctx, 'modeled', { value: diskBytesSpilled, unit: 'bytes' });
|
|
1321
|
+
},
|
|
1322
|
+
}),
|
|
1323
|
+
defineStageDetector({
|
|
1324
|
+
type: 'gc', order: 50, fixEffort: 'config', version: 1,
|
|
1325
|
+
emits: ['gc'],
|
|
1003
1326
|
docAnchor: '#bottleneck-gc',
|
|
1004
|
-
validationRequired: 'This finding is gated by a 10-second minimum-runtime floor, our own noise floor for this metric.',
|
|
1005
1327
|
thresholds: {
|
|
1006
1328
|
warnPct100: 10,
|
|
1007
1329
|
// Descending tier: ExecutorGcHeuristic, ported as-is.
|
|
@@ -1014,44 +1336,62 @@ export const DETECTORS = [
|
|
|
1014
1336
|
// findings); the low-GC pattern is still true on those stages, the floor is why they're dropped.
|
|
1015
1337
|
lowInfoFloorPct: 0.005,
|
|
1016
1338
|
},
|
|
1017
|
-
detect(
|
|
1018
|
-
|
|
1019
|
-
|
|
1020
|
-
|
|
1021
|
-
|
|
1022
|
-
stage ,
|
|
1023
|
-
ctx ,
|
|
1024
|
-
) {
|
|
1339
|
+
detect(stage, ctx, thresholds) {
|
|
1025
1340
|
const pct = stage.gcPct;
|
|
1026
|
-
if ((stage.executorRunTime ?? 0) >=
|
|
1027
|
-
&& pct >
|
|
1341
|
+
if ((stage.executorRunTime ?? 0) >= thresholds.minRunTimeMs
|
|
1342
|
+
&& pct > thresholds.warnPct100) {
|
|
1028
1343
|
const value = Math.round(pct * 10) / 10;
|
|
1029
1344
|
return {
|
|
1030
1345
|
type: 'gc', stageId: stage.id,
|
|
1031
1346
|
impactBand: 'warning',
|
|
1032
1347
|
metric: 'gcPct', value,
|
|
1033
|
-
confidence: gcConfidence(pct,
|
|
1348
|
+
confidence: gcConfidence(pct, thresholds, 'high'), validationRequired: gcValidation(thresholds.minRunTimeMs),
|
|
1034
1349
|
recommendation: `GC consumed ${value}% of executor run time: reduce object creation, use primitive types, avoid UDFs, increase executor memory.`,
|
|
1350
|
+
remediation: [increaseConf('spark.executor.memory')],
|
|
1035
1351
|
};
|
|
1036
1352
|
}
|
|
1037
1353
|
// Low-GC (cost) branch: only for stages that ran long enough to be meaningful.
|
|
1038
|
-
if ((stage.executorRunTime ?? 0) >=
|
|
1039
|
-
&& pct <
|
|
1040
|
-
&& !stageBelowRuntimeFloor(stage, ctx,
|
|
1354
|
+
if ((stage.executorRunTime ?? 0) >= thresholds.minRunTimeMs
|
|
1355
|
+
&& pct < thresholds.lowInfoPct100
|
|
1356
|
+
&& !stageBelowRuntimeFloor(stage, ctx, thresholds.lowInfoFloorPct)) {
|
|
1041
1357
|
const value = Math.round(pct * 10) / 10;
|
|
1042
1358
|
return {
|
|
1043
1359
|
type: 'gc', stageId: stage.id, direction: 'low',
|
|
1044
1360
|
impactBand: 'info',
|
|
1045
1361
|
metric: 'gcPct', value,
|
|
1046
|
-
confidence: gcConfidence(pct,
|
|
1362
|
+
confidence: gcConfidence(pct, thresholds, 'low'), validationRequired: gcValidation(thresholds.minRunTimeMs),
|
|
1047
1363
|
recommendation: `GC consumed only ${value}% of executor run time: memory may be over-provisioned; consider reducing spark.executor.memory for cost savings.`,
|
|
1364
|
+
remediation: [decreaseConf('spark.executor.memory')],
|
|
1048
1365
|
};
|
|
1049
1366
|
}
|
|
1050
1367
|
return null;
|
|
1051
1368
|
},
|
|
1052
|
-
|
|
1053
|
-
|
|
1054
|
-
|
|
1369
|
+
estimate(finding, ctx) {
|
|
1370
|
+
// The low-GC branch is an over-provisioning signal whose fix (less executor memory) raises
|
|
1371
|
+
// GC rather than recovering it: the stage's GC time is no saving there, so no waste model.
|
|
1372
|
+
if (finding.direction === 'low') return costOnly('none');
|
|
1373
|
+
if (finding.stageId == null) return null;
|
|
1374
|
+
const stage = ctx.stages.get(finding.stageId);
|
|
1375
|
+
if (!stage) return null;
|
|
1376
|
+
const stageDurationMs = (stage.completedAt ?? 0) - (stage.submittedAt ?? 0);
|
|
1377
|
+
const executorRunTime = stage.executorRunTime ?? 0;
|
|
1378
|
+
const jvmGCTime = stage.jvmGCTime ?? 0;
|
|
1379
|
+
// The raw cross-task core-time sum, before any conversion: the one figure here
|
|
1380
|
+
// that is straight from the log rather than modeled.
|
|
1381
|
+
const rawWaste = { value: jvmGCTime, unit: 'coreMs' } ;
|
|
1382
|
+
if (executorRunTime <= 0 || stageDurationMs <= 0) {
|
|
1383
|
+
return costOnly('modeled', rawWaste);
|
|
1384
|
+
}
|
|
1385
|
+
const avgConcurrency = executorRunTime / stageDurationMs;
|
|
1386
|
+
// jvmGCTime is a cross-task core-time sum (same shape as executorRunTime); dividing by the
|
|
1387
|
+
// stage's average concurrency converts it to an approximate wall-clock figure. Modeled, not exact.
|
|
1388
|
+
const wasteMs = jvmGCTime / avgConcurrency;
|
|
1389
|
+
return singleStageImpact(wasteMs, finding.stageId, ctx, 'modeled', rawWaste);
|
|
1390
|
+
},
|
|
1391
|
+
}),
|
|
1392
|
+
defineStageDetector({
|
|
1393
|
+
type: 'slowHost', order: 60, fixEffort: 'config', version: 1,
|
|
1394
|
+
emits: ['slowHost'],
|
|
1055
1395
|
docAnchor: '#bottleneck-slow-host',
|
|
1056
1396
|
thresholds: {
|
|
1057
1397
|
minHosts: 3, minTasks: 15, ratioWarn: 2.0, minShare: 0.20, shareWarn: 0.75, taskShareWarn: 0.50, ratioTiers: [1.33, 1.78, 3.16, 10],
|
|
@@ -1067,23 +1407,16 @@ export const DETECTORS = [
|
|
|
1067
1407
|
// floor is why they're dropped.
|
|
1068
1408
|
stageFloorPct: 0.005,
|
|
1069
1409
|
},
|
|
1070
|
-
detect(
|
|
1071
|
-
|
|
1072
|
-
|
|
1073
|
-
|
|
1074
|
-
|
|
1075
|
-
|
|
1076
|
-
|
|
1077
|
-
|
|
1078
|
-
stage ,
|
|
1079
|
-
ctx ,
|
|
1080
|
-
) {
|
|
1410
|
+
detect(stage, ctx, thresholds) {
|
|
1081
1411
|
const hosts = stage.hostStats ?? [];
|
|
1082
1412
|
const execs0 = stage.executorStats ?? [];
|
|
1083
|
-
if ((hosts.length <
|
|
1084
|
-
if (stageBelowRuntimeFloor(stage, ctx,
|
|
1413
|
+
if ((hosts.length < thresholds.minHosts && execs0.length < thresholds.minHosts) || stage.taskCount < thresholds.minTasks) return null;
|
|
1414
|
+
if (stageBelowRuntimeFloor(stage, ctx, thresholds.stageFloorPct)) return null;
|
|
1085
1415
|
const out = [];
|
|
1086
|
-
|
|
1416
|
+
const speculation = switchFix(loggedAs(ctx.app, 'spark.speculation', true), 'spark.speculation', true,
|
|
1417
|
+
'check what it was running, and consider enabling spark.speculation to relaunch a lagging task automatically',
|
|
1418
|
+
'check what it was running; speculation is already on, so a lagging task there is already relaunched');
|
|
1419
|
+
if (hosts.length >= thresholds.minHosts) {
|
|
1087
1420
|
const means = hosts.map(h => ({ host: h.host, taskCount: h.taskCount, mean: h.totalDuration / h.taskCount }));
|
|
1088
1421
|
const sorted = [...means].map(h => h.mean).sort((a, b) => a - b);
|
|
1089
1422
|
const overallMedian = sorted[Math.floor(sorted.length / 2)];
|
|
@@ -1091,7 +1424,7 @@ export const DETECTORS = [
|
|
|
1091
1424
|
for (const h of means) {
|
|
1092
1425
|
const ratio = h.mean / overallMedian;
|
|
1093
1426
|
const share = h.taskCount / stage.taskCount;
|
|
1094
|
-
if (ratio <
|
|
1427
|
+
if (ratio < thresholds.ratioWarn || share < thresholds.minShare || h.mean < thresholds.floorMs) continue;
|
|
1095
1428
|
out.push({
|
|
1096
1429
|
type: 'slowHost', stageId: stage.id,
|
|
1097
1430
|
impactBand: 'warning',
|
|
@@ -1099,7 +1432,8 @@ export const DETECTORS = [
|
|
|
1099
1432
|
// `value` is a ratio; the estimator needs the absolute per-host mean.
|
|
1100
1433
|
hostMeanMs: h.mean,
|
|
1101
1434
|
host: h.host, hostTaskShare: Math.round(share * 100) / 100,
|
|
1102
|
-
recommendation: `${h.host} may just hold data locality for its tasks or carry one heavy stage, not necessarily a hardware fault:
|
|
1435
|
+
recommendation: `${h.host} may just hold data locality for its tasks or carry one heavy stage, not necessarily a hardware fault: ${speculation.text}.`,
|
|
1436
|
+
remediation: speculation.remediation,
|
|
1103
1437
|
});
|
|
1104
1438
|
}
|
|
1105
1439
|
}
|
|
@@ -1108,7 +1442,7 @@ export const DETECTORS = [
|
|
|
1108
1442
|
for (const h of hosts) {
|
|
1109
1443
|
const durationShare = h.totalDuration / totalDuration;
|
|
1110
1444
|
const taskShare = h.taskCount / stage.taskCount;
|
|
1111
|
-
if (durationShare >=
|
|
1445
|
+
if (durationShare >= thresholds.shareWarn && taskShare >= thresholds.taskShareWarn) {
|
|
1112
1446
|
out.push({
|
|
1113
1447
|
type: 'slowHost', stageId: stage.id, impactBand: 'warning',
|
|
1114
1448
|
variant: 'durationShare',
|
|
@@ -1123,11 +1457,11 @@ export const DETECTORS = [
|
|
|
1123
1457
|
}
|
|
1124
1458
|
}
|
|
1125
1459
|
const execs = stage.executorStats ?? [];
|
|
1126
|
-
const tiers =
|
|
1460
|
+
const tiers = thresholds.ratioTiers;
|
|
1127
1461
|
const impactBandFor = (r ) =>
|
|
1128
1462
|
r >= tiers[3] ? 'critical' : (r >= tiers[1] ? 'warning' : (r >= tiers[0] ? 'info' : null));
|
|
1129
|
-
const floorMs =
|
|
1130
|
-
const dims
|
|
1463
|
+
const floorMs = thresholds.floorMs, floorBytes = thresholds.floorBytes;
|
|
1464
|
+
const dims = [
|
|
1131
1465
|
{ dimension: 'taskTime', floor: floorMs, samples: execs.filter(e => e.taskCount > 0).map(e => ({ key: e.executorId, value: e.totalDuration / e.taskCount })) },
|
|
1132
1466
|
{ dimension: 'inputBytes', floor: floorBytes, samples: execs.map(e => ({ key: e.executorId, value: e.inputBytes ?? 0 })) },
|
|
1133
1467
|
{ dimension: 'shuffleBytes', floor: floorBytes, samples: execs.map(e => ({ key: e.executorId, value: (e.shuffleReadBytes ?? 0) + (e.shuffleWriteBytes ?? 0) })) },
|
|
@@ -1155,31 +1489,46 @@ export const DETECTORS = [
|
|
|
1155
1489
|
// in this dimension's own unit (ms for taskTime, bytes for the rest).
|
|
1156
1490
|
execMaxValue: r.value,
|
|
1157
1491
|
executorId: r.key,
|
|
1158
|
-
recommendation: `Executor ${r.key}
|
|
1492
|
+
recommendation: `Executor ${r.key}'s ${SLOW_HOST_DIMENSION_LABEL[d.dimension]} is ${Math.round(r.ratio * 10) / 10}× the median: investigate uneven partition assignment or a degraded executor.`,
|
|
1159
1493
|
});
|
|
1160
1494
|
}
|
|
1161
1495
|
return out;
|
|
1162
1496
|
},
|
|
1163
|
-
|
|
1164
|
-
|
|
1165
|
-
|
|
1497
|
+
estimate(finding, ctx) {
|
|
1498
|
+
// Three duration-based shapes, each carrying its absolute-ms figure under a different field
|
|
1499
|
+
// (`value` is always a ratio/share, never ms): the per-host mean branch (discriminated by
|
|
1500
|
+
// `metric`), the duration-share branch (`variant`), and the multiDim taskTime dimension. Every
|
|
1501
|
+
// byte-based multiDim dimension has no absolute figure today, so it stays informational.
|
|
1502
|
+
const absoluteMs =
|
|
1503
|
+
finding.metric === 'hostMeanRatio' || finding.variant === 'durationShare'
|
|
1504
|
+
? (finding.hostMeanMs )
|
|
1505
|
+
: finding.variant === 'multiDim' && finding.dimension === 'taskTime'
|
|
1506
|
+
? (finding.execMaxValue )
|
|
1507
|
+
: null;
|
|
1508
|
+
if (absoluteMs == null) {
|
|
1509
|
+
return costOnly('none'); // byte-based multiDim dims: no absolute figure today, no model applied
|
|
1510
|
+
}
|
|
1511
|
+
if (finding.stageId == null) return null;
|
|
1512
|
+
const stage = ctx.stages.get(finding.stageId);
|
|
1513
|
+
if (!stage) return null;
|
|
1514
|
+
const wasteMs = Math.max(0, absoluteMs - (stage.taskDurationP50 ?? 0));
|
|
1515
|
+
return singleStageImpact(wasteMs, finding.stageId, ctx, 'measured', { value: wasteMs, unit: 'ms' });
|
|
1516
|
+
},
|
|
1517
|
+
}),
|
|
1518
|
+
defineStageDetector({
|
|
1519
|
+
type: 'stageSlowness', order: 65, fixEffort: 'code', version: 2,
|
|
1520
|
+
emits: ['stageSlowness'],
|
|
1166
1521
|
docAnchor: '#bottleneck-stage-slowness',
|
|
1167
1522
|
thresholds: { infoMin: 15 },
|
|
1168
|
-
//
|
|
1169
|
-
|
|
1170
|
-
|
|
1171
|
-
return out.some(o => o.type === 'slowHost' && o.stageId === finding.stageId);
|
|
1172
|
-
},
|
|
1173
|
-
detect(
|
|
1174
|
-
|
|
1175
|
-
stage ,
|
|
1176
|
-
) {
|
|
1523
|
+
// A stage slowHost already explains needs no generic "this stage is slow" finding on top.
|
|
1524
|
+
suppressedBy: 'slowHost',
|
|
1525
|
+
detect(stage, _ctx, thresholds) {
|
|
1177
1526
|
// Basis is real wall-clock stage duration, not per-executor average; the impact-estimator
|
|
1178
1527
|
// formula reuses this exact stageDurationMs computation.
|
|
1179
1528
|
const stageDurationMs = (stage.completedAt ?? 0) - (stage.submittedAt ?? 0);
|
|
1180
1529
|
if (!(stageDurationMs > 0)) return null;
|
|
1181
1530
|
const durationMinutes = stageDurationMs / 60000;
|
|
1182
|
-
const t =
|
|
1531
|
+
const t = thresholds;
|
|
1183
1532
|
const impactBand = durationMinutes >= t.infoMin ? 'info' : null;
|
|
1184
1533
|
if (!impactBand) return null;
|
|
1185
1534
|
const value = Math.round(durationMinutes * 10) / 10;
|
|
@@ -1187,38 +1536,60 @@ export const DETECTORS = [
|
|
|
1187
1536
|
type: 'stageSlowness', stageId: stage.id, impactBand,
|
|
1188
1537
|
metric: 'stageDurationMinutes', value,
|
|
1189
1538
|
recommendation: `This stage ran ${value} minutes with no more specific cause flagged: often a partition-count problem, raise parallelism via spark.sql.shuffle.partitions or spark.default.parallelism, or check for a large per-task data volume driving heavy shuffle and spill.`,
|
|
1539
|
+
remediation: [increaseConf('spark.sql.shuffle.partitions'), increaseConf('spark.default.parallelism')],
|
|
1190
1540
|
};
|
|
1191
1541
|
},
|
|
1192
|
-
|
|
1193
|
-
|
|
1194
|
-
|
|
1542
|
+
estimate(finding, ctx) {
|
|
1543
|
+
if (finding.stageId == null) return null;
|
|
1544
|
+
const stage = ctx.stages.get(finding.stageId);
|
|
1545
|
+
if (!stage) return null;
|
|
1546
|
+
// The recommended fix is more partitions, which only helps a stage that ran fewer tasks than
|
|
1547
|
+
// the cluster has cores: the time its tasks were running could then spread over up to
|
|
1548
|
+
// totalCores (lowShuffleParallelism's shape). Time the stage sat open with no task running
|
|
1549
|
+
// is queueing no partition count recovers. Splitting partitions splits the longest task
|
|
1550
|
+
// too, hence TAIL_CLAIM's post-fix floor. Unknown cluster size: no defensible figure.
|
|
1551
|
+
if (ctx.totalCores <= 0) return costOnly('modeled');
|
|
1552
|
+
// A stage that read no input and no shuffle, its tasks idle waiting on an external system,
|
|
1553
|
+
// gains nothing from more partitions (a 1-task JDBC count stage open 27 minutes on 5s of CPU
|
|
1554
|
+
// was claimed 99% recoverable): claim 0. Stages that read bytes keep their claim.
|
|
1555
|
+
const readBytes = (stage.inputBytes ?? 0) + (stage.shuffleReadBytes ?? 0);
|
|
1556
|
+
const activeMs = typeof stage.taskActiveMs === 'number'
|
|
1557
|
+
? stage.taskActiveMs
|
|
1558
|
+
: Math.max(0, (stage.completedAt ?? 0) - (stage.submittedAt ?? 0));
|
|
1559
|
+
const taskCount = stage.taskCount ?? 0;
|
|
1560
|
+
const wasteMs = readBytes <= 0 && tasksMostlyIdle(stage, ctx.sql) ? 0 : activeMs * Math.max(0, 1 - taskCount / ctx.totalCores);
|
|
1561
|
+
return singleStageImpact(wasteMs, finding.stageId, ctx, 'modeled', { value: wasteMs, unit: 'ms' }, TAIL_CLAIM);
|
|
1562
|
+
},
|
|
1563
|
+
}),
|
|
1564
|
+
defineStageDetector({
|
|
1565
|
+
type: 'stageFailed', order: 42, fixEffort: 'code', version: 1,
|
|
1566
|
+
emits: ['stageFailed'],
|
|
1195
1567
|
docAnchor: '#bottleneck-failures',
|
|
1196
1568
|
thresholds: {},
|
|
1197
|
-
detect(stage
|
|
1569
|
+
detect(stage) {
|
|
1198
1570
|
if (stage.stageFailureReason == null) return null;
|
|
1199
1571
|
return {
|
|
1200
1572
|
type: 'stageFailed', stageId: stage.id, impactBand: 'critical',
|
|
1201
1573
|
variant: 'stageFailure',
|
|
1202
|
-
metric: 'stageFailureReason',
|
|
1574
|
+
metric: 'stageFailureReason', valueText: stage.stageFailureReason,
|
|
1203
1575
|
numTasks: stage.taskCount,
|
|
1204
1576
|
memoryBytesSpilled: stage.memoryBytesSpilled,
|
|
1205
1577
|
failedTaskDetails: stage.failedTaskSamples ?? [],
|
|
1206
1578
|
recommendation: `This stage attempt failed outright. Inspect the driver log for the failure reason and the job that triggered it.`,
|
|
1207
1579
|
};
|
|
1208
1580
|
},
|
|
1209
|
-
|
|
1210
|
-
|
|
1211
|
-
|
|
1581
|
+
estimate: noWasteModel,
|
|
1582
|
+
}),
|
|
1583
|
+
defineStageDetector({
|
|
1584
|
+
type: 'failures', order: 40, fixEffort: 'code', version: 2,
|
|
1585
|
+
emits: ['failures'],
|
|
1212
1586
|
docAnchor: '#bottleneck-failures',
|
|
1213
1587
|
thresholds: { minTasks: 10, warnRate: 0.05, critRate: 0.20 },
|
|
1214
|
-
detect(
|
|
1215
|
-
|
|
1216
|
-
stage ,
|
|
1217
|
-
) {
|
|
1218
|
-
if (stage.taskCount < this.thresholds.minTasks) return null;
|
|
1588
|
+
detect(stage, _ctx, thresholds) {
|
|
1589
|
+
if (stage.taskCount < thresholds.minTasks) return null;
|
|
1219
1590
|
if (!stage.failedTasks) return null;
|
|
1220
1591
|
const failureRate = stage.failedTasks / stage.taskCount;
|
|
1221
|
-
if (failureRate <=
|
|
1592
|
+
if (failureRate <= thresholds.warnRate) return null;
|
|
1222
1593
|
const value = Math.round(failureRate * 1000) / 10;
|
|
1223
1594
|
const dominantReason = pickDominantReason(stage.failureReasons);
|
|
1224
1595
|
// Groups arrive most frequent first. Name the dominant error from the largest group under the
|
|
@@ -1230,7 +1601,7 @@ export const DETECTORS = [
|
|
|
1230
1601
|
const groupedTasks = failureGroups.reduce((sum, g) => sum + g.count, 0);
|
|
1231
1602
|
return {
|
|
1232
1603
|
type: 'failures', stageId: stage.id,
|
|
1233
|
-
impactBand: failureRate >
|
|
1604
|
+
impactBand: failureRate > thresholds.critRate ? 'critical' : 'warning',
|
|
1234
1605
|
metric: 'failureRate', value,
|
|
1235
1606
|
failedTasks: stage.failedTasks,
|
|
1236
1607
|
dominantReason,
|
|
@@ -1242,12 +1613,14 @@ export const DETECTORS = [
|
|
|
1242
1613
|
recommendation: `${value}% of tasks failed${dominantError ? ` (dominant error: ${dominantError})` : ''}: investigate driver logs for executor instability or data-driven errors.`,
|
|
1243
1614
|
};
|
|
1244
1615
|
},
|
|
1245
|
-
|
|
1246
|
-
|
|
1247
|
-
|
|
1616
|
+
estimate: noWasteModel,
|
|
1617
|
+
}),
|
|
1618
|
+
defineStageDetector({
|
|
1619
|
+
type: 'straggler', order: 70, fixEffort: 'code', version: 1,
|
|
1620
|
+
emits: ['straggler'],
|
|
1248
1621
|
docAnchor: '#bottleneck-straggler',
|
|
1249
|
-
// floorPctWarn/floorPctCrit
|
|
1250
|
-
//
|
|
1622
|
+
// floorPctWarn/floorPctCrit default to impact-band.ts's run-wide noise floor, so a tail this
|
|
1623
|
+
// gate admits at its warn floor grades at least warning there too.
|
|
1251
1624
|
// shareWarnAtFloor: in a large stage, the few stragglers that gate it for tens of seconds can be
|
|
1252
1625
|
// only 2.5-5% of its tasks. Scored against a task-level replay of every stage on 14 real logs
|
|
1253
1626
|
// (recoverable = replay with each task over 4x P50 capped at P50), admitting 2.5-5% shares
|
|
@@ -1257,40 +1630,27 @@ export const DETECTORS = [
|
|
|
1257
1630
|
// than the stage's own duration, so every finding there graded info. On the 14 real logs that
|
|
1258
1631
|
// was 671 of 753 straggler findings, none above info; the slow tail is still real on those
|
|
1259
1632
|
// stages, the floor is why they're dropped.
|
|
1260
|
-
thresholds: { minTasks: 10, shareWarn: 0.05, shareWarnAtFloor: 0.025, warnPct: 0.10, critPct: 0.20, floorPctWarn:
|
|
1261
|
-
detect(
|
|
1262
|
-
|
|
1263
|
-
|
|
1264
|
-
|
|
1265
|
-
|
|
1266
|
-
|
|
1267
|
-
|
|
1268
|
-
stage ,
|
|
1269
|
-
ctx ,
|
|
1270
|
-
) {
|
|
1271
|
-
if (stage.taskCount < this.thresholds.minTasks) return null;
|
|
1272
|
-
const appDurationMs = computeAppDurationMs(ctx);
|
|
1273
|
-
if (stageBelowRuntimeFloor(stage, ctx, this.thresholds.floorPctWarn)) return null;
|
|
1633
|
+
thresholds: { minTasks: 10, shareWarn: 0.05, shareWarnAtFloor: 0.025, warnPct: 0.10, critPct: 0.20, floorPctWarn: IMPACT_FLOOR_PCT_WARN, floorPctCrit: IMPACT_FLOOR_PCT_CRIT },
|
|
1634
|
+
detect(stage, ctx, thresholds) {
|
|
1635
|
+
if (stage.taskCount < thresholds.minTasks) return null;
|
|
1636
|
+
const runMs = appDurationMs(ctx.app);
|
|
1637
|
+
if (stageBelowRuntimeFloor(stage, ctx, thresholds.floorPctWarn)) return null;
|
|
1274
1638
|
const stragglerShare = (stage.stragglerCount ?? 0) / stage.taskCount;
|
|
1275
1639
|
const useSpeculative = (stage.speculativeTasks ?? 0) > 0;
|
|
1276
|
-
if (!useSpeculative && stragglerShare <=
|
|
1640
|
+
if (!useSpeculative && stragglerShare <= thresholds.shareWarnAtFloor) return null;
|
|
1277
1641
|
const speculativeShare = useSpeculative ? stage.speculativeTasks / stage.taskCount : 0;
|
|
1278
|
-
//
|
|
1279
|
-
//
|
|
1280
|
-
|
|
1281
|
-
const
|
|
1282
|
-
const
|
|
1283
|
-
const wasteMs = tailRecoveryMs(stage, singleDelta);
|
|
1284
|
-
const floorWasteMs = clippedWasteMs(wasteMs, stage.id, ctx, tailRemovedWorkMs(stage, singleDelta), longestTaskAfterFixMs);
|
|
1285
|
-
const meetsWarnFloor = meetsRuntimeFloor(floorWasteMs, appDurationMs, this.thresholds.floorPctWarn);
|
|
1286
|
-
const meetsCritFloor = meetsRuntimeFloor(floorWasteMs, appDurationMs, this.thresholds.floorPctCrit);
|
|
1642
|
+
// The claim estimate() reports as savings: a high straggler/speculative share on a stage
|
|
1643
|
+
// whose tasks barely vary models near-zero savings, so it must not outrank 'info'.
|
|
1644
|
+
const floorWasteMs = tailClaimFloorMs(stragglerTailClaim(stage), stage.id, ctx);
|
|
1645
|
+
const meetsWarnFloor = meetsRuntimeFloor(floorWasteMs, runMs, thresholds.floorPctWarn);
|
|
1646
|
+
const meetsCritFloor = meetsRuntimeFloor(floorWasteMs, runMs, thresholds.floorPctCrit);
|
|
1287
1647
|
// The lower gate needs positive evidence the tail matters: meetsRuntimeFloor passes by
|
|
1288
1648
|
// default when the app's duration is unknown (an incomplete run), which isn't that.
|
|
1289
|
-
const stragglerShareFires = stragglerShare >
|
|
1290
|
-
|| (stragglerShare >
|
|
1649
|
+
const stragglerShareFires = stragglerShare > thresholds.shareWarn
|
|
1650
|
+
|| (stragglerShare > thresholds.shareWarnAtFloor && runMs != null && meetsWarnFloor);
|
|
1291
1651
|
if (!useSpeculative && !stragglerShareFires) return null;
|
|
1292
|
-
const speculativeTier = speculativeShare >=
|
|
1293
|
-
: speculativeShare >=
|
|
1652
|
+
const speculativeTier = speculativeShare >= thresholds.critPct && meetsCritFloor ? 'critical'
|
|
1653
|
+
: speculativeShare >= thresholds.warnPct && meetsWarnFloor ? 'warning' : 'info';
|
|
1294
1654
|
// Straggler share has no dedicated critical tier per detector-contract.md; only warning.
|
|
1295
1655
|
const stragglerTier = stragglerShareFires && meetsWarnFloor ? 'warning' : 'info';
|
|
1296
1656
|
// Fixed fallback: overwritten by deriveImpactBand when this finding gets a real wallClock
|
|
@@ -1312,46 +1672,54 @@ export const DETECTORS = [
|
|
|
1312
1672
|
speculativeTasks: stage.speculativeTasks ?? 0,
|
|
1313
1673
|
stragglerCount: stage.stragglerCount ?? 0,
|
|
1314
1674
|
confidence: useSpeculativeMetric
|
|
1315
|
-
? stragglerConfidence(speculativeShare,
|
|
1316
|
-
: stragglerConfidence(stragglerShare,
|
|
1317
|
-
validationRequired:
|
|
1675
|
+
? stragglerConfidence(speculativeShare, thresholds.warnPct, thresholds.critPct)
|
|
1676
|
+
: stragglerConfidence(stragglerShare, thresholds.shareWarn, thresholds.critPct),
|
|
1677
|
+
validationRequired: `Warning needs at least ${shareLabel(thresholds.floorPctWarn)} of run time at stake, critical ${shareLabel(thresholds.floorPctCrit)}.`,
|
|
1318
1678
|
recommendation: `${detail}: rule out a GC pause or a slow shuffle fetch before assuming a hardware issue; if a skewed key is the real cause, that's a candidate for AQE's skew-join handling.`,
|
|
1319
1679
|
};
|
|
1320
1680
|
},
|
|
1321
|
-
|
|
1322
|
-
|
|
1323
|
-
|
|
1681
|
+
estimate(finding, ctx) {
|
|
1682
|
+
if (finding.stageId == null) return null;
|
|
1683
|
+
const stage = ctx.stages.get(finding.stageId);
|
|
1684
|
+
if (!stage) return null;
|
|
1685
|
+
return tailClaimImpact(stragglerTailClaim(stage), finding.stageId, ctx);
|
|
1686
|
+
},
|
|
1687
|
+
}),
|
|
1688
|
+
defineStageDetector({
|
|
1689
|
+
type: 'speculationWaste', order: 71, fixEffort: 'config', version: 1,
|
|
1690
|
+
emits: ['speculationWaste'],
|
|
1324
1691
|
docAnchor: '#bottleneck-speculation-waste',
|
|
1325
1692
|
thresholds: { minWasted: 5, minWasteMs: 60000 },
|
|
1326
|
-
detect(
|
|
1327
|
-
|
|
1328
|
-
|
|
1329
|
-
|
|
1330
|
-
stage ,
|
|
1331
|
-
) {
|
|
1693
|
+
detect(stage, _ctx, thresholds) {
|
|
1332
1694
|
const wasted = stage.speculationWastedAttempts ?? 0;
|
|
1333
1695
|
const wastedMs = stage.speculationWasteMs ?? 0;
|
|
1334
|
-
if (wasted <
|
|
1696
|
+
if (wasted < thresholds.minWasted || wastedMs < thresholds.minWasteMs) return null;
|
|
1335
1697
|
return {
|
|
1336
1698
|
type: 'speculationWaste', stageId: stage.id,
|
|
1337
1699
|
impactBand: 'warning',
|
|
1338
1700
|
metric: 'speculationWasteMs', value: wastedMs,
|
|
1339
|
-
confidence: speculationWasteConfidence(wastedMs,
|
|
1340
|
-
recommendation: `Speculative execution discarded ${Math.round(wastedMs / 1000)}s of executor time in this stage
|
|
1701
|
+
confidence: speculationWasteConfidence(wastedMs, thresholds.minWasteMs),
|
|
1702
|
+
recommendation: `Speculative execution discarded ${Math.round(wastedMs / 1000)}s of executor time in this stage: if task durations are naturally variable rather than genuine stragglers, consider tuning spark.speculation.multiplier/quantile.`,
|
|
1703
|
+
remediation: [increaseConf('spark.speculation.multiplier'), increaseConf('spark.speculation.quantile')],
|
|
1341
1704
|
};
|
|
1342
1705
|
},
|
|
1343
|
-
|
|
1344
|
-
|
|
1345
|
-
|
|
1706
|
+
estimate(finding, ctx) {
|
|
1707
|
+
if (finding.stageId == null) return null;
|
|
1708
|
+
const stage = ctx.stages.get(finding.stageId);
|
|
1709
|
+
if (!stage) return null;
|
|
1710
|
+
const wasteMs = (stage.speculationWasteMs ) ?? 0;
|
|
1711
|
+
return singleStageImpact(wasteMs, finding.stageId, ctx, 'measured', { value: wasteMs, unit: 'ms' });
|
|
1712
|
+
},
|
|
1713
|
+
}),
|
|
1714
|
+
defineStageDetector({
|
|
1715
|
+
type: 'retryWaste', order: 45, fixEffort: 'code', version: 1,
|
|
1716
|
+
emits: ['retryWaste'],
|
|
1346
1717
|
docAnchor: '#bottleneck-retry-waste',
|
|
1347
1718
|
thresholds: { minWasted: 3, minWasteMs: 30000 },
|
|
1348
|
-
detect(
|
|
1349
|
-
|
|
1350
|
-
stage ,
|
|
1351
|
-
) {
|
|
1719
|
+
detect(stage, _ctx, thresholds) {
|
|
1352
1720
|
const wasted = stage.wastedAttempts ?? 0;
|
|
1353
1721
|
const wastedMs = stage.retryWasteMs ?? 0;
|
|
1354
|
-
if (wasted <
|
|
1722
|
+
if (wasted < thresholds.minWasted || wastedMs < thresholds.minWasteMs) return null;
|
|
1355
1723
|
return {
|
|
1356
1724
|
type: 'retryWaste', stageId: stage.id,
|
|
1357
1725
|
impactBand: 'warning',
|
|
@@ -1363,22 +1731,29 @@ export const DETECTORS = [
|
|
|
1363
1731
|
extended: `${wasted} task attempts were superseded by a later retry, wasting ${Math.round(wastedMs / 1000)}s of executor time. Common causes: executor loss (OOM-kill, node death) or shuffle FetchFailed forcing a stage-map recompute. Check driver logs for the dominant reason (see the Failures widget) even if the final failure rate looks low; retries hide the true cost.`,
|
|
1364
1732
|
};
|
|
1365
1733
|
},
|
|
1366
|
-
|
|
1367
|
-
|
|
1368
|
-
|
|
1734
|
+
estimate(finding, ctx) {
|
|
1735
|
+
// The waste figure lives on the Stage, not the Finding: detect() only re-publishes it as metric/value.
|
|
1736
|
+
if (finding.stageId == null) return null;
|
|
1737
|
+
const stage = ctx.stages.get(finding.stageId);
|
|
1738
|
+
if (!stage) return null;
|
|
1739
|
+
const wasteMs = (stage.retryWasteMs ) ?? 0;
|
|
1740
|
+
const wallClockMs = retryWallClockMs(stage);
|
|
1741
|
+
return singleStageImpact(wallClockMs, finding.stageId, ctx,
|
|
1742
|
+
wallClockMs === wasteMs ? 'measured' : 'modeled', { value: wasteMs, unit: 'ms' });
|
|
1743
|
+
},
|
|
1744
|
+
}),
|
|
1745
|
+
defineStageDetector({
|
|
1746
|
+
type: 'tinyTask', order: 80, fixEffort: 'code', version: 1,
|
|
1747
|
+
emits: ['tinyTask'],
|
|
1369
1748
|
docAnchor: '#bottleneck-tiny-tasks',
|
|
1370
1749
|
// stageFloorPct: the tiered detectors' 0.5% runtime floor. Coalescing can't save more than the
|
|
1371
1750
|
// stage's own duration, so on a shorter stage every finding graded info: on the 14 real logs
|
|
1372
1751
|
// 132 of 151 tinyTask findings. The tasks are still tiny there; the floor is why they're dropped.
|
|
1373
1752
|
thresholds: { minTasks: 100, maxP50: 500, maxP95: 1000, stageFloorPct: 0.005 },
|
|
1374
|
-
detect(
|
|
1375
|
-
|
|
1376
|
-
stage
|
|
1377
|
-
|
|
1378
|
-
) {
|
|
1379
|
-
if (stage.taskCount < this.thresholds.minTasks) return null;
|
|
1380
|
-
if (stageBelowRuntimeFloor(stage, ctx, this.thresholds.stageFloorPct)) return null;
|
|
1381
|
-
if (stage.taskDurationP50 > this.thresholds.maxP50 || stage.taskDurationP95 > this.thresholds.maxP95) return null;
|
|
1753
|
+
detect(stage, ctx, thresholds) {
|
|
1754
|
+
if (stage.taskCount < thresholds.minTasks) return null;
|
|
1755
|
+
if (stageBelowRuntimeFloor(stage, ctx, thresholds.stageFloorPct)) return null;
|
|
1756
|
+
if (stage.taskDurationP50 > thresholds.maxP50 || stage.taskDurationP95 > thresholds.maxP95) return null;
|
|
1382
1757
|
const coalesceTo = Math.max(1, Math.round(stage.taskCount / 10));
|
|
1383
1758
|
const fix = stage.shuffleReadBytes > 0
|
|
1384
1759
|
? `lower spark.sql.shuffle.partitions or .coalesce(${coalesceTo})`
|
|
@@ -1387,32 +1762,49 @@ export const DETECTORS = [
|
|
|
1387
1762
|
type: 'tinyTask', stageId: stage.id, impactBand: 'info',
|
|
1388
1763
|
metric: 'taskDurationP50', value: Math.round(stage.taskDurationP50),
|
|
1389
1764
|
recommendation: `Many small tasks (${stage.taskCount}, P50 ${Math.round(stage.taskDurationP50)}ms): scheduler overhead may dominate. Try ${fix}.`,
|
|
1765
|
+
remediation: stage.shuffleReadBytes > 0 ? [decreaseConf('spark.sql.shuffle.partitions')] : [],
|
|
1390
1766
|
};
|
|
1391
1767
|
},
|
|
1392
|
-
|
|
1393
|
-
|
|
1768
|
+
estimate(finding, ctx) {
|
|
1769
|
+
if (finding.stageId == null) return null;
|
|
1770
|
+
const stage = ctx.stages.get(finding.stageId);
|
|
1771
|
+
if (!stage) return null;
|
|
1772
|
+
const taskCount = stage.taskCount ?? 0;
|
|
1773
|
+
const excessTaskCount = Math.max(0, taskCount - Math.round(taskCount / 10));
|
|
1774
|
+
const measured = measuredTaskOverhead(stage);
|
|
1775
|
+
if (measured) {
|
|
1776
|
+
// Coalescing to a tenth of the tasks removes the excess tasks' per-task overhead: core
|
|
1777
|
+
// time spent in parallel, so wall-clock at the stage's achieved concurrency (floored at
|
|
1778
|
+
// 1: a mostly-idle stage can't save more wall-clock than the task time it removes).
|
|
1779
|
+
const wasteMs = (excessTaskCount * measured.perTaskMs) / Math.max(1, measured.concurrency);
|
|
1780
|
+
return singleStageImpact(wasteMs, finding.stageId, ctx, 'measured', { value: wasteMs, unit: 'ms' });
|
|
1781
|
+
}
|
|
1782
|
+
const wasteMs = excessTaskCount * TASK_SCHEDULING_OVERHEAD_MS;
|
|
1783
|
+
return singleStageImpact(wasteMs, finding.stageId, ctx, 'modeled', { value: wasteMs, unit: 'ms' });
|
|
1784
|
+
},
|
|
1785
|
+
}),
|
|
1786
|
+
defineAppDetector({
|
|
1394
1787
|
// No docAnchor: the upstream spark-tuning-reference docs have no section for this
|
|
1395
1788
|
// tool-specific "capture stopped early" signal.
|
|
1396
|
-
type: 'incompleteRun',
|
|
1789
|
+
type: 'incompleteRun', order: 5, fixEffort: 'code', version: 1,
|
|
1790
|
+
emits: ['incompleteRun'],
|
|
1397
1791
|
thresholds: {},
|
|
1398
|
-
|
|
1399
|
-
detect( ctx ) {
|
|
1792
|
+
detect(ctx) {
|
|
1400
1793
|
if (!ctx.app || ctx.app.startTime == null || ctx.app.endTime != null) return null;
|
|
1401
1794
|
return {
|
|
1402
1795
|
type: 'incompleteRun', stageId: null, impactBand: 'warning',
|
|
1403
|
-
metric: 'applicationEnd',
|
|
1404
|
-
recommendation:
|
|
1796
|
+
metric: 'applicationEnd', valueText: 'missing',
|
|
1797
|
+
recommendation: INCOMPLETE_RUN_RECOMMENDATION,
|
|
1405
1798
|
};
|
|
1406
1799
|
},
|
|
1407
|
-
|
|
1408
|
-
|
|
1409
|
-
|
|
1800
|
+
estimate: noWasteModel,
|
|
1801
|
+
}),
|
|
1802
|
+
defineAppDetector({
|
|
1803
|
+
type: 'coldStart', order: 90, fixEffort: 'code', version: 1,
|
|
1804
|
+
emits: ['coldStart'],
|
|
1410
1805
|
docAnchor: '#bottleneck-cold-start',
|
|
1411
1806
|
thresholds: { gapSeconds: 30 },
|
|
1412
|
-
detect(
|
|
1413
|
-
|
|
1414
|
-
ctx ,
|
|
1415
|
-
) {
|
|
1807
|
+
detect(ctx, thresholds) {
|
|
1416
1808
|
const { app, stages, executorsAdded, executorsRemoved } = ctx;
|
|
1417
1809
|
// Nullish (not falsy) check: a literal startTime:0 must not be treated as "missing".
|
|
1418
1810
|
if (!app || app.startTime == null || stages.size === 0) return null;
|
|
@@ -1443,23 +1835,30 @@ export const DETECTORS = [
|
|
|
1443
1835
|
}
|
|
1444
1836
|
if (!Number.isFinite(firstExecutorAdded)) return null;
|
|
1445
1837
|
const gapSeconds = (firstExecutorAdded - firstStageSubmitted) / 1000;
|
|
1446
|
-
if (gapSeconds <=
|
|
1838
|
+
if (gapSeconds <= thresholds.gapSeconds) return null;
|
|
1447
1839
|
const value = Math.round(gapSeconds);
|
|
1448
1840
|
return {
|
|
1449
1841
|
type: 'coldStart', stageId: null, impactBand: 'warning',
|
|
1450
1842
|
metric: 'startupGapSeconds', value,
|
|
1451
1843
|
recommendation: `The first stage waited ${value}s for an executor to become available: keep a warm pool of idle executors, or if using dynamic allocation, raise the minimum/initial executor count so it doesn't scale up from zero.`,
|
|
1844
|
+
remediation: dynamicAllocationOff(ctx.app) ? [] : [increaseConf('spark.dynamicAllocation.minExecutors'), increaseConf('spark.dynamicAllocation.initialExecutors')],
|
|
1452
1845
|
};
|
|
1453
1846
|
},
|
|
1454
|
-
|
|
1455
|
-
|
|
1456
|
-
|
|
1847
|
+
estimate(finding) {
|
|
1848
|
+
// detect() reports the gap as `metric: 'startupGapSeconds', value: <seconds>`.
|
|
1849
|
+
if (typeof finding.value !== 'number') return null;
|
|
1850
|
+
const wasteMs = finding.value * 1000;
|
|
1851
|
+
// Time before any task starts can never overlap any stage; a genuine unclipped point estimate,
|
|
1852
|
+
// not tied to any stage's gate (coldStart is app-scoped, stageId: null).
|
|
1853
|
+
return { basis: 'serial', wallClock: { low: wasteMs, high: wasteMs }, estimateMethod: 'measured' };
|
|
1854
|
+
},
|
|
1855
|
+
}),
|
|
1856
|
+
defineAppDetector({
|
|
1857
|
+
type: 'utilization', order: 100, fixEffort: 'config', version: 1,
|
|
1858
|
+
emits: ['utilization'],
|
|
1457
1859
|
docAnchor: '#bottleneck-utilization',
|
|
1458
1860
|
thresholds: { minUtil: 0.60 },
|
|
1459
|
-
detect(
|
|
1460
|
-
|
|
1461
|
-
ctx ,
|
|
1462
|
-
) {
|
|
1861
|
+
detect(ctx, thresholds) {
|
|
1463
1862
|
const { app, executorsAdded, executorsRemoved, runAggregates } = ctx;
|
|
1464
1863
|
// Nullish (not falsy) check: a literal startTime:0 must not be treated as "missing".
|
|
1465
1864
|
if (!app || executorsAdded.length === 0 || app.startTime == null || app.endTime == null) return null;
|
|
@@ -1478,18 +1877,19 @@ export const DETECTORS = [
|
|
|
1478
1877
|
// lifetime-based measure this replaces.
|
|
1479
1878
|
const busyCoreMs = runAggregates?.busyCoreMs ?? 0;
|
|
1480
1879
|
const utilization = busyCoreMs / capacityCoreMs;
|
|
1481
|
-
if (utilization >=
|
|
1880
|
+
if (utilization >= thresholds.minUtil) return null;
|
|
1482
1881
|
|
|
1483
1882
|
// CPU-time-based utilization (sparkMeasure): metric only, no threshold.
|
|
1484
1883
|
let cpuUtilizationPct = null;
|
|
1485
1884
|
if (totalCores > 0) {
|
|
1486
|
-
|
|
1487
|
-
|
|
1488
|
-
|
|
1489
|
-
cpuUtilizationPct = Math.round((cpuMs / (appDuration * totalCores)) * 100);
|
|
1885
|
+
// Null when no stage recorded CPU time (older Spark), the same rule as the CLI metrics block.
|
|
1886
|
+
const cpuMs = totalExecutorCpuMs(ctx.stages.values());
|
|
1887
|
+
cpuUtilizationPct = cpuMs == null ? null : Math.round((cpuMs / (appDuration * totalCores)) * 100);
|
|
1490
1888
|
}
|
|
1491
1889
|
|
|
1492
1890
|
const value = Math.round(utilization * 100);
|
|
1891
|
+
const fix = dynamicAllocationFix(app, 'consider reducing cluster size or enabling dynamic allocation',
|
|
1892
|
+
'dynamic allocation is already on, so consider reducing cluster size');
|
|
1493
1893
|
return {
|
|
1494
1894
|
type: 'utilization', stageId: null, impactBand: 'info',
|
|
1495
1895
|
metric: 'avgUtilization', value,
|
|
@@ -1497,12 +1897,24 @@ export const DETECTORS = [
|
|
|
1497
1897
|
appDurationMs: appDuration,
|
|
1498
1898
|
totalCores,
|
|
1499
1899
|
cpuUtilizationPct,
|
|
1500
|
-
recommendation: `Average executor utilization was only ${value}%:
|
|
1900
|
+
recommendation: `Average executor utilization was only ${value}%: ${fix.text}.`,
|
|
1901
|
+
remediation: fix.remediation,
|
|
1501
1902
|
};
|
|
1502
1903
|
},
|
|
1503
|
-
|
|
1504
|
-
|
|
1505
|
-
|
|
1904
|
+
estimate(finding) {
|
|
1905
|
+
const fraction = finding.utilizationFraction ;
|
|
1906
|
+
const appDurationMs = finding.appDurationMs ;
|
|
1907
|
+
const totalCores = finding.totalCores ;
|
|
1908
|
+
if (fraction == null || appDurationMs == null || totalCores == null) {
|
|
1909
|
+
return costOnly('measured');
|
|
1910
|
+
}
|
|
1911
|
+
const idleCoreHours = (1 - fraction) * appDurationMs * totalCores / 3.6e6;
|
|
1912
|
+
return costOnly('measured', { value: idleCoreHours, unit: 'coreHours', idle: true });
|
|
1913
|
+
},
|
|
1914
|
+
}),
|
|
1915
|
+
defineAppDetector({
|
|
1916
|
+
type: 'memoryUtilization', order: 102, fixEffort: 'config', version: 1,
|
|
1917
|
+
emits: ['memoryUtilization'],
|
|
1506
1918
|
docAnchor: '#bottleneck-memory-utilization',
|
|
1507
1919
|
thresholds: {
|
|
1508
1920
|
idleCoreWarn: 0.50, // WastedCoresAlertsReducer
|
|
@@ -1510,14 +1922,7 @@ export const DETECTORS = [
|
|
|
1510
1922
|
bandTooHigh: 0.70, // below this => over-provisioned (cost signal)
|
|
1511
1923
|
wasteBufferMultiplier: 1.5, // UNVERIFIED
|
|
1512
1924
|
},
|
|
1513
|
-
detect(
|
|
1514
|
-
|
|
1515
|
-
|
|
1516
|
-
|
|
1517
|
-
|
|
1518
|
-
|
|
1519
|
-
ctx ,
|
|
1520
|
-
) {
|
|
1925
|
+
detect(ctx, thresholds) {
|
|
1521
1926
|
const { app, executorsAdded, executorsRemoved, runAggregates, stages } = ctx;
|
|
1522
1927
|
const out = [];
|
|
1523
1928
|
// Nullish (not falsy) check: a literal startTime:0 must not be treated as "missing".
|
|
@@ -1535,17 +1940,21 @@ export const DETECTORS = [
|
|
|
1535
1940
|
const allocatedMB = app.resources?.executor?.memoryMB ?? null;
|
|
1536
1941
|
|
|
1537
1942
|
// ── 1a idle-cores rate ────────────────────────────────────────────────
|
|
1538
|
-
|
|
1943
|
+
const busyCoreMs = runAggregates?.busyCoreMs;
|
|
1944
|
+
if (busyCoreMs != null && totalCores > 0) {
|
|
1539
1945
|
const capacityCoreMs = totalCores * appDurationMs;
|
|
1540
|
-
const idleRate = capacityCoreMs > 0 ? 1 - (
|
|
1541
|
-
if (idleRate >
|
|
1946
|
+
const idleRate = capacityCoreMs > 0 ? 1 - (busyCoreMs / capacityCoreMs) : 0;
|
|
1947
|
+
if (idleRate > thresholds.idleCoreWarn) {
|
|
1542
1948
|
const value = Math.round(idleRate * 100);
|
|
1949
|
+
const fix = dynamicAllocationFix(app, 'reduce cluster size or enable dynamic allocation',
|
|
1950
|
+
'dynamic allocation is already on, so reduce cluster size');
|
|
1543
1951
|
out.push({
|
|
1544
1952
|
type: 'memoryUtilization', variant: 'idleCores', stageId: null,
|
|
1545
1953
|
impactBand: 'warning', metric: 'idleCoreRate', value,
|
|
1546
1954
|
// Raw (unrounded) rate plus sizing inputs for the impact estimator: `value` is rounded pct.
|
|
1547
1955
|
idleRateFraction: idleRate, allocatedMB, peakExecutors, appDurationMs,
|
|
1548
|
-
recommendation: `${value}% of
|
|
1956
|
+
recommendation: `${value}% of available core-time ran no task: ${fix.text}.`,
|
|
1957
|
+
remediation: fix.remediation,
|
|
1549
1958
|
});
|
|
1550
1959
|
}
|
|
1551
1960
|
}
|
|
@@ -1562,10 +1971,15 @@ export const DETECTORS = [
|
|
|
1562
1971
|
}
|
|
1563
1972
|
}
|
|
1564
1973
|
if (peakHeapByExec.size === 0) {
|
|
1974
|
+
const key = 'spark.eventLog.logStageExecutorMetrics';
|
|
1975
|
+
const fix = switchFix(loggedAs(app, key, true), key, true,
|
|
1976
|
+
`Per-executor memory usage requires ${key}=true: not enabled for this run.`,
|
|
1977
|
+
'Per-executor memory usage is missing from this log even though executor metrics logging is on for this run.');
|
|
1565
1978
|
out.push({
|
|
1566
1979
|
type: 'memoryUtilization', variant: 'memoryBand', stageId: null,
|
|
1567
1980
|
impactBand: 'info', metric: 'memoryBand', dataUnavailable: true,
|
|
1568
|
-
recommendation:
|
|
1981
|
+
recommendation: fix.text,
|
|
1982
|
+
remediation: fix.remediation,
|
|
1569
1983
|
});
|
|
1570
1984
|
} else if (allocatedMB != null && allocatedMB > 0) {
|
|
1571
1985
|
const allocatedBytes = allocatedMB * 1024 * 1024;
|
|
@@ -1573,14 +1987,15 @@ export const DETECTORS = [
|
|
|
1573
1987
|
const ratio = heap / allocatedBytes;
|
|
1574
1988
|
// The two bands are opposite signals: an explicit `rule` discriminator lets consumers
|
|
1575
1989
|
// tell OOM-risk from over-provisioning without re-deriving the ratio.
|
|
1576
|
-
if (ratio >
|
|
1990
|
+
if (ratio > thresholds.bandTooSmall) {
|
|
1577
1991
|
out.push({
|
|
1578
1992
|
type: 'memoryUtilization', variant: 'memoryBand', rule: 'heapNearCapacity',
|
|
1579
1993
|
stageId: null, executorId: execId,
|
|
1580
1994
|
impactBand: 'warning', metric: 'heapUsedRatio', value: Math.round(ratio * 100),
|
|
1581
1995
|
recommendation: `Executor ${execId} peaked at ${Math.round(ratio * 100)}% of allocated heap: memory may be too small; raise spark.executor.memory to avoid OOM/spill.`,
|
|
1996
|
+
remediation: [increaseConf('spark.executor.memory')],
|
|
1582
1997
|
});
|
|
1583
|
-
} else if (ratio <
|
|
1998
|
+
} else if (ratio < thresholds.bandTooHigh) {
|
|
1584
1999
|
out.push({
|
|
1585
2000
|
type: 'memoryUtilization', variant: 'memoryBand', rule: 'heapOverProvisioned',
|
|
1586
2001
|
stageId: null, executorId: execId,
|
|
@@ -1589,6 +2004,7 @@ export const DETECTORS = [
|
|
|
1589
2004
|
// unused-memory-over-time model.
|
|
1590
2005
|
allocatedBytes, heap, appDurationMs,
|
|
1591
2006
|
recommendation: `Executor ${execId} used only ${Math.round(ratio * 100)}% of allocated heap: memory may be over-provisioned; consider reducing spark.executor.memory for cost savings.`,
|
|
2007
|
+
remediation: [decreaseConf('spark.executor.memory')],
|
|
1592
2008
|
});
|
|
1593
2009
|
}
|
|
1594
2010
|
}
|
|
@@ -1601,111 +2017,168 @@ export const DETECTORS = [
|
|
|
1601
2017
|
for (const s of stages.values()) usedRunTimeMs += s.executorRunTime ?? 0;
|
|
1602
2018
|
const usedMBSeconds = allocatedMB * (usedRunTimeMs / 1000);
|
|
1603
2019
|
const wastedMBSeconds = allocatedMBSeconds - usedMBSeconds;
|
|
1604
|
-
if (wastedMBSeconds >
|
|
2020
|
+
if (wastedMBSeconds > thresholds.wasteBufferMultiplier * usedMBSeconds) {
|
|
1605
2021
|
const value = Math.round(wastedMBSeconds);
|
|
1606
2022
|
out.push({
|
|
1607
2023
|
type: 'memoryUtilization', variant: 'wasteModel', stageId: null,
|
|
1608
2024
|
impactBand: 'info', metric: 'wastedMBSeconds', value,
|
|
1609
|
-
confidence: memoryWasteConfidence(wastedMBSeconds, usedMBSeconds,
|
|
1610
|
-
validationRequired:
|
|
2025
|
+
confidence: memoryWasteConfidence(wastedMBSeconds, usedMBSeconds, thresholds.wasteBufferMultiplier),
|
|
2026
|
+
validationRequired: `Memory-waste estimate uses allocated-vs-used memory-time and a ${thresholds.wasteBufferMultiplier}x buffer: confirm against the Spark UI before acting.`,
|
|
1611
2027
|
recommendation: `Allocated executor memory sat largely idle over the run (~${value.toLocaleString('en-US')} MB-seconds wasted): review spark.executor.memory and executor count.`,
|
|
2028
|
+
remediation: [decreaseConf('spark.executor.memory')],
|
|
1612
2029
|
});
|
|
1613
2030
|
}
|
|
1614
2031
|
}
|
|
1615
2032
|
|
|
1616
2033
|
return out;
|
|
1617
2034
|
},
|
|
1618
|
-
|
|
1619
|
-
|
|
2035
|
+
estimate(finding) {
|
|
2036
|
+
// The wasteModel variant reports metric: 'wastedMBSeconds', value: <MB-seconds>.
|
|
2037
|
+
if (finding.variant === 'wasteModel' && typeof finding.value === 'number') {
|
|
2038
|
+
return costOnly('measured', { value: finding.value, unit: 'mbSeconds' });
|
|
2039
|
+
}
|
|
2040
|
+
if (finding.variant === 'idleCores') {
|
|
2041
|
+
// Idle core-time priced as memory held but unused: the same MB-seconds unit as wasteModel, so comparable.
|
|
2042
|
+
const idleRateFraction = finding.idleRateFraction ;
|
|
2043
|
+
const allocatedMB = finding.allocatedMB ;
|
|
2044
|
+
const peakExecutors = finding.peakExecutors ;
|
|
2045
|
+
const appDurationMs = finding.appDurationMs ;
|
|
2046
|
+
if (idleRateFraction != null && allocatedMB != null && peakExecutors != null && appDurationMs != null) {
|
|
2047
|
+
const wastedMBSeconds = idleRateFraction * allocatedMB * peakExecutors * (appDurationMs / 1000);
|
|
2048
|
+
return costOnly('modeled', { value: wastedMBSeconds, unit: 'mbSeconds' });
|
|
2049
|
+
}
|
|
2050
|
+
return costOnly('modeled');
|
|
2051
|
+
}
|
|
2052
|
+
// Only the over-provisioned band is a waste; the near-capacity band is an OOM-risk signal with
|
|
2053
|
+
// no magnitude, and the dataUnavailable shape has no inputs: both stay informational.
|
|
2054
|
+
if (finding.variant === 'memoryBand' && finding.rule === 'heapOverProvisioned') {
|
|
2055
|
+
const allocatedBytes = finding.allocatedBytes ;
|
|
2056
|
+
const heap = finding.heap ;
|
|
2057
|
+
const appDurationMs = finding.appDurationMs ;
|
|
2058
|
+
if (allocatedBytes != null && heap != null && appDurationMs != null) {
|
|
2059
|
+
const unusedMB = (allocatedBytes - heap) / (1024 * 1024);
|
|
2060
|
+
const wastedMBSeconds = unusedMB * (appDurationMs / 1000);
|
|
2061
|
+
return costOnly('modeled', { value: wastedMBSeconds, unit: 'mbSeconds' });
|
|
2062
|
+
}
|
|
2063
|
+
}
|
|
2064
|
+
return costOnly('modeled');
|
|
2065
|
+
},
|
|
2066
|
+
}),
|
|
2067
|
+
defineAppDetector({
|
|
1620
2068
|
// Per-RDD cache-utilization proxies (this repo's own design: Spark event logs carry no
|
|
1621
2069
|
// block-access events, so a literal cache hit rate isn't derivable). Two per-RDD tiered
|
|
1622
|
-
// checks over rddInfo
|
|
1623
|
-
|
|
2070
|
+
// checks over rddInfo: partial caching and disk spillover. An RDD can produce both. rddInfo's
|
|
2071
|
+
// sizes come from SparkListenerBlockUpdated when the log has it, else from StageSubmitted's
|
|
2072
|
+
// RDD Info (0 since Spark 2.3; Spark 1.x fills it only on StageCompleted); with neither, on
|
|
2073
|
+
// Spark 2.3+ with logBlockUpdates off, a storageUnobserved caveat replaces them.
|
|
2074
|
+
type: 'cacheUtilization', order: 103, fixEffort: 'code', version: 2,
|
|
2075
|
+
emits: ['cacheUtilization'],
|
|
1624
2076
|
docAnchor: '#bottleneck-cache-utilization',
|
|
1625
2077
|
thresholds: {
|
|
1626
2078
|
cachedRatioWarn: 0.50, cachedRatioInfo: 0.90,
|
|
1627
2079
|
diskRatioWarn: 0.40, diskRatioInfo: 0.15,
|
|
1628
2080
|
},
|
|
1629
|
-
detect(
|
|
1630
|
-
|
|
1631
|
-
|
|
1632
|
-
|
|
1633
|
-
|
|
1634
|
-
|
|
1635
|
-
ctx ,
|
|
1636
|
-
) {
|
|
2081
|
+
detect(ctx, thresholds) {
|
|
1637
2082
|
const rddInfo = ctx.app?.rddInfo;
|
|
1638
2083
|
if (!(rddInfo instanceof Map)) return null;
|
|
1639
2084
|
const out = [];
|
|
2085
|
+
let persistedRddCount = 0;
|
|
2086
|
+
// With block-update logging on, zero rdd_* updates means nothing was ever cached, not a gap.
|
|
2087
|
+
const blockUpdatesLogged = String(ctx.app?.config?.['spark.eventLog.logBlockUpdates.enabled']).toLowerCase() === 'true';
|
|
2088
|
+
// Spark before 2.3 has no block-update logging and writes RDD Info's cache figures only on
|
|
2089
|
+
// StageCompleted, which isn't read: the caveat's advice doesn't apply there. A log with no
|
|
2090
|
+
// version is pre-1.3 (no SparkListenerLogStart); every 2.3+ log records one.
|
|
2091
|
+
const version = /^(\d+)\.(\d+)/.exec(ctx.app?.sparkVersion ?? '');
|
|
2092
|
+
const preBlockUpdates = version == null || Number(version[1]) < 2 || (Number(version[1]) === 2 && Number(version[2]) < 3);
|
|
2093
|
+
let anyStorageEvidence = blockUpdatesLogged || preBlockUpdates || (ctx.app?.rddBlockUpdates ?? 0) > 0;
|
|
1640
2094
|
for (const rdd of rddInfo.values()) {
|
|
1641
2095
|
const sl = rdd.storageLevel ?? {};
|
|
1642
2096
|
if (!(sl.useMemory || sl.useDisk)) continue;
|
|
2097
|
+
persistedRddCount++;
|
|
1643
2098
|
if (!((rdd.numCachedPartitions ?? 0) > 0)) continue;
|
|
2099
|
+
anyStorageEvidence = true;
|
|
1644
2100
|
|
|
1645
2101
|
if ((rdd.numPartitions ?? 0) > 0) {
|
|
1646
2102
|
const cachedRatio = rdd.numCachedPartitions / rdd.numPartitions;
|
|
1647
|
-
if (cachedRatio <
|
|
1648
|
-
else if (cachedRatio <
|
|
2103
|
+
if (cachedRatio < thresholds.cachedRatioWarn) out.push(partialCacheFinding(rdd, cachedRatio, 'warning'));
|
|
2104
|
+
else if (cachedRatio < thresholds.cachedRatioInfo) out.push(partialCacheFinding(rdd, cachedRatio, 'info'));
|
|
1649
2105
|
}
|
|
1650
2106
|
|
|
1651
2107
|
if (sl.useMemory && sl.useDisk) {
|
|
1652
2108
|
const total = (rdd.memorySize ?? 0) + (rdd.diskSize ?? 0);
|
|
1653
2109
|
if (total > 0) {
|
|
1654
2110
|
const diskRatio = (rdd.diskSize ?? 0) / total;
|
|
1655
|
-
if (diskRatio >
|
|
1656
|
-
else if (diskRatio >
|
|
2111
|
+
if (diskRatio > thresholds.diskRatioWarn) out.push(diskSpilloverFinding(rdd, diskRatio, 'warning'));
|
|
2112
|
+
else if (diskRatio > thresholds.diskRatioInfo) out.push(diskSpilloverFinding(rdd, diskRatio, 'info'));
|
|
1657
2113
|
}
|
|
1658
2114
|
}
|
|
1659
2115
|
}
|
|
2116
|
+
if (persistedRddCount > 0 && !anyStorageEvidence) out.push(storageUnobservedFinding(persistedRddCount, ctx.app));
|
|
1660
2117
|
return out;
|
|
1661
2118
|
},
|
|
1662
|
-
|
|
1663
|
-
|
|
2119
|
+
estimate(finding) {
|
|
2120
|
+
// storageUnobserved reports missing evidence: no sizes, so nothing to model.
|
|
2121
|
+
if (finding.dataUnavailable) return costOnly('none');
|
|
2122
|
+
const memorySize = (finding.memorySize ) ?? 0;
|
|
2123
|
+
const diskSize = (finding.diskSize ) ?? 0;
|
|
2124
|
+
const numCachedPartitions = (finding.numCachedPartitions ) ?? 0;
|
|
2125
|
+
const numPartitions = (finding.numPartitions ) ?? 0;
|
|
2126
|
+
const numUncachedPartitions = Math.max(0, numPartitions - numCachedPartitions);
|
|
2127
|
+
const cachedBytes = memorySize + diskSize;
|
|
2128
|
+
// Extrapolate never-cached partitions' size from the CACHED partitions' average (uncached/
|
|
2129
|
+
// cached, not uncached/total: numCachedPartitions produced cachedBytes). diskSize is added
|
|
2130
|
+
// once more: those bytes are cached but on disk, so re-reading them still costs I/O like an uncached partition.
|
|
2131
|
+
const uncachedBytes = numCachedPartitions > 0 ? (cachedBytes / numCachedPartitions) * numUncachedPartitions : 0;
|
|
2132
|
+
const uncachedOrSpilledBytes = uncachedBytes + diskSize;
|
|
2133
|
+
const wasteMs = (uncachedOrSpilledBytes / RE_READ_THROUGHPUT_BPS) * 1000;
|
|
2134
|
+
return costOnly('modeled', { value: wasteMs, unit: 'ms' });
|
|
2135
|
+
},
|
|
2136
|
+
}),
|
|
2137
|
+
defineAppDetector({
|
|
1664
2138
|
// Non-local task ratio across stage.localityStats (RACK_LOCAL + ANY vs all tasks), the other
|
|
1665
2139
|
// half of the "Wasted Cores Ratio" (idle-core half is memoryUtilization's idleCores).
|
|
1666
2140
|
// NO_PREF stays in the denominator only: shuffle-read stages legitimately report it.
|
|
1667
|
-
type: 'coreLocality',
|
|
2141
|
+
type: 'coreLocality', order: 103, fixEffort: 'config', version: 1,
|
|
2142
|
+
emits: ['coreLocality'],
|
|
1668
2143
|
docAnchor: '#bottleneck-core-locality',
|
|
1669
2144
|
thresholds: { minTasks: 50, warnRatio: 0.15, critRatio: 0.35 },
|
|
1670
|
-
detect(
|
|
1671
|
-
|
|
1672
|
-
ctx ,
|
|
1673
|
-
) {
|
|
2145
|
+
detect(ctx, thresholds) {
|
|
1674
2146
|
const { totalTasks, nonLocalTasks, ratio } = computeCoreLocalityRatio([...ctx.stages.values()]);
|
|
1675
|
-
if (totalTasks == null || totalTasks <
|
|
2147
|
+
if (totalTasks == null || totalTasks < thresholds.minTasks) return null;
|
|
1676
2148
|
// computeCoreLocalityRatio only returns ratio:null together with totalTasks:null (shared
|
|
1677
2149
|
// EMPTY sentinel); the totalTasks guard above rules that out, so ratio is non-null here.
|
|
1678
|
-
if (ratio <
|
|
2150
|
+
if (ratio < thresholds.warnRatio) return null;
|
|
1679
2151
|
|
|
1680
2152
|
const value = Math.round(ratio * 100);
|
|
1681
2153
|
return {
|
|
1682
2154
|
type: 'coreLocality', stageId: null,
|
|
1683
|
-
impactBand: ratio >=
|
|
2155
|
+
impactBand: ratio >= thresholds.critRatio ? 'critical' : 'warning',
|
|
1684
2156
|
metric: 'nonLocalRatio', value,
|
|
1685
2157
|
// Raw count behind the ratio, for the impact estimator. Non-null whenever totalTasks is.
|
|
1686
2158
|
nonLocalTaskCount: nonLocalTasks ,
|
|
1687
|
-
confidence: coreLocalityConfidence(ratio , totalTasks,
|
|
1688
|
-
validationRequired:
|
|
1689
|
-
recommendation: `${value}% of tasks (${nonLocalTasks }) ran without process- or node-local data placement: check
|
|
2159
|
+
confidence: coreLocalityConfidence(ratio , totalTasks, thresholds),
|
|
2160
|
+
validationRequired: `Flagged when at least ${shareLabel(thresholds.warnRatio)} of tasks run non-local (critical at ${shareLabel(thresholds.critRatio)}), on runs of ${thresholds.minTasks}+ tasks.`,
|
|
2161
|
+
recommendation: `${value}% of tasks (${nonLocalTasks }) ran without process- or node-local data placement: check executor/data colocation.`,
|
|
1690
2162
|
};
|
|
1691
2163
|
},
|
|
1692
|
-
|
|
1693
|
-
|
|
2164
|
+
estimate(finding) {
|
|
2165
|
+
const nonLocal = (finding.nonLocalTaskCount ) ?? 0;
|
|
2166
|
+
const coreMs = nonLocal * NETWORK_FETCH_PENALTY_MS;
|
|
2167
|
+
return costOnly('modeled', { value: coreMs, unit: 'coreMs' });
|
|
2168
|
+
},
|
|
2169
|
+
}),
|
|
2170
|
+
defineAppDetector({
|
|
1694
2171
|
// Short-lived executors: stood up and torn down before doing useful work (wasteful
|
|
1695
2172
|
// re-provisioning, not normal scale-down). Reuses utilization's add/remove matching, but
|
|
1696
2173
|
// measures lifetime against a threshold instead of aggregate active-time.
|
|
1697
|
-
type: 'autoscalingChurn',
|
|
2174
|
+
type: 'autoscalingChurn', order: 103, fixEffort: 'config', version: 1,
|
|
2175
|
+
emits: ['autoscalingChurn'],
|
|
1698
2176
|
docAnchor: '#bottleneck-autoscaling-churn',
|
|
1699
2177
|
thresholds: { shortLivedMs: 120_000, warningPct: 0.30, criticalPct: 0.60, minExecutors: 5 },
|
|
1700
|
-
detect(
|
|
1701
|
-
|
|
1702
|
-
|
|
1703
|
-
|
|
1704
|
-
ctx ,
|
|
1705
|
-
) {
|
|
2178
|
+
detect(ctx, thresholds) {
|
|
1706
2179
|
const { app, executorsAdded, executorsRemoved } = ctx;
|
|
1707
2180
|
if (!app || executorsAdded.length === 0 || app.endTime == null) return null;
|
|
1708
|
-
if (executorsAdded.length <
|
|
2181
|
+
if (executorsAdded.length < thresholds.minExecutors) return null;
|
|
1709
2182
|
|
|
1710
2183
|
const removedAt = new Map ();
|
|
1711
2184
|
for (const ev of executorsRemoved) removedAt.set(ev.executorId, ev.timestamp);
|
|
@@ -1714,12 +2187,12 @@ export const DETECTORS = [
|
|
|
1714
2187
|
for (const ev of executorsAdded) {
|
|
1715
2188
|
const endedAt = removedAt.has(ev.executorId) ? removedAt.get(ev.executorId) : app.endTime;
|
|
1716
2189
|
const lifetime = endedAt - ev.timestamp;
|
|
1717
|
-
if (lifetime <
|
|
2190
|
+
if (lifetime < thresholds.shortLivedMs) shortLivedCount++;
|
|
1718
2191
|
}
|
|
1719
2192
|
|
|
1720
2193
|
const shortLivedPct = shortLivedCount / executorsAdded.length;
|
|
1721
|
-
const impactBand = shortLivedPct >
|
|
1722
|
-
: shortLivedPct >
|
|
2194
|
+
const impactBand = shortLivedPct > thresholds.criticalPct ? 'critical'
|
|
2195
|
+
: shortLivedPct > thresholds.warningPct ? 'warning' : null;
|
|
1723
2196
|
if (!impactBand) return null;
|
|
1724
2197
|
|
|
1725
2198
|
const pct = Math.round(shortLivedPct * 100);
|
|
@@ -1728,18 +2201,29 @@ export const DETECTORS = [
|
|
|
1728
2201
|
metric: 'shortLivedExecutorPct', value: pct,
|
|
1729
2202
|
// Raw count behind the percentage, for the impact estimator's startup-overhead figure.
|
|
1730
2203
|
shortLivedExecutorCount: shortLivedCount,
|
|
1731
|
-
confidence: autoscalingChurnConfidence(shortLivedPct,
|
|
1732
|
-
recommendation: `${pct}% of executors ran for under 2 minutes before being removed
|
|
2204
|
+
confidence: autoscalingChurnConfidence(shortLivedPct, thresholds.warningPct, thresholds.criticalPct),
|
|
2205
|
+
recommendation: `${pct}% of executors ran for under 2 minutes before being removed: this looks like wasteful re-provisioning rather than normal scale-down; consider raising spark.dynamicAllocation.executorIdleTimeout or widening the minExecutors/maxExecutors bounds to reduce flapping.`,
|
|
2206
|
+
remediation: [
|
|
2207
|
+
increaseConf('spark.dynamicAllocation.executorIdleTimeout'),
|
|
2208
|
+
decreaseConf('spark.dynamicAllocation.minExecutors'),
|
|
2209
|
+
increaseConf('spark.dynamicAllocation.maxExecutors'),
|
|
2210
|
+
],
|
|
1733
2211
|
};
|
|
1734
2212
|
},
|
|
1735
|
-
|
|
1736
|
-
|
|
2213
|
+
estimate(finding) {
|
|
2214
|
+
const shortLived = (finding.shortLivedExecutorCount ) ?? 0;
|
|
2215
|
+
const executorHours = (shortLived * EXECUTOR_STARTUP_OVERHEAD_MS) / 3.6e6;
|
|
2216
|
+
return costOnly('modeled', { value: executorHours, unit: 'coreHours' });
|
|
2217
|
+
},
|
|
2218
|
+
}),
|
|
2219
|
+
defineAppDetector({
|
|
1737
2220
|
// Cross-execution relation reuse: flags an input relation scanned by two or more SQL
|
|
1738
2221
|
// executions in one run, firing on real relation names (parquet:..., jdbc:...).
|
|
1739
|
-
type: 'cachingOpportunity',
|
|
2222
|
+
type: 'cachingOpportunity', order: 105, fixEffort: 'code', version: 1,
|
|
2223
|
+
emits: ['cachingOpportunity'],
|
|
1740
2224
|
docAnchor: '#bottleneck-caching-opportunity',
|
|
1741
2225
|
thresholds: { minExecutions: 2 },
|
|
1742
|
-
detect(
|
|
2226
|
+
detect(ctx, thresholds) {
|
|
1743
2227
|
const sql = ctx.sql;
|
|
1744
2228
|
if (!(sql instanceof Map) || sql.size === 0) return null;
|
|
1745
2229
|
|
|
@@ -1827,7 +2311,7 @@ export const DETECTORS = [
|
|
|
1827
2311
|
// Qualifying = enough distinct executions on its own. Nested-dedupe: a qualifying composite
|
|
1828
2312
|
// with a qualifying ANCESTOR is subsumed, fully (equal sets) or partially (residual).
|
|
1829
2313
|
const isQualifying = (fp ) =>
|
|
1830
|
-
byComposite.has(fp) && byComposite.get(fp) .executionIds.size >=
|
|
2314
|
+
byComposite.has(fp) && byComposite.get(fp) .executionIds.size >= thresholds.minExecutions;
|
|
1831
2315
|
const compositeResolutions = new Map ();
|
|
1832
2316
|
for (const [fingerprint, agg] of byComposite) {
|
|
1833
2317
|
if (!isQualifying(fingerprint)) { compositeResolutions.set(fingerprint, { finalExecutionIds: agg.executionIds, suppressed: true }); continue; }
|
|
@@ -1836,7 +2320,7 @@ export const DETECTORS = [
|
|
|
1836
2320
|
const coveredByAncestors = new Set(qualifyingAncestors.flatMap(outer => [...outer.executionIds]));
|
|
1837
2321
|
const residual = new Set([...agg.executionIds].filter(id => !coveredByAncestors.has(id)));
|
|
1838
2322
|
if (residual.size === 0) compositeResolutions.set(fingerprint, { finalExecutionIds: residual, suppressed: true });
|
|
1839
|
-
else if (residual.size <
|
|
2323
|
+
else if (residual.size < thresholds.minExecutions) compositeResolutions.set(fingerprint, { finalExecutionIds: residual, suppressed: true });
|
|
1840
2324
|
else compositeResolutions.set(fingerprint, { finalExecutionIds: residual, suppressed: false });
|
|
1841
2325
|
}
|
|
1842
2326
|
|
|
@@ -1861,15 +2345,15 @@ export const DETECTORS = [
|
|
|
1861
2345
|
const relationDisplay = rids.map(relationDisplayName).join(` ${connector} `);
|
|
1862
2346
|
const value = finalExecutionIds.length;
|
|
1863
2347
|
const recommendation = totalReadBytes >= 128 * MB
|
|
1864
|
-
? `${verb[0].toUpperCase()}${verb.slice(1)} result read by ${value} queries (~${formatBytes(totalReadBytes)})
|
|
1865
|
-
: `${verb[0].toUpperCase()}${verb.slice(1)} result read by ${value} queries
|
|
2348
|
+
? `${verb[0].toUpperCase()}${verb.slice(1)} result read by ${value} queries (~${formatBytes(totalReadBytes)}): cache/persist the ${verb} DataFrame so it is computed once.`
|
|
2349
|
+
: `${verb[0].toUpperCase()}${verb.slice(1)} result read by ${value} queries: cache the ${verb} DataFrame, or reconsider whether it needs to be recomputed each time.`;
|
|
1866
2350
|
|
|
1867
2351
|
out.push({
|
|
1868
2352
|
type: 'cachingOpportunity', variant: 'composite', stageId: null, impactBand: 'info',
|
|
1869
2353
|
metric: 'executionReuse', value,
|
|
1870
2354
|
format: 'derived', relations, operator: agg.operator, relation: relationDisplay,
|
|
1871
2355
|
executionIds: finalExecutionIds, totalReadBytes,
|
|
1872
|
-
confidence: cachingReuseConfidence(value,
|
|
2356
|
+
confidence: cachingReuseConfidence(value, thresholds.minExecutions),
|
|
1873
2357
|
validationRequired:
|
|
1874
2358
|
'Composite reuse is inferred from a structural plan-shape match (operator + normalized ' +
|
|
1875
2359
|
'join/filter condition + child shapes) across SQL executions; confirm these executions ' +
|
|
@@ -1888,19 +2372,19 @@ export const DETECTORS = [
|
|
|
1888
2372
|
const residualExecutionIds = covered
|
|
1889
2373
|
? [...agg.executionIds].filter(id => !covered.has(id))
|
|
1890
2374
|
: [...agg.executionIds];
|
|
1891
|
-
if (residualExecutionIds.length <
|
|
2375
|
+
if (residualExecutionIds.length < thresholds.minExecutions) continue;
|
|
1892
2376
|
const value = residualExecutionIds.length;
|
|
1893
2377
|
const totalReadBytes = residualExecutionIds.reduce((sum, id) => sum + (agg.executionBytes.get(id) ?? 0), 0);
|
|
1894
2378
|
const recommendation = totalReadBytes >= 128 * MB
|
|
1895
|
-
? `Read by ${value} queries (~${formatBytes(totalReadBytes)})
|
|
1896
|
-
: `Read by ${value} queries
|
|
2379
|
+
? `Read by ${value} queries (~${formatBytes(totalReadBytes)}): cache/persist the shared DataFrame so it is scanned once.`
|
|
2380
|
+
: `Read by ${value} queries: cache the shared DataFrame, or broadcast it if it is a small join lookup.`;
|
|
1897
2381
|
out.push({
|
|
1898
2382
|
type: 'cachingOpportunity', stageId: null, impactBand: 'info',
|
|
1899
2383
|
metric: 'executionReuse', value,
|
|
1900
2384
|
relation: agg.relation, format: agg.format,
|
|
1901
2385
|
executionIds: residualExecutionIds.sort((a, b) => a - b),
|
|
1902
2386
|
totalReadBytes,
|
|
1903
|
-
confidence: cachingReuseConfidence(value,
|
|
2387
|
+
confidence: cachingReuseConfidence(value, thresholds.minExecutions),
|
|
1904
2388
|
validationRequired:
|
|
1905
2389
|
'Relation-reuse is inferred from the pre-AQE plan scan identity across SQL ' +
|
|
1906
2390
|
'executions; confirm the reads are the same data and cacheable within one ' +
|
|
@@ -1910,15 +2394,18 @@ export const DETECTORS = [
|
|
|
1910
2394
|
}
|
|
1911
2395
|
return out;
|
|
1912
2396
|
},
|
|
1913
|
-
|
|
1914
|
-
|
|
1915
|
-
|
|
2397
|
+
estimate(finding) {
|
|
2398
|
+
const totalReadBytes = (finding.totalReadBytes ) ?? 0;
|
|
2399
|
+
const wasteMs = (totalReadBytes / RE_READ_THROUGHPUT_BPS) * 1000;
|
|
2400
|
+
return costOnly('modeled', { value: wasteMs, unit: 'ms' });
|
|
2401
|
+
},
|
|
2402
|
+
}),
|
|
2403
|
+
defineAppDetector({
|
|
2404
|
+
type: 'jobFailureRate', order: 110, fixEffort: 'code', version: 1,
|
|
2405
|
+
emits: ['jobFailureRate'],
|
|
1916
2406
|
docAnchor: '#bottleneck-job-failure-rate',
|
|
1917
2407
|
thresholds: { infoRate: 0.10, warnRate: 0.30, critRate: 0.50 },
|
|
1918
|
-
detect(
|
|
1919
|
-
|
|
1920
|
-
ctx ,
|
|
1921
|
-
) {
|
|
2408
|
+
detect(ctx, thresholds) {
|
|
1922
2409
|
const { jobs, stages } = ctx;
|
|
1923
2410
|
const all = jobs ? [...jobs.values()] : [];
|
|
1924
2411
|
const completed = all.filter(j => j.result != null);
|
|
@@ -1926,7 +2413,7 @@ export const DETECTORS = [
|
|
|
1926
2413
|
const failedJobList = completed.filter(j => j.succeeded === false);
|
|
1927
2414
|
const failedJobs = failedJobList.length;
|
|
1928
2415
|
const rate = failedJobs / completed.length;
|
|
1929
|
-
if (rate <
|
|
2416
|
+
if (rate < thresholds.infoRate) return null;
|
|
1930
2417
|
let totalTasks = 0, failedTasks = 0;
|
|
1931
2418
|
for (const s of stages.values()) { totalTasks += s.taskCount ?? 0; failedTasks += s.failedTasks ?? 0; }
|
|
1932
2419
|
const taskFailureRate = totalTasks > 0 ? failedTasks / totalTasks : 0;
|
|
@@ -1939,113 +2426,126 @@ export const DETECTORS = [
|
|
|
1939
2426
|
const totalJobs = completed.length;
|
|
1940
2427
|
return {
|
|
1941
2428
|
type: 'jobFailureRate', stageId: null,
|
|
1942
|
-
impactBand: rate >=
|
|
2429
|
+
impactBand: rate >= thresholds.critRate ? 'critical' : rate >= thresholds.warnRate ? 'warning' : 'info',
|
|
1943
2430
|
metric: 'jobFailureRate', value: Math.round(rate * 1000) / 10,
|
|
1944
2431
|
failedJobs, totalJobs, failedTasks, totalTasks, avgJobDurationMs,
|
|
1945
2432
|
taskFailureRate: Math.round(taskFailureRate * 1000) / 10,
|
|
1946
2433
|
recommendation: `${failedJobs} of ${totalJobs} jobs never recovered: inspect the driver log for the failed job(s) and the stage failures that triggered them.`,
|
|
1947
2434
|
};
|
|
1948
2435
|
},
|
|
1949
|
-
|
|
2436
|
+
estimate(finding) {
|
|
2437
|
+
const failedJobs = (finding.failedJobs ) ?? 0;
|
|
2438
|
+
const avgJobDurationMs = (finding.avgJobDurationMs ) ?? 0;
|
|
2439
|
+
const coreHoursIsh = (failedJobs * avgJobDurationMs) / 3.6e6;
|
|
2440
|
+
return costOnly('modeled', { value: coreHoursIsh, unit: 'coreHours' });
|
|
2441
|
+
},
|
|
2442
|
+
}),
|
|
1950
2443
|
// ── Config-sanity entries (scope:'config', inScorecard:false) ────────────────
|
|
1951
|
-
{
|
|
1952
|
-
type: 'configAudit',
|
|
2444
|
+
defineConfigDetector({
|
|
2445
|
+
type: 'configAudit', order: 120, fixEffort: 'config', version: 1, inScorecard: false,
|
|
2446
|
+
emits: ['configAudit'],
|
|
1953
2447
|
docAnchor: '#config-shuffle-service', thresholds: {}, property: 'spark.shuffle.service.enabled',
|
|
1954
|
-
detect(
|
|
1955
|
-
const res =
|
|
2448
|
+
detect(target) {
|
|
2449
|
+
const res = target.app?.resources ?? null;
|
|
1956
2450
|
if (res?.dynamicAllocationEnabled === true && res?.shuffleServiceEnabled === false) {
|
|
1957
2451
|
return {
|
|
1958
2452
|
type: 'configAudit', property: 'spark.shuffle.service.enabled',
|
|
1959
|
-
impactBand: 'warning', metric: 'config',
|
|
2453
|
+
impactBand: 'warning', metric: 'config', valueText: 'false',
|
|
1960
2454
|
recommendation: 'Dynamic allocation is on but the external shuffle service is off: set spark.shuffle.service.enabled=true so shuffle data survives executor removal.',
|
|
2455
|
+
remediation: setConfUnlessLogged(target.app, 'spark.shuffle.service.enabled', true),
|
|
1961
2456
|
};
|
|
1962
2457
|
}
|
|
1963
2458
|
return null;
|
|
1964
2459
|
},
|
|
1965
|
-
|
|
1966
|
-
|
|
1967
|
-
|
|
2460
|
+
estimate: noWasteModel,
|
|
2461
|
+
}),
|
|
2462
|
+
defineConfigDetector({
|
|
2463
|
+
type: 'configAudit', order: 121, fixEffort: 'config', version: 1, inScorecard: false,
|
|
2464
|
+
emits: ['configAudit'],
|
|
1968
2465
|
docAnchor: '#config-autoscale-bounds', thresholds: {}, property: 'spark.dynamicAllocation.maxExecutors',
|
|
1969
|
-
detect(
|
|
1970
|
-
const app =
|
|
2466
|
+
detect(target) {
|
|
2467
|
+
const app = target.app; const config = app?.config ?? {}; const res = app?.resources ?? null;
|
|
1971
2468
|
if (res?.dynamicAllocationEnabled !== true) return null;
|
|
1972
2469
|
const minN = config['spark.dynamicAllocation.minExecutors'] != null ? parseInt(config['spark.dynamicAllocation.minExecutors'], 10) : null;
|
|
1973
2470
|
const maxN = config['spark.dynamicAllocation.maxExecutors'] != null ? parseInt(config['spark.dynamicAllocation.maxExecutors'], 10) : null;
|
|
1974
2471
|
if (minN != null && maxN != null && minN > maxN) {
|
|
1975
2472
|
return {
|
|
1976
2473
|
type: 'configAudit', property: 'spark.dynamicAllocation.minExecutors',
|
|
1977
|
-
impactBand: 'critical', metric: 'config',
|
|
1978
|
-
recommendation: `
|
|
2474
|
+
impactBand: 'critical', metric: 'config', valueText: `${minN} > ${maxN}`,
|
|
2475
|
+
recommendation: `spark.dynamicAllocation.minExecutors (${minN}) exceeds maxExecutors (${maxN}): set min ≤ max.`,
|
|
2476
|
+
remediation: [decreaseConf('spark.dynamicAllocation.minExecutors', maxN)],
|
|
1979
2477
|
};
|
|
1980
2478
|
}
|
|
1981
2479
|
if (maxN == null) {
|
|
1982
2480
|
return {
|
|
1983
2481
|
type: 'configAudit', property: 'spark.dynamicAllocation.maxExecutors',
|
|
1984
|
-
impactBand: 'info', metric: 'config',
|
|
2482
|
+
impactBand: 'info', metric: 'config', valueText: '(unset)',
|
|
1985
2483
|
recommendation: 'Dynamic allocation is on with no upper bound: set spark.dynamicAllocation.maxExecutors to cap cluster growth.',
|
|
2484
|
+
remediation: [setConf('spark.dynamicAllocation.maxExecutors')],
|
|
1986
2485
|
};
|
|
1987
2486
|
}
|
|
1988
2487
|
return null;
|
|
1989
2488
|
},
|
|
1990
|
-
|
|
1991
|
-
|
|
1992
|
-
|
|
2489
|
+
estimate: noWasteModel,
|
|
2490
|
+
}),
|
|
2491
|
+
defineConfigDetector({
|
|
2492
|
+
type: 'configAudit', order: 122, fixEffort: 'config', version: 1, inScorecard: false,
|
|
2493
|
+
emits: ['configAudit'],
|
|
1993
2494
|
docAnchor: '#config-serializer', thresholds: {}, property: 'spark.serializer',
|
|
1994
|
-
detect(
|
|
1995
|
-
const app =
|
|
2495
|
+
detect(target) {
|
|
2496
|
+
const app = target.app; const config = app?.config ?? {}; const res = app?.resources ?? null;
|
|
1996
2497
|
if (Object.keys(config).length === 0) return null;
|
|
1997
2498
|
const ser = res?.serializer ?? config['spark.serializer'] ?? null;
|
|
1998
2499
|
const isKryo = typeof ser === 'string' && /kryo/i.test(ser);
|
|
1999
2500
|
if (isKryo) return null;
|
|
2000
2501
|
return {
|
|
2001
2502
|
type: 'configAudit', property: 'spark.serializer',
|
|
2002
|
-
impactBand: 'info', metric: 'config',
|
|
2503
|
+
impactBand: 'info', metric: 'config', valueText: ser ?? '(default JavaSerializer)',
|
|
2003
2504
|
recommendation: `Current serializer is ${ser ?? 'the default JavaSerializer'}: consider spark.serializer=org.apache.spark.serializer.KryoSerializer for faster, smaller buffers.`,
|
|
2505
|
+
remediation: setConfUnlessLogged(app, 'spark.serializer', 'org.apache.spark.serializer.KryoSerializer'),
|
|
2004
2506
|
};
|
|
2005
2507
|
},
|
|
2006
|
-
|
|
2007
|
-
|
|
2008
|
-
|
|
2508
|
+
estimate: noWasteModel,
|
|
2509
|
+
}),
|
|
2510
|
+
defineConfigDetector({
|
|
2511
|
+
type: 'configAudit', order: 123, fixEffort: 'config', version: 1, inScorecard: false,
|
|
2512
|
+
emits: ['configAudit'],
|
|
2009
2513
|
docAnchor: '#config-memory-overhead', thresholds: { floorMB: 384, floorPct: 0.1 }, property: 'spark.executor.memoryOverhead',
|
|
2010
|
-
detect(
|
|
2011
|
-
|
|
2012
|
-
ctx ,
|
|
2013
|
-
) {
|
|
2014
|
-
const res = ctx.app?.resources ?? null;
|
|
2514
|
+
detect(target, thresholds) {
|
|
2515
|
+
const res = target.app?.resources ?? null;
|
|
2015
2516
|
const memMB = res?.executor?.memoryMB ?? null;
|
|
2016
2517
|
const ovMB = res?.executor?.memoryOverheadMB ?? null;
|
|
2017
2518
|
if (memMB == null || ovMB == null) return null;
|
|
2018
|
-
const floor = Math.max(
|
|
2519
|
+
const floor = Math.max(thresholds.floorMB, Math.round(memMB * thresholds.floorPct));
|
|
2019
2520
|
if (ovMB >= floor) return null;
|
|
2020
2521
|
return {
|
|
2021
2522
|
type: 'configAudit', property: 'spark.executor.memoryOverhead',
|
|
2022
|
-
impactBand: 'info', metric: 'config',
|
|
2523
|
+
impactBand: 'info', metric: 'config', valueText: `${ovMB} MiB`,
|
|
2023
2524
|
recommendation: `Executor memoryOverhead (${ovMB} MiB) is below Spark's default floor of ${floor} MiB (max of 384 MiB or 10% of executor memory): raise it to avoid off-heap OOM-kills.`,
|
|
2525
|
+
remediation: [increaseConf('spark.executor.memoryOverhead', `${floor}m`)],
|
|
2024
2526
|
};
|
|
2025
2527
|
},
|
|
2026
|
-
|
|
2528
|
+
estimate: noWasteModel,
|
|
2529
|
+
}),
|
|
2027
2530
|
// ── Plan-metric entries (scope:'sql') ────────────────────────────────────
|
|
2028
|
-
{
|
|
2029
|
-
type: 'duplicatePlanSubtree',
|
|
2531
|
+
defineSqlDetector({
|
|
2532
|
+
type: 'duplicatePlanSubtree', order: 130, fixEffort: 'code', version: 2,
|
|
2533
|
+
emits: ['duplicatePlanSubtree'],
|
|
2030
2534
|
docAnchor: '#bottleneck-duplicate-plan-subtree',
|
|
2031
2535
|
// stageFloorPct: the 0.5% runtime floor. The claim counts at most each linked stage's own
|
|
2032
2536
|
// task-active time, so a repeat whose stages together lasted less than this share of the run
|
|
2033
2537
|
// graded info: 340 of 545 findings on the 14 real logs. The repeat is still in the plan; the
|
|
2034
2538
|
// floor is why they're dropped. A repeat with no linked stage time is kept.
|
|
2035
2539
|
thresholds: { minSubtreeSize: 3, minOccurrences: 2, stageFloorPct: 0.005 },
|
|
2036
|
-
detect(
|
|
2037
|
-
|
|
2038
|
-
sqlExec ,
|
|
2039
|
-
ctx ,
|
|
2040
|
-
) {
|
|
2540
|
+
detect(sqlExec, ctx, thresholds) {
|
|
2041
2541
|
if (!sqlExec.planTree) return null;
|
|
2042
|
-
const groups = findDuplicateSubtrees(sqlExec.planTree,
|
|
2542
|
+
const groups = findDuplicateSubtrees(sqlExec.planTree, thresholds);
|
|
2043
2543
|
if (groups.length === 0) return null;
|
|
2044
2544
|
const fallbackStageIds = stageIdsForSqlExec(sqlExec.id, ctx.stages);
|
|
2045
2545
|
const executionNodes = [];
|
|
2046
2546
|
walkPlanTree(sqlExec.planTree, (node) => executionNodes.push(node));
|
|
2047
2547
|
const operatorsByStage = operatorCountByStage(executionNodes);
|
|
2048
|
-
const
|
|
2548
|
+
const runMs = appDurationMs(ctx.app);
|
|
2049
2549
|
const findings = groups.map((g) => {
|
|
2050
2550
|
const nodes = [];
|
|
2051
2551
|
for (const n of g.nodes) walkPlanTree(n, (node) => nodes.push(node));
|
|
@@ -2055,13 +2555,13 @@ export const DETECTORS = [
|
|
|
2055
2555
|
const stage = ctx.stages.get(id);
|
|
2056
2556
|
if (stage) stagesMs += Math.max(0, (stage.completedAt ?? 0) - (stage.submittedAt ?? 0));
|
|
2057
2557
|
}
|
|
2058
|
-
if (
|
|
2558
|
+
if (runMs != null && stagesMs > 0 && stagesMs < runMs * thresholds.stageFloorPct) return null;
|
|
2059
2559
|
const occurrencesIdentical = occurrencesHaveIdenticalDetails(g.nodes);
|
|
2060
2560
|
const stageShares = stageOperatorShares(nodes, operatorsByStage);
|
|
2061
2561
|
// resolvePlanTree always sets id; safe downstream of it.
|
|
2062
2562
|
const planNodeIds = nodes.map((n) => n.id ).filter(Boolean);
|
|
2063
|
-
const
|
|
2064
|
-
const differing = occurrencesIdentical ? '' :
|
|
2563
|
+
const detail = duplicateSubtreeDetail({ ...g, value: g.occurrences });
|
|
2564
|
+
const differing = occurrencesIdentical ? '' : ` ${DUPLICATE_SUBTREE_DIFFERING_NOTE}`;
|
|
2065
2565
|
return {
|
|
2066
2566
|
type: 'duplicatePlanSubtree', executionId: sqlExec.id, stageIds, planNodeIds,
|
|
2067
2567
|
stageShares, occurrencesIdentical,
|
|
@@ -2072,27 +2572,59 @@ export const DETECTORS = [
|
|
|
2072
2572
|
metric: 'subtreeOccurrences', value: g.occurrences,
|
|
2073
2573
|
rootName: g.rootName, subtreeSize: g.subtreeSize, sampleRelation: g.sampleRelation,
|
|
2074
2574
|
groupIndex: g.groupIndex,
|
|
2075
|
-
confidence: occurrencesIdentical ? duplicateSubtreeConfidence(g.subtreeSize, g.occurrences,
|
|
2076
|
-
validationRequired: '
|
|
2575
|
+
confidence: occurrencesIdentical ? duplicateSubtreeConfidence(g.subtreeSize, g.occurrences, thresholds) : 'low',
|
|
2576
|
+
validationRequired: 'Matching compares operator and metric names only, not literals or expression IDs: confirm the repeat in the Spark UI SQL tab before acting.',
|
|
2077
2577
|
recommendation: (g.isExchangeRoot
|
|
2078
|
-
?
|
|
2079
|
-
:
|
|
2578
|
+
? `${detail}: this looks like a possible missed exchange reuse; check whether the same shuffle could be computed once and reused.`
|
|
2579
|
+
: `${detail}: consider caching/persisting the shared computation or check for a duplicated query branch.`) + differing,
|
|
2080
2580
|
};
|
|
2081
2581
|
}).filter((f) => f !== null);
|
|
2082
2582
|
return findings.length > 0 ? findings : null;
|
|
2083
2583
|
},
|
|
2084
|
-
|
|
2085
|
-
|
|
2086
|
-
|
|
2584
|
+
estimate(finding, ctx) {
|
|
2585
|
+
const stageIds = finding.stageIds ;
|
|
2586
|
+
if (!stageIds || stageIds.length === 0) return null;
|
|
2587
|
+
// detect() reports subtreeOccurrences >= 2. Only repeats past the first are redundant:
|
|
2588
|
+
// computing the subtree once is real work, so waste is (occurrences-1)/occurrences of the stages' time.
|
|
2589
|
+
const occurrences = typeof finding.value === 'number' ? finding.value : 0;
|
|
2590
|
+
// Defensive only: minOccurrences guarantees occurrences >= 2 on real data; a malformed-value fallback.
|
|
2591
|
+
if (occurrences < 2) return costOnly('none');
|
|
2592
|
+
// Same-shaped repeats whose details differ compute different data: nothing is known to be
|
|
2593
|
+
// recomputed, so there is no time to claim.
|
|
2594
|
+
if (finding.occurrencesIdentical === false) return costOnly('none');
|
|
2595
|
+
// Each stage contributes the share of its operators inside the repeated subtree: a stage it
|
|
2596
|
+
// shares with other operators (the consuming join, the join's other side) isn't all its
|
|
2597
|
+
// time, and claiming whole stages let sibling groups claim the same stage twice. Findings
|
|
2598
|
+
// built without the field (hand-made fixtures) count every linked stage whole.
|
|
2599
|
+
const shares = (finding.stageShares ?? null) ;
|
|
2600
|
+
const redundantFraction = (occurrences - 1) / occurrences;
|
|
2601
|
+
const wasteMsByStage = new Map ();
|
|
2602
|
+
for (const id of stageIds) {
|
|
2603
|
+
const s = ctx.stages.get(id);
|
|
2604
|
+
const share = shares ? (shares[id] ?? 0) : 1;
|
|
2605
|
+
if (s && share > 0) {
|
|
2606
|
+
// Time with tasks running, not submit-to-complete: a stage left waiting for cores
|
|
2607
|
+
// (2491s open, 60s of tasks on a real log) isn't recomputing anything while it waits.
|
|
2608
|
+
const activeMs = s.taskActiveMs ?? Math.max(0, (s.completedAt ?? 0) - (s.submittedAt ?? 0));
|
|
2609
|
+
wasteMsByStage.set(id, activeMs * share * redundantFraction);
|
|
2610
|
+
}
|
|
2611
|
+
}
|
|
2612
|
+
// No operator of the subtree ran in a known stage: no time to attribute.
|
|
2613
|
+
if (wasteMsByStage.size === 0) return costOnly('none');
|
|
2614
|
+
const totalWasteMs = [...wasteMsByStage.values()].reduce((sum, ms) => sum + ms, 0);
|
|
2615
|
+
const rawWaste = { value: totalWasteMs, unit: 'ms' };
|
|
2616
|
+
return multiStageImpact([...wasteMsByStage.keys()], wasteMsByStage, ctx, 'measured', rawWaste)
|
|
2617
|
+
?? costOnly('measured', rawWaste);
|
|
2618
|
+
},
|
|
2619
|
+
}),
|
|
2620
|
+
defineSqlDetector({
|
|
2621
|
+
type: 'smallFiles', order: 131, fixEffort: 'config', version: 2,
|
|
2622
|
+
emits: ['smallFiles'],
|
|
2087
2623
|
docAnchor: '#bottleneck-small-files',
|
|
2088
2624
|
thresholds: { minFiles: 100, maxAvgFileSizeMB: 3 },
|
|
2089
|
-
detect(
|
|
2090
|
-
|
|
2091
|
-
sqlExec ,
|
|
2092
|
-
ctx ,
|
|
2093
|
-
) {
|
|
2625
|
+
detect(sqlExec, ctx, thresholds) {
|
|
2094
2626
|
if (!sqlExec.planTree) return null;
|
|
2095
|
-
const { minFiles, maxAvgFileSizeMB } =
|
|
2627
|
+
const { minFiles, maxAvgFileSizeMB } = thresholds;
|
|
2096
2628
|
|
|
2097
2629
|
const hits = [];
|
|
2098
2630
|
walkPlanTree(sqlExec.planTree, (node) => {
|
|
@@ -2127,27 +2659,36 @@ export const DETECTORS = [
|
|
|
2127
2659
|
};
|
|
2128
2660
|
});
|
|
2129
2661
|
},
|
|
2130
|
-
|
|
2131
|
-
|
|
2662
|
+
estimate(finding, ctx) {
|
|
2663
|
+
const fileMs = ((finding.fileCount ) ?? 0) * FILE_OPEN_OVERHEAD_MS;
|
|
2664
|
+
// A read's files are opened by the scan's tasks, in parallel: spread the per-file cost over
|
|
2665
|
+
// the most tasks its stages ran at once (91344 files x 10ms is 913s, claimed against a 117s
|
|
2666
|
+
// stage that ran 314 tasks at once). A write keeps the serial sum: the job commit moves each
|
|
2667
|
+
// output file on the driver, one after another.
|
|
2668
|
+
const stageIds = finding.stageIds ;
|
|
2669
|
+
let slots = 1;
|
|
2670
|
+
if (finding.direction === 'read') {
|
|
2671
|
+
for (const id of stageIds ?? []) slots = Math.max(slots, ctx.stages.get(id)?.peakConcurrentTasks ?? 1);
|
|
2672
|
+
}
|
|
2673
|
+
return stageMappableWasteOrCostOnly(fileMs / slots, stageIds, ctx);
|
|
2674
|
+
},
|
|
2675
|
+
}),
|
|
2676
|
+
defineSqlDetector({
|
|
2132
2677
|
// Entry-level type is an identifier only; it never appears on an emitted finding. Findings
|
|
2133
2678
|
// carry 'underBroadcast'/'overBroadcast' since one shared plan-walk covers both
|
|
2134
2679
|
// opposite-direction rules (JoinToBroadcastAlert / BroadcastTooLargeAlert).
|
|
2135
|
-
type: 'broadcastSizing',
|
|
2680
|
+
type: 'broadcastSizing', order: 132, fixEffort: 'config', version: 2,
|
|
2681
|
+
// Listed over-first: the two share order 132, and this list order is their display tie-break.
|
|
2682
|
+
emits: ['overBroadcast', 'underBroadcast'],
|
|
2136
2683
|
docAnchor: '#bottleneck-broadcast-sizing',
|
|
2137
2684
|
thresholds: {
|
|
2138
2685
|
broadcastTiers: [10 * MB, 100 * MB, GB, 5 * GB],
|
|
2139
2686
|
comparisonTiers: [10 * GB, 300 * GB, TB],
|
|
2140
2687
|
overBroadcastBytes: GB,
|
|
2141
2688
|
},
|
|
2142
|
-
detect(
|
|
2143
|
-
|
|
2144
|
-
|
|
2145
|
-
|
|
2146
|
-
sqlExec ,
|
|
2147
|
-
ctx ,
|
|
2148
|
-
) {
|
|
2689
|
+
detect(sqlExec, ctx, thresholds) {
|
|
2149
2690
|
if (!sqlExec.planTree) return null;
|
|
2150
|
-
const { broadcastTiers, comparisonTiers, overBroadcastBytes } =
|
|
2691
|
+
const { broadcastTiers, comparisonTiers, overBroadcastBytes } = thresholds;
|
|
2151
2692
|
const fallbackStageIds = stageIdsForSqlExec(sqlExec.id, ctx.stages);
|
|
2152
2693
|
const out = [];
|
|
2153
2694
|
walkPlanTree(sqlExec.planTree, (node) => {
|
|
@@ -2171,6 +2712,7 @@ export const DETECTORS = [
|
|
|
2171
2712
|
impactBand: 'info', metric: 'smallerSideBytes', value: smaller,
|
|
2172
2713
|
largerSideBytes: larger,
|
|
2173
2714
|
recommendation: `The smaller input to this Sort Merge Join (${formatBytes(smaller)}) is well under the broadcast threshold relative to the larger side (${formatBytes(larger)}): this could have been a broadcast join. Consider a broadcast() hint or raising spark.sql.autoBroadcastJoinThreshold.`,
|
|
2715
|
+
remediation: [increaseConf('spark.sql.autoBroadcastJoinThreshold')],
|
|
2174
2716
|
});
|
|
2175
2717
|
}
|
|
2176
2718
|
}
|
|
@@ -2182,18 +2724,70 @@ export const DETECTORS = [
|
|
|
2182
2724
|
// (node.stageIds always empty in real data); its child carries the executor-side
|
|
2183
2725
|
// metrics, so only the child unions in.
|
|
2184
2726
|
const child = (node.children ?? [])[0];
|
|
2727
|
+
const autoBroadcastOff = ctx.app?.config?.['spark.sql.autoBroadcastJoinThreshold']?.trim() === '-1';
|
|
2185
2728
|
out.push({
|
|
2186
2729
|
type: 'overBroadcast', executionId: sqlExec.id,
|
|
2187
2730
|
stageIds: unionStageIds(child ? [child] : [], fallbackStageIds),
|
|
2188
2731
|
// resolvePlanTree always sets id; safe downstream of it.
|
|
2189
2732
|
planNodeIds: [node.id ].filter(Boolean),
|
|
2190
2733
|
impactBand: 'warning', metric: 'broadcastBytes', value: m.value,
|
|
2191
|
-
recommendation:
|
|
2734
|
+
recommendation: autoBroadcastOff
|
|
2735
|
+
? `This broadcast (${formatBytes(m.value)}) exceeds the ${binaryThresholdLabel(overBroadcastBytes)} threshold: automatic broadcast is already disabled, so remove the broadcast() hint that forced it.`
|
|
2736
|
+
: `This broadcast (${formatBytes(m.value)}) exceeds the ${binaryThresholdLabel(overBroadcastBytes)} threshold: check for a misapplied broadcast hint or a misconfigured spark.sql.autoBroadcastJoinThreshold.`,
|
|
2737
|
+
remediation: autoBroadcastOff ? [] : [decreaseConf('spark.sql.autoBroadcastJoinThreshold')],
|
|
2192
2738
|
});
|
|
2193
2739
|
}
|
|
2194
2740
|
}
|
|
2195
2741
|
});
|
|
2196
2742
|
return out.length ? out : null;
|
|
2197
2743
|
},
|
|
2198
|
-
|
|
2199
|
-
|
|
2744
|
+
estimate(finding, ctx) {
|
|
2745
|
+
// Both finding types carry bytes as `value`: overBroadcast's broadcastBytes, underBroadcast's
|
|
2746
|
+
// smallerSideBytes (the smaller join side), each priced as one broadcast transfer.
|
|
2747
|
+
const wasteMs = (((finding.value ) ?? 0) / BROADCAST_BANDWIDTH_BPS) * 1000;
|
|
2748
|
+
return stageMappableWasteOrCostOnly(wasteMs, finding.stageIds , ctx);
|
|
2749
|
+
},
|
|
2750
|
+
}),
|
|
2751
|
+
] ;
|
|
2752
|
+
|
|
2753
|
+
/** The entry that emits each finding type: the one whose estimate prices it and whose thresholds
|
|
2754
|
+
* and order describe it. Several entries can emit one type (the four configAudit audits), and the
|
|
2755
|
+
* first declared wins. */
|
|
2756
|
+
export const ENTRY_BY_TYPE = (() => {
|
|
2757
|
+
const byType = new Map ();
|
|
2758
|
+
for (const entry of DETECTORS ) {
|
|
2759
|
+
for (const type of entry.emits) if (!byType.has(type)) byType.set(type, entry);
|
|
2760
|
+
}
|
|
2761
|
+
return byType;
|
|
2762
|
+
})();
|
|
2763
|
+
|
|
2764
|
+
/** A `DETECTORS` entry's own `type`: every emitted finding type, plus broadcastSizing. */
|
|
2765
|
+
|
|
2766
|
+
|
|
2767
|
+
/** Every finding `type` a detector can emit, from the entries' `emits` lists. */
|
|
2768
|
+
|
|
2769
|
+
|
|
2770
|
+
|
|
2771
|
+
|
|
2772
|
+
|
|
2773
|
+
|
|
2774
|
+
/** The `thresholds` of the entry (or entries, for configAudit) that emit finding type `T`. */
|
|
2775
|
+
|
|
2776
|
+
|
|
2777
|
+
// `emits` already only names Finding members (Detector.emits); this makes the reverse hold too, so a
|
|
2778
|
+
// Finding member no detector emits, or a detector whose type has no Finding member, fails to compile.
|
|
2779
|
+
|
|
2780
|
+
|
|
2781
|
+
|
|
2782
|
+
|
|
2783
|
+
// A `suppressedBy` that names no entry would never suppress anything; this makes it a compile error.
|
|
2784
|
+
// An entry without one infers the bare `string` constraint, which contributes nothing here.
|
|
2785
|
+
|
|
2786
|
+
|
|
2787
|
+
|
|
2788
|
+
|
|
2789
|
+
/** Per-detector threshold overrides, keyed by entry `type`: each value a partial of that entry's
|
|
2790
|
+
* own thresholds. Built by threshold-overrides.ts from a user's config file. */
|
|
2791
|
+
|
|
2792
|
+
|
|
2793
|
+
|