sparkforensics-mcp 0.2.0 → 0.2.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/package.json +5 -4
- package/vendor-core/cli/collect-run.js +3 -2
- package/vendor-core/cli/native-zstd.js +351 -0
- package/vendor-core/detectors.js +390 -75
- package/vendor-core/docs-config.js +34 -8
- package/vendor-core/docs-content/chapters/03-memory-model.md +39 -0
- package/vendor-core/docs-content/chapters/11-cluster-config.md +40 -0
- package/vendor-core/docs-content/detection/cache.md +4 -3
- package/vendor-core/docs-content/detection/chrn.md +4 -2
- package/vendor-core/docs-content/detection/gc.md +2 -0
- package/vendor-core/docs-content/detection/host.md +2 -1
- package/vendor-core/docs-content/detection/local.md +2 -3
- package/vendor-core/docs-content/detection/mem.md +3 -3
- package/vendor-core/docs-content/detection/plan.md +3 -1
- package/vendor-core/docs-content/detection/shape.md +2 -1
- package/vendor-core/docs-content/detection/shfl.md +2 -1
- package/vendor-core/docs-content/detection/spec.md +4 -3
- package/vendor-core/docs-content/detection/spill.md +1 -1
- package/vendor-core/docs-content/detection/strag.md +2 -1
- package/vendor-core/docs-content/detection/tiny.md +2 -1
- package/vendor-core/docs-content/tuning/failures.md +1 -1
- package/vendor-core/docs-content/tuning/gc.md +11 -4
- package/vendor-core/docs-content/tuning/shuffle.md +25 -5
- package/vendor-core/docs-content/tuning/skew.md +14 -6
- package/vendor-core/docs-content/tuning/small-files.md +12 -7
- package/vendor-core/docs-content/tuning/straggler.md +34 -0
- package/vendor-core/docs-content/tuning/tiny-tasks.md +1 -1
- package/vendor-core/docs-content/tuning/utilization.md +57 -7
- package/vendor-core/docs-content/upstream.json +4 -0
- package/vendor-core/event-handlers.js +321 -69
- package/vendor-core/event-schemas.js +8 -6
- package/vendor-core/evidence-report.js +4 -2
- package/vendor-core/impact-estimator.js +150 -34
- package/vendor-core/mcp-tools.js +20 -7
- package/vendor-core/occupancy.js +70 -2
- package/vendor-core/parser-worker.js +30 -14
- package/vendor-core/plan-summary.js +4 -0
- package/vendor-core/run-comparison.js +22 -17
- package/vendor-core/shs-fetch.js +18 -7
- package/vendor-core/shs-load.js +2 -1
- package/vendor-core/stage-quantiles.js +59 -3
- package/vendor-core/string-hash.js +15 -0
- package/vendor-core/types.js +9 -1
- package/vendor-core/vendor/fzstd.js +94 -18
package/vendor-core/detectors.js
CHANGED
|
@@ -3,8 +3,9 @@ import { scanRelationId } from './plan-summary.js';
|
|
|
3
3
|
import { computePeakConcurrentCores, computePeakConcurrentExecutorCount } from './core-count.js';
|
|
4
4
|
import { walkPlanTree } from './plan-tree-walk.js';
|
|
5
5
|
import { computeCoreLocalityRatio } from './core-locality-ratio.js';
|
|
6
|
-
import { estimateSingleStage, } from './occupancy.js';
|
|
6
|
+
import { estimateSingleStage, tailRecoveryMs, tailRemovedWorkMs, stragglerFixLongestTaskMs, } from './occupancy.js';
|
|
7
7
|
import { isExchangeNode, isBroadcastExchangeNode } from './plan-node-detail.js';
|
|
8
|
+
import { cyrb53 } from './string-hash.js';
|
|
8
9
|
|
|
9
10
|
|
|
10
11
|
const MB = 1024 * 1024;
|
|
@@ -74,6 +75,9 @@ const TB = 1024 * GB;
|
|
|
74
75
|
|
|
75
76
|
|
|
76
77
|
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
|
|
77
81
|
|
|
78
82
|
|
|
79
83
|
|
|
@@ -154,6 +158,7 @@ const TB = 1024 * GB;
|
|
|
154
158
|
|
|
155
159
|
|
|
156
160
|
|
|
161
|
+
|
|
157
162
|
|
|
158
163
|
|
|
159
164
|
// SpillPressureDetector (5a) + SpillSkewDetector (5b).
|
|
@@ -206,8 +211,10 @@ export function unionStageIds(nodes , fallback ) {
|
|
|
206
211
|
|
|
207
212
|
// Bottom-up shape computation for duplicate-subtree detection and the
|
|
208
213
|
// cachingOpportunity composite detector. `size` is the subtree node count; the default
|
|
209
|
-
// fingerprint encodes operator name + sorted metric NAMES (never values, per spec) +
|
|
210
|
-
//
|
|
214
|
+
// fingerprint encodes operator name + sorted metric NAMES (never values, per spec) + a digest
|
|
215
|
+
// of each child's fingerprint, so two subtrees with the same shape but different values still
|
|
216
|
+
// collide. Digesting children keeps each fingerprint O(own size): embedding the full child
|
|
217
|
+
// strings made every node carry its whole subtree, O(n x depth) text on deep real plans.
|
|
211
218
|
//
|
|
212
219
|
// opts.includeDetail (default false) folds root's own detail (via opts.normalizeDetail) into
|
|
213
220
|
// ONLY root's fingerprint, never a child's: lets a caller match "this node's own detail + shape"
|
|
@@ -239,7 +246,7 @@ export function computePlanShapes(
|
|
|
239
246
|
const childShapes = realChildren.map((c) => visit(c, false));
|
|
240
247
|
const size = 1 + childShapes.reduce((sum, c) => sum + c.size, 0);
|
|
241
248
|
const metricNames = (realMetrics ?? []).map((m) => m.name).sort().join(',');
|
|
242
|
-
const childFingerprints = childShapes.map((c) => c.fingerprint).join(',');
|
|
249
|
+
const childFingerprints = childShapes.map((c) => cyrb53(c.fingerprint)).join(',');
|
|
243
250
|
const fingerprint = isRoot && includeDetail
|
|
244
251
|
? `${node.name}[${metricNames}]<${normalize(node.detail ?? '')}>{${childFingerprints}}`
|
|
245
252
|
: `${node.name}[${metricNames}]{${childFingerprints}}`;
|
|
@@ -261,7 +268,11 @@ export function normalizeDetail(detail ) {
|
|
|
261
268
|
.replace(/,?\s*plan_id=\d+/g, '')
|
|
262
269
|
.replace(/\[codegen id\s*:\s*\d+\]/gi, '[codegen id]')
|
|
263
270
|
.replace(/\bBuild(Left|Right)\b/g, 'BuildSide');
|
|
264
|
-
|
|
271
|
+
// The lookbehind only skips starts inside an identifier, which can't match unless the
|
|
272
|
+
// identifier's own start already did: same output, without re-scanning each identifier from
|
|
273
|
+
// every one of its letters (O(length^2) on the 6.8 KB average join detail of the largest real
|
|
274
|
+
// log; 64ms -> 40ms per analyze()).
|
|
275
|
+
s = s.replace(/(?<![A-Za-z_])([A-Za-z_][\w.]*)\s*=\s*([A-Za-z_][\w.]*)/g, (_m, l, r) => {
|
|
265
276
|
const [a, b] = [l, r].sort();
|
|
266
277
|
return `${a} = ${b}`;
|
|
267
278
|
});
|
|
@@ -270,6 +281,21 @@ export function normalizeDetail(detail ) {
|
|
|
270
281
|
|
|
271
282
|
const JOIN_NAME_RE = /Join/i;
|
|
272
283
|
|
|
284
|
+
// scanRelationId per plan node, memoized: cachingOpportunity's relation walk and
|
|
285
|
+
// findCompositeCandidates both classify every node of every execution, and a JDBC scan's detail
|
|
286
|
+
// carries its whole inner SQL, so the repeated regex passes were a measurable share of
|
|
287
|
+
// analyze(). Keyed by node identity: a resolved plan tree is never mutated after the parser
|
|
288
|
+
// posts it.
|
|
289
|
+
const relationIdByNode = new WeakMap ();
|
|
290
|
+
function relationIdOf(node ) {
|
|
291
|
+
let rid = relationIdByNode.get(node);
|
|
292
|
+
if (rid === undefined) {
|
|
293
|
+
rid = scanRelationId(node.name ?? '', node.detail ?? '');
|
|
294
|
+
relationIdByNode.set(node, rid);
|
|
295
|
+
}
|
|
296
|
+
return rid;
|
|
297
|
+
}
|
|
298
|
+
|
|
273
299
|
// Structural operator kind for cachingOpportunity's composite detection: 'join' covers every
|
|
274
300
|
// Spark join physical operator; 'union' is Spark's exact `Union` node. CartesianProduct is
|
|
275
301
|
// deliberately excluded (out of scope, as in plan-summary.ts).
|
|
@@ -302,11 +328,12 @@ export function findCompositeCandidates(root ) {
|
|
|
302
328
|
path.pop();
|
|
303
329
|
|
|
304
330
|
const metricNames = (node.metrics ?? []).map((m) => m.name).sort().join(',');
|
|
305
|
-
|
|
331
|
+
// Child digests, as in computePlanShapes: deterministic, so still comparable across executions.
|
|
332
|
+
const childFingerprints = childResults.map((r) => cyrb53(r.fingerprint)).join(',');
|
|
306
333
|
const fingerprint = `${node.name}[${metricNames}]{${childFingerprints}}`;
|
|
307
334
|
|
|
308
335
|
const leafRelationBytes = new Map ();
|
|
309
|
-
const rid =
|
|
336
|
+
const rid = relationIdOf(node);
|
|
310
337
|
if (rid) {
|
|
311
338
|
const bytesMetric = (node.metrics ?? []).find((m) => m.name === FILES_READ_BYTES);
|
|
312
339
|
leafRelationBytes.set(rid, (leafRelationBytes.get(rid) ?? 0) + (bytesMetric ? bytesMetric.value : 0));
|
|
@@ -343,7 +370,7 @@ function firstLeafRelationId(node ) {
|
|
|
343
370
|
let found = null;
|
|
344
371
|
walkPlanTree(node, (n) => {
|
|
345
372
|
if (found) return;
|
|
346
|
-
found =
|
|
373
|
+
found = relationIdOf(n);
|
|
347
374
|
});
|
|
348
375
|
return found;
|
|
349
376
|
}
|
|
@@ -411,6 +438,74 @@ export function findDuplicateSubtrees(
|
|
|
411
438
|
return results;
|
|
412
439
|
}
|
|
413
440
|
|
|
441
|
+
// True when every occurrence also agrees node-for-node on normalized detail (filters, columns,
|
|
442
|
+
// scanned table, literals). The fingerprint ignores detail, so occurrences can be same-shaped
|
|
443
|
+
// branches over different data; only identical ones are repeated work that computing once would
|
|
444
|
+
// save. AQE query-stage numbers (`ShuffleQueryStage 718` vs `720`) name the same computation's
|
|
445
|
+
// runtime stage and are ignored too. On the 14 real logs, 270 of 546 groups differed beyond that.
|
|
446
|
+
function occurrencesHaveIdenticalDetails(occurrences ) {
|
|
447
|
+
const detailsOf = (root ) => {
|
|
448
|
+
const out = [];
|
|
449
|
+
walkPlanTree(root, (n) => out.push(n.detail ?? ''));
|
|
450
|
+
return out;
|
|
451
|
+
};
|
|
452
|
+
const normalize = (d ) => normalizeDetail(d).replace(/\b(\w+QueryStage)\s+\d+/g, '$1');
|
|
453
|
+
const first = detailsOf(occurrences[0]);
|
|
454
|
+
const firstNormalized = [];
|
|
455
|
+
for (const other of occurrences.slice(1)) {
|
|
456
|
+
const details = detailsOf(other);
|
|
457
|
+
if (details.length !== first.length) return false;
|
|
458
|
+
for (let i = 0; i < first.length; i++) {
|
|
459
|
+
if (details[i] === first[i]) continue;
|
|
460
|
+
firstNormalized[i] ??= normalize(first[i]);
|
|
461
|
+
if (normalize(details[i]) !== firstNormalized[i]) return false;
|
|
462
|
+
}
|
|
463
|
+
}
|
|
464
|
+
return true;
|
|
465
|
+
}
|
|
466
|
+
|
|
467
|
+
// Stages that run nothing but the duplicated occurrences' operators: every operator attributed to
|
|
468
|
+
// the stage is inside `nodes`. A stage shared with operators outside them (the join consuming the
|
|
469
|
+
// subtree, the other join side) does other work too, so its time isn't the subtree's to claim;
|
|
470
|
+
// counting it also let sibling groups claim the same stage twice. A WholeStageCodegen wrapper and
|
|
471
|
+
// an Exchange's write half aren't other work (the fused pipeline around an operator, the shuffle
|
|
472
|
+
// write of its output), so they're left out of both counts; on the 14 real logs they were the
|
|
473
|
+
// only outside node on 426 stages.
|
|
474
|
+
function countsTowardStage(node ) {
|
|
475
|
+
return node.exchangeRole !== 'write' && !node.name.startsWith('WholeStageCodegen');
|
|
476
|
+
}
|
|
477
|
+
|
|
478
|
+
function operatorCountByStage(nodes ) {
|
|
479
|
+
const counts = new Map ();
|
|
480
|
+
for (const node of nodes) {
|
|
481
|
+
if (!countsTowardStage(node)) continue;
|
|
482
|
+
for (const sid of node.stageIds ?? []) counts.set(sid, (counts.get(sid) ?? 0) + 1);
|
|
483
|
+
}
|
|
484
|
+
return counts;
|
|
485
|
+
}
|
|
486
|
+
|
|
487
|
+
function stageOperatorShares(nodes , executionCounts ) {
|
|
488
|
+
const shares = {};
|
|
489
|
+
for (const [sid, count] of operatorCountByStage(nodes)) shares[sid] = count / (executionCounts.get(sid) ?? count);
|
|
490
|
+
return shares;
|
|
491
|
+
}
|
|
492
|
+
|
|
493
|
+
// Fingerprint matching compares operator + metric names only, not literal values or expr IDs
|
|
494
|
+
// (see the finding's validationRequired text), so a small pattern repeated the bare minimum
|
|
495
|
+
// number of times is the case most likely to be coincidental rather than real duplicated work.
|
|
496
|
+
// A bigger matched subtree, or more repeats, are each on their own strong corroborating evidence
|
|
497
|
+
// that the match is real: the odds of two semantically-different query branches producing an
|
|
498
|
+
// identical operator-name sequence shrink fast as the sequence grows or repeats.
|
|
499
|
+
function duplicateSubtreeConfidence(
|
|
500
|
+
subtreeSize ,
|
|
501
|
+
occurrences ,
|
|
502
|
+
thresholds ,
|
|
503
|
+
) {
|
|
504
|
+
if (subtreeSize <= thresholds.minSubtreeSize && occurrences <= thresholds.minOccurrences) return 'low';
|
|
505
|
+
if (subtreeSize >= thresholds.minSubtreeSize * 2 || occurrences >= thresholds.minOccurrences + 2) return 'high';
|
|
506
|
+
return 'medium';
|
|
507
|
+
}
|
|
508
|
+
|
|
414
509
|
// Exact metric names Spark emits, verified against a real SQLExecutionStart's sparkPlanInfo.
|
|
415
510
|
// The write-side byte metric is "written output", not "size of written files".
|
|
416
511
|
const FILES_READ_COUNT = 'number of files read';
|
|
@@ -505,12 +600,27 @@ function meetsRuntimeFloor(wasteMs , appDurationMs , floorP
|
|
|
505
600
|
return appDurationMs == null || wasteMs >= appDurationMs * floorPct;
|
|
506
601
|
}
|
|
507
602
|
|
|
603
|
+
// A stage that ran for less than floorPct of the run (a known duration): an estimate clipped to
|
|
604
|
+
// the stage can't reach floorPct, so every finding there grades info. A zero-length stage (no
|
|
605
|
+
// submission time on older Spark) is not skipped: it gets no estimate and keeps its own band.
|
|
606
|
+
function stageBelowRuntimeFloor(stage , ctx , floorPct ) {
|
|
607
|
+
const stageDurationMs = (stage.completedAt ?? 0) - (stage.submittedAt ?? 0);
|
|
608
|
+
const appDurationMs = computeAppDurationMs(ctx);
|
|
609
|
+
return appDurationMs != null && stageDurationMs > 0 && stageDurationMs < appDurationMs * floorPct;
|
|
610
|
+
}
|
|
611
|
+
|
|
508
612
|
// Runs a raw waste delta through the same occupancy clip impact-estimator.ts applies before
|
|
509
613
|
// display, so the runtime floor checks recoverable wall-clock, not a delta a physical floor
|
|
510
614
|
// leaves unrecoverable. Falls back to the raw delta when occupancy data is unavailable.
|
|
511
|
-
|
|
615
|
+
// skew/straggler claims shorten the stage's longest task, hence shortensLongestTask (see occupancy.ts).
|
|
616
|
+
function clippedWasteMs(
|
|
617
|
+
wasteMs , stageId , ctx , removedCoreWorkMs , longestTaskAfterFixMs = 0,
|
|
618
|
+
) {
|
|
512
619
|
if (!ctx) return wasteMs;
|
|
513
|
-
const est = estimateSingleStage(
|
|
620
|
+
const est = estimateSingleStage(
|
|
621
|
+
wasteMs, stageId, ctx.stages , ctx.occupancy,
|
|
622
|
+
{ shortensLongestTask: true, removedCoreWorkMs, longestTaskAfterFixMs },
|
|
623
|
+
);
|
|
514
624
|
return est ? est.wallClock.high : wasteMs;
|
|
515
625
|
}
|
|
516
626
|
|
|
@@ -519,6 +629,19 @@ function clippedWasteMs(wasteMs , stageId , ctx )
|
|
|
519
629
|
const CACHE_UTILIZATION_VALIDATION =
|
|
520
630
|
"This ratio is a point-in-time storage snapshot from stage-submission events, not a runtime read-count. Confirm against the Spark UI's Storage tab before acting.";
|
|
521
631
|
|
|
632
|
+
// Both cachedRatio and diskRatio are percentages of an RDD's partition/byte count: with few
|
|
633
|
+
// partitions, one partition flipping cached/evicted (or disk-resident/memory-resident) swings the
|
|
634
|
+
// reported percentage by a large amount, so the point estimate is noisy. More partitions average
|
|
635
|
+
// that noise out into a stable ratio. numPartitions is the only sample-size signal
|
|
636
|
+
// DetectorRddInfo carries, so it drives confidence for both variants rather than a flat guess.
|
|
637
|
+
// 10/50 mirror this file's other "trust the sample" cutoffs (skew's minTasksForP95: 20,
|
|
638
|
+
// coreLocality's minTasks: 50).
|
|
639
|
+
function cacheSampleConfidence(numPartitions ) {
|
|
640
|
+
if (numPartitions < 10) return 'low';
|
|
641
|
+
if (numPartitions >= 50) return 'high';
|
|
642
|
+
return 'medium';
|
|
643
|
+
}
|
|
644
|
+
|
|
522
645
|
function partialCacheFinding(rdd , cachedRatio , impactBand ) {
|
|
523
646
|
const rddName = rdd.name || `RDD ${rdd.id}`;
|
|
524
647
|
const cachedPct = Math.round(cachedRatio * 100);
|
|
@@ -527,7 +650,7 @@ function partialCacheFinding(rdd , cachedRatio , impactBa
|
|
|
527
650
|
type: 'cacheUtilization', variant: 'partialCache', stageId: null,
|
|
528
651
|
rddId: rdd.id, rddName,
|
|
529
652
|
impactBand, metric: 'cachedRatio', value: cachedPct,
|
|
530
|
-
confidence:
|
|
653
|
+
confidence: cacheSampleConfidence(rdd.numPartitions),
|
|
531
654
|
validationRequired: CACHE_UTILIZATION_VALIDATION,
|
|
532
655
|
memorySize: rdd.memorySize, diskSize: rdd.diskSize,
|
|
533
656
|
numCachedPartitions: rdd.numCachedPartitions, numPartitions: rdd.numPartitions,
|
|
@@ -542,7 +665,7 @@ function diskSpilloverFinding(rdd , diskRatio , impactBan
|
|
|
542
665
|
type: 'cacheUtilization', variant: 'diskSpillover', stageId: null,
|
|
543
666
|
rddId: rdd.id, rddName,
|
|
544
667
|
impactBand, metric: 'diskRatio', value: diskPct,
|
|
545
|
-
confidence:
|
|
668
|
+
confidence: cacheSampleConfidence(rdd.numPartitions),
|
|
546
669
|
validationRequired: CACHE_UTILIZATION_VALIDATION,
|
|
547
670
|
memorySize: rdd.memorySize, diskSize: rdd.diskSize,
|
|
548
671
|
numCachedPartitions: rdd.numCachedPartitions, numPartitions: rdd.numPartitions,
|
|
@@ -584,6 +707,103 @@ function diskSpilloverFinding(rdd , diskRatio , impactBan
|
|
|
584
707
|
export const STRAGGLER_FLOOR_PCT_WARN = 0.005;
|
|
585
708
|
export const STRAGGLER_FLOOR_PCT_CRIT = 0.02;
|
|
586
709
|
|
|
710
|
+
// A ratio just past ratioWarn is the case most likely to be ordinary task-duration variance
|
|
711
|
+
// rather than real skew; a ratio many multiples past it (a 50x P95/median vs. a 3.1x one) is
|
|
712
|
+
// unambiguous. 1.5x/5x mirror this file's other "just past the floor vs. clearly past it" splits
|
|
713
|
+
// (duplicateSubtreeConfidence's 2x, cacheSampleConfidence's 10/50-partition cutoffs).
|
|
714
|
+
function skewConfidence(ratio , ratioWarn ) {
|
|
715
|
+
if (ratio <= ratioWarn * 1.5) return 'low';
|
|
716
|
+
if (ratio >= ratioWarn * 5) return 'high';
|
|
717
|
+
return 'medium';
|
|
718
|
+
}
|
|
719
|
+
|
|
720
|
+
// warnPct100/lowInfoPct100 are this detector's own two thresholds; scale confidence as a multiple
|
|
721
|
+
// of whichever one gates the branch that fired, the same way skewConfidence scales off ratioWarn.
|
|
722
|
+
// High-GC: a pct just past warnPct100 (10%) is likely normal variance, 3x past it is unambiguous.
|
|
723
|
+
// Low-GC ("cost", over-provisioned): a pct just under lowInfoPct100 (5%) is borderline, a pct near
|
|
724
|
+
// zero is unambiguous idle GC.
|
|
725
|
+
function gcConfidence(
|
|
726
|
+
pct ,
|
|
727
|
+
thresholds ,
|
|
728
|
+
direction ,
|
|
729
|
+
) {
|
|
730
|
+
if (direction === 'high') {
|
|
731
|
+
if (pct <= thresholds.warnPct100 * 1.5) return 'low';
|
|
732
|
+
if (pct >= thresholds.warnPct100 * 3) return 'high';
|
|
733
|
+
return 'medium';
|
|
734
|
+
}
|
|
735
|
+
if (pct >= thresholds.lowInfoPct100 * 0.66) return 'low';
|
|
736
|
+
if (pct <= thresholds.lowInfoPct100 * 0.2) return 'high';
|
|
737
|
+
return 'medium';
|
|
738
|
+
}
|
|
739
|
+
|
|
740
|
+
// warnFloor gates the finding, so a value just past it is the weakest evidence this detector can
|
|
741
|
+
// produce. highFloor is critPct: straggler's own second (speculative-share) tier, 2x warnPct
|
|
742
|
+
// (0.10 -> 0.20); reused as the high-confidence bar for the stragglerShare path too since that
|
|
743
|
+
// metric has no dedicated critical tier of its own (see the "no dedicated critical tier" comment
|
|
744
|
+
// on the straggler detector) but is the same 0-1 task-share magnitude.
|
|
745
|
+
function stragglerConfidence(shareValue , warnFloor , highFloor ) {
|
|
746
|
+
if (shareValue < warnFloor * 1.5) return 'low';
|
|
747
|
+
if (shareValue >= highFloor) return 'high';
|
|
748
|
+
return 'medium';
|
|
749
|
+
}
|
|
750
|
+
|
|
751
|
+
// minWasteMs is speculationWaste's own floor; a run that barely clears it (under 1.5x) is the
|
|
752
|
+
// weakest case, one that clears it several times over (4x, i.e. 4 minutes against a 1-minute
|
|
753
|
+
// floor) is unambiguous.
|
|
754
|
+
function speculationWasteConfidence(wastedMs , minWasteMs ) {
|
|
755
|
+
if (wastedMs <= minWasteMs * 1.5) return 'low';
|
|
756
|
+
if (wastedMs >= minWasteMs * 4) return 'high';
|
|
757
|
+
return 'medium';
|
|
758
|
+
}
|
|
759
|
+
|
|
760
|
+
// The finding already gates on wastedMBSeconds > wasteBufferMultiplier * usedMBSeconds, i.e. a
|
|
761
|
+
// ratio of 1 at the floor; scale confidence off that same ratio the way skewConfidence scales off
|
|
762
|
+
// ratioWarn, instead of introducing a second, unrelated multiplier.
|
|
763
|
+
function memoryWasteConfidence(wastedMBSeconds , usedMBSeconds , wasteBufferMultiplier ) {
|
|
764
|
+
const ratio = usedMBSeconds > 0 ? wastedMBSeconds / (wasteBufferMultiplier * usedMBSeconds) : Infinity;
|
|
765
|
+
if (ratio <= 1.5) return 'low';
|
|
766
|
+
if (ratio >= 3) return 'high';
|
|
767
|
+
return 'medium';
|
|
768
|
+
}
|
|
769
|
+
|
|
770
|
+
// Two independent weak spots can each undercut this finding: a ratio just past warnRatio (could
|
|
771
|
+
// be one bad stage), or too few sampled tasks (mirrors cacheSampleConfidence's use of
|
|
772
|
+
// numPartitions as a sample-size signal). Report whichever signal is weaker rather than
|
|
773
|
+
// averaging them away. critRatio is this detector's own existing second tier, reused directly as
|
|
774
|
+
// the ratio high-bar; minTasks*2/*4 mirror the same "2x floor is still weak, 4x is strong" spread
|
|
775
|
+
// used elsewhere in this file.
|
|
776
|
+
function coreLocalityConfidence(
|
|
777
|
+
ratio ,
|
|
778
|
+
totalTasks ,
|
|
779
|
+
thresholds ,
|
|
780
|
+
) {
|
|
781
|
+
const rank = { low: 0, medium: 1, high: 2 } ;
|
|
782
|
+
const ratioTier = ratio < thresholds.warnRatio * 1.5 ? 'low' : ratio >= thresholds.critRatio ? 'high' : 'medium';
|
|
783
|
+
const sampleTier = totalTasks < thresholds.minTasks * 2 ? 'low' : totalTasks >= thresholds.minTasks * 4 ? 'high' : 'medium';
|
|
784
|
+
return rank[ratioTier] <= rank[sampleTier] ? ratioTier : sampleTier;
|
|
785
|
+
}
|
|
786
|
+
|
|
787
|
+
// warningPct/criticalPct are this detector's own two tiers (0.30/0.60); a churn rate just past
|
|
788
|
+
// warningPct is the borderline call the impact-band split already treats as the weaker tier, so
|
|
789
|
+
// reuse criticalPct directly as the high-confidence bar instead of inventing a third figure.
|
|
790
|
+
function autoscalingChurnConfidence(shortLivedPct , warningPct , criticalPct ) {
|
|
791
|
+
if (shortLivedPct <= warningPct * 1.5) return 'low';
|
|
792
|
+
if (shortLivedPct >= criticalPct) return 'high';
|
|
793
|
+
return 'medium';
|
|
794
|
+
}
|
|
795
|
+
|
|
796
|
+
// minExecutions is the bare minimum occurrence count this detector will even emit a finding for;
|
|
797
|
+
// a match at exactly that count is the weakest reuse signal (as likely to be coincidental overlap
|
|
798
|
+
// as real shared work), while 3x the floor is several independent executions all hitting the same
|
|
799
|
+
// relation/composite shape, unambiguous. Mirrors duplicateSubtreeConfidence's occurrences handling
|
|
800
|
+
// for the same reason: repetition count is the strength signal for a structural-match detector.
|
|
801
|
+
function cachingReuseConfidence(occurrences , minExecutions ) {
|
|
802
|
+
if (occurrences <= minExecutions) return 'low';
|
|
803
|
+
if (occurrences >= minExecutions * 3) return 'high';
|
|
804
|
+
return 'medium';
|
|
805
|
+
}
|
|
806
|
+
|
|
587
807
|
export const DETECTORS = [
|
|
588
808
|
{
|
|
589
809
|
type: 'skew', scope: 'stage', order: 30, fixEffort: 'code', version: 1,
|
|
@@ -602,17 +822,18 @@ export const DETECTORS = [
|
|
|
602
822
|
if (ratio <= this.thresholds.ratioWarn) return null;
|
|
603
823
|
// Same absolute delta impact-estimator.ts's 'skew' case reports as savings; clipped the
|
|
604
824
|
// same way before the floor check so the gate agrees with what's displayed.
|
|
605
|
-
const
|
|
825
|
+
const singleDelta = Math.max(0, metric === 'P95/median' ? stage.taskDurationP95 - stage.taskDurationP50 : stage.taskDurationMax - stage.taskDurationP50);
|
|
826
|
+
const wasteMs = tailRecoveryMs(stage, singleDelta);
|
|
606
827
|
const appDurationMs = computeAppDurationMs(ctx);
|
|
607
|
-
const floorWasteMs = clippedWasteMs(wasteMs, stage.id, ctx);
|
|
828
|
+
const floorWasteMs = clippedWasteMs(wasteMs, stage.id, ctx, tailRemovedWorkMs(stage, singleDelta), stragglerFixLongestTaskMs(stage));
|
|
608
829
|
if (!meetsRuntimeFloor(floorWasteMs, appDurationMs, this.thresholds.floorPctWarn)) return null;
|
|
609
830
|
const value = Math.round(ratio * 10) / 10;
|
|
610
831
|
return {
|
|
611
832
|
type: 'skew', stageId: stage.id,
|
|
612
833
|
impactBand: 'warning',
|
|
613
834
|
metric, value,
|
|
614
|
-
confidence:
|
|
615
|
-
validationRequired: '
|
|
835
|
+
confidence: skewConfidence(ratio, this.thresholds.ratioWarn),
|
|
836
|
+
validationRequired: 'This finding is gated by a 0.5% runtime-floor threshold, our own noise floor for this metric.',
|
|
616
837
|
recommendation: `Task duration ratio (${metric}) is ${value}×: for join-driven skew, enable AQE skew-join handling (spark.sql.adaptive.skewJoin.enabled); otherwise salt the key or repartition on a better key to reduce task skew.`,
|
|
617
838
|
};
|
|
618
839
|
},
|
|
@@ -620,9 +841,13 @@ export const DETECTORS = [
|
|
|
620
841
|
{
|
|
621
842
|
type: 'stageShape', scope: 'stage', order: 35, fixEffort: 'code', version: 1,
|
|
622
843
|
docAnchor: '#bottleneck-stage-shape',
|
|
623
|
-
|
|
844
|
+
// lowParallelismFloorPct: the same 0.5% runtime floor the tiered detectors use. Parallelizing a
|
|
845
|
+
// stage can't save more than the stage's own duration, so a shorter stage can't clear it; on
|
|
846
|
+
// the 14 real logs that was 2839 of 3005 lowParallelism findings (2168 on sub-second stages).
|
|
847
|
+
// App-wide idle capacity stays covered by utilization.
|
|
848
|
+
thresholds: { pRatioMax: 0.5, oiRatioMax: 10, skewWarn: 3, lowParallelismFloorPct: 0.005 },
|
|
624
849
|
detect(
|
|
625
|
-
|
|
850
|
+
|
|
626
851
|
stage ,
|
|
627
852
|
ctx ,
|
|
628
853
|
) {
|
|
@@ -630,8 +855,9 @@ export const DETECTORS = [
|
|
|
630
855
|
const execCount = (stage.executorStats ?? []).length;
|
|
631
856
|
const cores = ctx?.app?.resources?.executor?.cores ?? 1;
|
|
632
857
|
const totalCores = execCount * cores;
|
|
858
|
+
const stageDurationMs = (stage.completedAt ?? 0) - (stage.submittedAt ?? 0);
|
|
633
859
|
// PRatio: under-parallelization.
|
|
634
|
-
if (totalCores > 0) {
|
|
860
|
+
if (totalCores > 0 && !stageBelowRuntimeFloor(stage, ctx, this.thresholds.lowParallelismFloorPct)) {
|
|
635
861
|
const pRatio = stage.taskCount / totalCores;
|
|
636
862
|
if (pRatio < this.thresholds.pRatioMax) {
|
|
637
863
|
out.push({
|
|
@@ -657,7 +883,6 @@ export const DETECTORS = [
|
|
|
657
883
|
// TaskStageSkew: straggler cost vs stage wall-clock. Skip near-zero duration. Always info
|
|
658
884
|
// like its siblings: this trigger forces the occupancy-clipped estimate to exactly zero on
|
|
659
885
|
// every firing, so there's no wall-clock-backed tier left to gate on.
|
|
660
|
-
const stageDurationMs = (stage.completedAt ?? 0) - (stage.submittedAt ?? 0);
|
|
661
886
|
if (stageDurationMs > 0) {
|
|
662
887
|
const ratio = stage.taskDurationMax / stageDurationMs;
|
|
663
888
|
if (ratio > this.thresholds.skewWarn) {
|
|
@@ -676,13 +901,18 @@ export const DETECTORS = [
|
|
|
676
901
|
{
|
|
677
902
|
type: 'shuffle', scope: 'stage', order: 20, fixEffort: 'config', version: 1,
|
|
678
903
|
docAnchor: '#bottleneck-shuffle',
|
|
679
|
-
|
|
904
|
+
// stageFloorPct: the 0.5% runtime floor. The shuffle claim is clipped to the stage, so on a
|
|
905
|
+
// shorter stage it graded info: 182 of 284 shuffle findings on the 14 real logs. The shuffle
|
|
906
|
+
// is still there on those stages; the floor is why they're dropped.
|
|
907
|
+
thresholds: { minBytes: 50 * MB, stageFloorPct: 0.005 },
|
|
680
908
|
detect(
|
|
681
|
-
|
|
909
|
+
|
|
682
910
|
stage ,
|
|
911
|
+
ctx ,
|
|
683
912
|
) {
|
|
684
913
|
const bytes = stage.shuffleReadBytes;
|
|
685
914
|
if (bytes <= this.thresholds.minBytes) return null;
|
|
915
|
+
if (stageBelowRuntimeFloor(stage, ctx, this.thresholds.stageFloorPct)) return null;
|
|
686
916
|
return {
|
|
687
917
|
type: 'shuffle', stageId: stage.id,
|
|
688
918
|
impactBand: 'info',
|
|
@@ -693,7 +923,7 @@ export const DETECTORS = [
|
|
|
693
923
|
},
|
|
694
924
|
{
|
|
695
925
|
type: 'partitionSizing', scope: 'stage', order: 22, fixEffort: 'config', version: 1,
|
|
696
|
-
docAnchor: '#bottleneck-
|
|
926
|
+
docAnchor: '#bottleneck-partition-sizing',
|
|
697
927
|
thresholds: { skewRatio: 5, skewFloorBytes: 256 * MB, lowParTotalBytes: GB, lowParMaxTasks: 7, maxPartBytes: 5 * GB },
|
|
698
928
|
detect(
|
|
699
929
|
|
|
@@ -742,9 +972,12 @@ export const DETECTORS = [
|
|
|
742
972
|
{
|
|
743
973
|
type: 'spill', scope: 'stage', order: 10, fixEffort: 'code', version: 1,
|
|
744
974
|
docAnchor: '#bottleneck-spill',
|
|
745
|
-
|
|
746
|
-
|
|
975
|
+
// stageFloorPct: the 0.5% runtime floor, as for shuffle (12 of 38 spill findings on the 14 real
|
|
976
|
+
// logs, all info). The spill is still there on those stages; the floor is why they're dropped.
|
|
977
|
+
thresholds: { singleTaskDiskGiB: 1, singleTaskMemGiB: 4, highDiskGiB: 1, highTaskDiskMB: 512, highMemGiB: 4, medDiskMB: 256, medMemGiB: 1, skewRatio: 5, skewDiskFloorMB: 128, skewMemFloorMB: 256, skewMinTasks: 10, stageFloorPct: 0.005 },
|
|
978
|
+
detect( stage , ctx ) {
|
|
747
979
|
if (stage.memoryBytesSpilled === 0) return null;
|
|
980
|
+
if (stageBelowRuntimeFloor(stage, ctx, this.thresholds.stageFloorPct)) return null;
|
|
748
981
|
const cls = stage.spillClassification;
|
|
749
982
|
const classified = cls === 'skew' || cls === 'volume';
|
|
750
983
|
const mag = computeSpillMagnitude(stage, this.thresholds);
|
|
@@ -766,21 +999,26 @@ export const DETECTORS = [
|
|
|
766
999
|
{
|
|
767
1000
|
type: 'gc', scope: 'stage', order: 50, fixEffort: 'config', version: 1,
|
|
768
1001
|
docAnchor: '#bottleneck-gc',
|
|
769
|
-
|
|
770
|
-
validationRequired: 'The 10-second minimum-runtime floor that gates this finding is our own noise floor, unvalidated: no external tool publishes an equivalent metric to calibrate against. Confirm against known-good/known-bad real logs before trusting the impact-band split.',
|
|
1002
|
+
validationRequired: 'This finding is gated by a 10-second minimum-runtime floor, our own noise floor for this metric.',
|
|
771
1003
|
thresholds: {
|
|
772
1004
|
warnPct100: 10,
|
|
773
1005
|
// Descending tier: ExecutorGcHeuristic, ported as-is.
|
|
774
1006
|
lowInfoPct100: 5,
|
|
775
1007
|
// NOT SOURCED: our own noise floor so a stage that barely ran doesn't flag either direction.
|
|
776
1008
|
minRunTimeMs: 10000,
|
|
1009
|
+
// lowInfoFloorPct: the 0.5% runtime floor stageShape's lowParallelism uses. The low-GC note is
|
|
1010
|
+
// an app-level memory-sizing signal; on a stage shorter than this share of the run it adds
|
|
1011
|
+
// nothing to that call. On the 14 real logs that was 464 of 685 low-GC notes (of 701 gc
|
|
1012
|
+
// findings); the low-GC pattern is still true on those stages, the floor is why they're dropped.
|
|
1013
|
+
lowInfoFloorPct: 0.005,
|
|
777
1014
|
},
|
|
778
1015
|
detect(
|
|
779
1016
|
|
|
780
|
-
|
|
781
|
-
|
|
1017
|
+
|
|
1018
|
+
|
|
782
1019
|
|
|
783
1020
|
stage ,
|
|
1021
|
+
ctx ,
|
|
784
1022
|
) {
|
|
785
1023
|
const pct = stage.gcPct;
|
|
786
1024
|
if ((stage.executorRunTime ?? 0) >= this.thresholds.minRunTimeMs
|
|
@@ -790,19 +1028,20 @@ export const DETECTORS = [
|
|
|
790
1028
|
type: 'gc', stageId: stage.id,
|
|
791
1029
|
impactBand: 'warning',
|
|
792
1030
|
metric: 'gcPct', value,
|
|
793
|
-
confidence: this.
|
|
1031
|
+
confidence: gcConfidence(pct, this.thresholds, 'high'), validationRequired: this.validationRequired,
|
|
794
1032
|
recommendation: `GC consumed ${value}% of executor run time: reduce object creation, use primitive types, avoid UDFs, increase executor memory.`,
|
|
795
1033
|
};
|
|
796
1034
|
}
|
|
797
1035
|
// Low-GC (cost) branch: only for stages that ran long enough to be meaningful.
|
|
798
1036
|
if ((stage.executorRunTime ?? 0) >= this.thresholds.minRunTimeMs
|
|
799
|
-
&& pct < this.thresholds.lowInfoPct100
|
|
1037
|
+
&& pct < this.thresholds.lowInfoPct100
|
|
1038
|
+
&& !stageBelowRuntimeFloor(stage, ctx, this.thresholds.lowInfoFloorPct)) {
|
|
800
1039
|
const value = Math.round(pct * 10) / 10;
|
|
801
1040
|
return {
|
|
802
1041
|
type: 'gc', stageId: stage.id, direction: 'low',
|
|
803
1042
|
impactBand: 'info',
|
|
804
1043
|
metric: 'gcPct', value,
|
|
805
|
-
confidence: this.
|
|
1044
|
+
confidence: gcConfidence(pct, this.thresholds, 'low'), validationRequired: this.validationRequired,
|
|
806
1045
|
recommendation: `GC consumed only ${value}% of executor run time: memory may be over-provisioned; consider reducing spark.executor.memory for cost savings.`,
|
|
807
1046
|
};
|
|
808
1047
|
}
|
|
@@ -818,19 +1057,29 @@ export const DETECTORS = [
|
|
|
818
1057
|
// stages, sub-second/sub-64MB host differences produce huge noise ratios. 1s is above
|
|
819
1058
|
// per-task jitter but below genuine slow-host stages; 64MB mirrors the spill disk-skew floor.
|
|
820
1059
|
floorMs: 1000, floorBytes: 64 * MB,
|
|
1060
|
+
// stageFloorPct: the tiered detectors' 0.5% runtime floor. On a stage shorter than this share
|
|
1061
|
+
// of the run a slow host can't cost that much: duration findings are clipped to the stage
|
|
1062
|
+
// and graded info, and a byte-dimension imbalance (no time estimate, so its ratio tier was
|
|
1063
|
+
// its band) was graded info here since 6298149. So the stage is skipped: on the 14 real logs
|
|
1064
|
+
// 324 of 452 slowHost findings, all info. The imbalance is still there on those stages; the
|
|
1065
|
+
// floor is why they're dropped.
|
|
1066
|
+
stageFloorPct: 0.005,
|
|
821
1067
|
},
|
|
822
1068
|
detect(
|
|
823
1069
|
|
|
824
1070
|
|
|
825
1071
|
|
|
826
1072
|
|
|
1073
|
+
|
|
827
1074
|
|
|
828
1075
|
|
|
829
1076
|
stage ,
|
|
1077
|
+
ctx ,
|
|
830
1078
|
) {
|
|
831
1079
|
const hosts = stage.hostStats ?? [];
|
|
832
1080
|
const execs0 = stage.executorStats ?? [];
|
|
833
1081
|
if ((hosts.length < this.thresholds.minHosts && execs0.length < this.thresholds.minHosts) || stage.taskCount < this.thresholds.minTasks) return null;
|
|
1082
|
+
if (stageBelowRuntimeFloor(stage, ctx, this.thresholds.stageFloorPct)) return null;
|
|
834
1083
|
const out = [];
|
|
835
1084
|
if (hosts.length >= this.thresholds.minHosts) {
|
|
836
1085
|
const means = hosts.map(h => ({ host: h.host, taskCount: h.taskCount, mean: h.totalDuration / h.taskCount }));
|
|
@@ -985,31 +1234,51 @@ export const DETECTORS = [
|
|
|
985
1234
|
docAnchor: '#bottleneck-straggler',
|
|
986
1235
|
// floorPctWarn/floorPctCrit are re-exported as STRAGGLER_FLOOR_PCT_WARN/CRIT and reused as
|
|
987
1236
|
// impact-band.ts's global noise floor: keep the two in sync.
|
|
988
|
-
|
|
1237
|
+
// shareWarnAtFloor: in a large stage, the few stragglers that gate it for tens of seconds can be
|
|
1238
|
+
// only 2.5-5% of its tasks. Scored against a task-level replay of every stage on 14 real logs
|
|
1239
|
+
// (recoverable = replay with each task over 4x P50 capped at P50), admitting 2.5-5% shares
|
|
1240
|
+
// whose clipped tail already clears floorPctWarn found 3 such stages (10-48s) for 1 borderline
|
|
1241
|
+
// miss; admitting every 2.5% share instead added 86 findings below the floor.
|
|
1242
|
+
// A stage shorter than floorPctWarn of the run is skipped outright: its tail can't cost more
|
|
1243
|
+
// than the stage's own duration, so every finding there graded info. On the 14 real logs that
|
|
1244
|
+
// was 671 of 753 straggler findings, none above info; the slow tail is still real on those
|
|
1245
|
+
// stages, the floor is why they're dropped.
|
|
1246
|
+
thresholds: { minTasks: 10, shareWarn: 0.05, shareWarnAtFloor: 0.025, warnPct: 0.10, critPct: 0.20, floorPctWarn: STRAGGLER_FLOOR_PCT_WARN, floorPctCrit: STRAGGLER_FLOOR_PCT_CRIT },
|
|
989
1247
|
detect(
|
|
990
1248
|
|
|
991
|
-
|
|
1249
|
+
|
|
1250
|
+
|
|
1251
|
+
|
|
1252
|
+
|
|
992
1253
|
|
|
993
1254
|
stage ,
|
|
994
1255
|
ctx ,
|
|
995
1256
|
) {
|
|
996
1257
|
if (stage.taskCount < this.thresholds.minTasks) return null;
|
|
1258
|
+
const appDurationMs = computeAppDurationMs(ctx);
|
|
1259
|
+
if (stageBelowRuntimeFloor(stage, ctx, this.thresholds.floorPctWarn)) return null;
|
|
997
1260
|
const stragglerShare = (stage.stragglerCount ?? 0) / stage.taskCount;
|
|
998
|
-
if ((stage.speculativeTasks ?? 0) === 0 && stragglerShare <= this.thresholds.shareWarn) return null;
|
|
999
1261
|
const useSpeculative = (stage.speculativeTasks ?? 0) > 0;
|
|
1262
|
+
if (!useSpeculative && stragglerShare <= this.thresholds.shareWarnAtFloor) return null;
|
|
1000
1263
|
const speculativeShare = useSpeculative ? stage.speculativeTasks / stage.taskCount : 0;
|
|
1001
1264
|
// Same absolute delta impact-estimator.ts's straggler/stageShape case reports as savings: a
|
|
1002
1265
|
// high straggler/speculative share on a stage whose tasks barely vary models near-zero
|
|
1003
1266
|
// savings, so it must not outrank 'info'. Clipped the same way before the floor check.
|
|
1004
|
-
const
|
|
1005
|
-
const
|
|
1006
|
-
const
|
|
1267
|
+
const longestTaskAfterFixMs = stragglerFixLongestTaskMs(stage);
|
|
1268
|
+
const singleDelta = Math.max(0, stage.taskDurationMax - longestTaskAfterFixMs);
|
|
1269
|
+
const wasteMs = tailRecoveryMs(stage, singleDelta);
|
|
1270
|
+
const floorWasteMs = clippedWasteMs(wasteMs, stage.id, ctx, tailRemovedWorkMs(stage, singleDelta), longestTaskAfterFixMs);
|
|
1007
1271
|
const meetsWarnFloor = meetsRuntimeFloor(floorWasteMs, appDurationMs, this.thresholds.floorPctWarn);
|
|
1008
1272
|
const meetsCritFloor = meetsRuntimeFloor(floorWasteMs, appDurationMs, this.thresholds.floorPctCrit);
|
|
1273
|
+
// The lower gate needs positive evidence the tail matters: meetsRuntimeFloor passes by
|
|
1274
|
+
// default when the app's duration is unknown (an incomplete run), which isn't that.
|
|
1275
|
+
const stragglerShareFires = stragglerShare > this.thresholds.shareWarn
|
|
1276
|
+
|| (stragglerShare > this.thresholds.shareWarnAtFloor && appDurationMs != null && meetsWarnFloor);
|
|
1277
|
+
if (!useSpeculative && !stragglerShareFires) return null;
|
|
1009
1278
|
const speculativeTier = speculativeShare >= this.thresholds.critPct && meetsCritFloor ? 'critical'
|
|
1010
1279
|
: speculativeShare >= this.thresholds.warnPct && meetsWarnFloor ? 'warning' : 'info';
|
|
1011
1280
|
// Straggler share has no dedicated critical tier per detector-contract.md; only warning.
|
|
1012
|
-
const stragglerTier =
|
|
1281
|
+
const stragglerTier = stragglerShareFires && meetsWarnFloor ? 'warning' : 'info';
|
|
1013
1282
|
// Fixed fallback: overwritten by deriveImpactBand when this finding gets a real wallClock
|
|
1014
1283
|
// estimate (the common case). Only surfaces on the rare occupancy-sweep miss.
|
|
1015
1284
|
const impactBand = 'info';
|
|
@@ -1028,20 +1297,21 @@ export const DETECTORS = [
|
|
|
1028
1297
|
unit: useSpeculativeMetric ? 'count' : 'pct',
|
|
1029
1298
|
speculativeTasks: stage.speculativeTasks ?? 0,
|
|
1030
1299
|
stragglerCount: stage.stragglerCount ?? 0,
|
|
1031
|
-
confidence:
|
|
1032
|
-
|
|
1300
|
+
confidence: useSpeculativeMetric
|
|
1301
|
+
? stragglerConfidence(speculativeShare, this.thresholds.warnPct, this.thresholds.critPct)
|
|
1302
|
+
: stragglerConfidence(stragglerShare, this.thresholds.shareWarn, this.thresholds.critPct),
|
|
1303
|
+
validationRequired: 'This finding is gated by 0.5%/2% runtime-floor thresholds, our own noise floor for this metric.',
|
|
1033
1304
|
recommendation: `${detail}: rule out a GC pause or a slow shuffle fetch before assuming a hardware issue; if a skewed key is the real cause, that's a candidate for AQE's skew-join handling.`,
|
|
1034
1305
|
};
|
|
1035
1306
|
},
|
|
1036
1307
|
},
|
|
1037
1308
|
{
|
|
1038
1309
|
type: 'speculationWaste', scope: 'stage', order: 71, fixEffort: 'config', version: 1,
|
|
1039
|
-
docAnchor: '#bottleneck-
|
|
1310
|
+
docAnchor: '#bottleneck-speculation-waste',
|
|
1040
1311
|
thresholds: { minWasted: 5, minWasteMs: 60000 },
|
|
1041
1312
|
detect(
|
|
1042
1313
|
|
|
1043
1314
|
|
|
1044
|
-
|
|
1045
1315
|
|
|
1046
1316
|
stage ,
|
|
1047
1317
|
) {
|
|
@@ -1052,7 +1322,7 @@ export const DETECTORS = [
|
|
|
1052
1322
|
type: 'speculationWaste', stageId: stage.id,
|
|
1053
1323
|
impactBand: 'warning',
|
|
1054
1324
|
metric: 'speculationWasteMs', value: wastedMs,
|
|
1055
|
-
confidence: this.
|
|
1325
|
+
confidence: speculationWasteConfidence(wastedMs, this.thresholds.minWasteMs),
|
|
1056
1326
|
recommendation: `Speculative execution discarded ${Math.round(wastedMs / 1000)}s of executor time in this stage; if task durations are naturally variable rather than genuine stragglers, consider tuning spark.speculation.multiplier/quantile.`,
|
|
1057
1327
|
};
|
|
1058
1328
|
},
|
|
@@ -1083,12 +1353,17 @@ export const DETECTORS = [
|
|
|
1083
1353
|
{
|
|
1084
1354
|
type: 'tinyTask', scope: 'stage', order: 80, fixEffort: 'code', version: 1,
|
|
1085
1355
|
docAnchor: '#bottleneck-tiny-tasks',
|
|
1086
|
-
|
|
1356
|
+
// stageFloorPct: the tiered detectors' 0.5% runtime floor. Coalescing can't save more than the
|
|
1357
|
+
// stage's own duration, so on a shorter stage every finding graded info: on the 14 real logs
|
|
1358
|
+
// 132 of 151 tinyTask findings. The tasks are still tiny there; the floor is why they're dropped.
|
|
1359
|
+
thresholds: { minTasks: 100, maxP50: 500, maxP95: 1000, stageFloorPct: 0.005 },
|
|
1087
1360
|
detect(
|
|
1088
|
-
|
|
1361
|
+
|
|
1089
1362
|
stage ,
|
|
1363
|
+
ctx ,
|
|
1090
1364
|
) {
|
|
1091
1365
|
if (stage.taskCount < this.thresholds.minTasks) return null;
|
|
1366
|
+
if (stageBelowRuntimeFloor(stage, ctx, this.thresholds.stageFloorPct)) return null;
|
|
1092
1367
|
if (stage.taskDurationP50 > this.thresholds.maxP50 || stage.taskDurationP95 > this.thresholds.maxP95) return null;
|
|
1093
1368
|
const coalesceTo = Math.max(1, Math.round(stage.taskCount / 10));
|
|
1094
1369
|
const fix = stage.shuffleReadBytes > 0
|
|
@@ -1124,23 +1399,42 @@ export const DETECTORS = [
|
|
|
1124
1399
|
|
|
1125
1400
|
ctx ,
|
|
1126
1401
|
) {
|
|
1127
|
-
const { app, stages } = ctx;
|
|
1402
|
+
const { app, stages, executorsAdded, executorsRemoved } = ctx;
|
|
1128
1403
|
// Nullish (not falsy) check: a literal startTime:0 must not be treated as "missing".
|
|
1129
1404
|
if (!app || app.startTime == null || stages.size === 0) return null;
|
|
1130
|
-
let
|
|
1405
|
+
let firstStageSubmitted = Infinity;
|
|
1131
1406
|
for (const stage of stages.values()) {
|
|
1132
|
-
if (stage.submittedAt > 0 && stage.submittedAt <
|
|
1407
|
+
if (stage.submittedAt > 0 && stage.submittedAt < firstStageSubmitted) firstStageSubmitted = stage.submittedAt;
|
|
1133
1408
|
}
|
|
1134
1409
|
// No stage ever recorded a submission timestamp: no basis to measure a startup gap against.
|
|
1135
|
-
|
|
1136
|
-
|
|
1137
|
-
|
|
1410
|
+
if (!Number.isFinite(firstStageSubmitted)) return null;
|
|
1411
|
+
// The wait is from the first runnable stage to the first executor, not from app start: the
|
|
1412
|
+
// driver's own startup before its first job (36-47s on every real log, whatever the
|
|
1413
|
+
// executors did) isn't something executors could shorten. On the 9 real logs that fired,
|
|
1414
|
+
// the first executor arrived 146-192s after the first stage on two whose old gap read ~40s,
|
|
1415
|
+
// and 2s after it on one the old gap flagged critical at 42s.
|
|
1416
|
+
// An executor added before the first stage only counts if it was still alive at submission:
|
|
1417
|
+
// with dynamic allocation scaling to zero, early executors can idle out before any job runs.
|
|
1418
|
+
const removedAt = new Map ();
|
|
1419
|
+
for (const ev of executorsRemoved) removedAt.set(ev.executorId, ev.timestamp);
|
|
1420
|
+
let firstExecutorAdded = Infinity;
|
|
1421
|
+
for (const e of executorsAdded) {
|
|
1422
|
+
if (!(e.timestamp > 0)) continue;
|
|
1423
|
+
if (e.timestamp <= firstStageSubmitted) {
|
|
1424
|
+
const removed = removedAt.get(e.executorId);
|
|
1425
|
+
if (removed == null || removed > firstStageSubmitted) return null;
|
|
1426
|
+
} else if (e.timestamp < firstExecutorAdded) {
|
|
1427
|
+
firstExecutorAdded = e.timestamp;
|
|
1428
|
+
}
|
|
1429
|
+
}
|
|
1430
|
+
if (!Number.isFinite(firstExecutorAdded)) return null;
|
|
1431
|
+
const gapSeconds = (firstExecutorAdded - firstStageSubmitted) / 1000;
|
|
1138
1432
|
if (gapSeconds <= this.thresholds.gapSeconds) return null;
|
|
1139
1433
|
const value = Math.round(gapSeconds);
|
|
1140
1434
|
return {
|
|
1141
1435
|
type: 'coldStart', stageId: null, impactBand: 'warning',
|
|
1142
1436
|
metric: 'startupGapSeconds', value,
|
|
1143
|
-
recommendation: `The first
|
|
1437
|
+
recommendation: `The first stage waited ${value}s for an executor to become available: keep a warm pool of idle executors, or if using dynamic allocation, raise the minimum/initial executor count so it doesn't scale up from zero.`,
|
|
1144
1438
|
};
|
|
1145
1439
|
},
|
|
1146
1440
|
},
|
|
@@ -1298,8 +1592,8 @@ export const DETECTORS = [
|
|
|
1298
1592
|
out.push({
|
|
1299
1593
|
type: 'memoryUtilization', variant: 'wasteModel', stageId: null,
|
|
1300
1594
|
impactBand: 'info', metric: 'wastedMBSeconds', value,
|
|
1301
|
-
confidence:
|
|
1302
|
-
validationRequired: 'Memory-waste estimate uses allocated-vs-used memory-time and
|
|
1595
|
+
confidence: memoryWasteConfidence(wastedMBSeconds, usedMBSeconds, this.thresholds.wasteBufferMultiplier),
|
|
1596
|
+
validationRequired: 'Memory-waste estimate uses allocated-vs-used memory-time and a 1.5x buffer: confirm against the Spark UI before acting.',
|
|
1303
1597
|
recommendation: `Allocated executor memory sat largely idle over the run (~${value.toLocaleString('en-US')} MB-seconds wasted): review spark.executor.memory and executor count.`,
|
|
1304
1598
|
});
|
|
1305
1599
|
}
|
|
@@ -1313,7 +1607,7 @@ export const DETECTORS = [
|
|
|
1313
1607
|
// block-access events, so a literal cache hit rate isn't derivable). Two per-RDD tiered
|
|
1314
1608
|
// checks over rddInfo snapshots: partial caching and disk spillover. An RDD can produce both.
|
|
1315
1609
|
type: 'cacheUtilization', scope: 'app', order: 103, fixEffort: 'code', version: 1,
|
|
1316
|
-
docAnchor: '#
|
|
1610
|
+
docAnchor: '#bottleneck-cache-utilization',
|
|
1317
1611
|
thresholds: {
|
|
1318
1612
|
cachedRatioWarn: 0.50, cachedRatioInfo: 0.90,
|
|
1319
1613
|
diskRatioWarn: 0.40, diskRatioInfo: 0.15,
|
|
@@ -1357,7 +1651,7 @@ export const DETECTORS = [
|
|
|
1357
1651
|
// half of the "Wasted Cores Ratio" (idle-core half is memoryUtilization's idleCores).
|
|
1358
1652
|
// NO_PREF stays in the denominator only: shuffle-read stages legitimately report it.
|
|
1359
1653
|
type: 'coreLocality', scope: 'app', order: 103, fixEffort: 'config', version: 1,
|
|
1360
|
-
docAnchor: '#bottleneck-
|
|
1654
|
+
docAnchor: '#bottleneck-core-locality',
|
|
1361
1655
|
thresholds: { minTasks: 50, warnRatio: 0.15, critRatio: 0.35 },
|
|
1362
1656
|
detect(
|
|
1363
1657
|
|
|
@@ -1376,8 +1670,8 @@ export const DETECTORS = [
|
|
|
1376
1670
|
metric: 'nonLocalRatio', value,
|
|
1377
1671
|
// Raw count behind the ratio, for the impact estimator. Non-null whenever totalTasks is.
|
|
1378
1672
|
nonLocalTaskCount: nonLocalTasks ,
|
|
1379
|
-
confidence:
|
|
1380
|
-
validationRequired: '
|
|
1673
|
+
confidence: coreLocalityConfidence(ratio , totalTasks, this.thresholds),
|
|
1674
|
+
validationRequired: 'This finding is gated by 15%/35% non-local-ratio thresholds (and a 50-task minimum), our own noise floor for this metric.',
|
|
1381
1675
|
recommendation: `${value}% of tasks (${nonLocalTasks }) ran without process- or node-local data placement: check spark.locality.wait settings and executor/data colocation.`,
|
|
1382
1676
|
};
|
|
1383
1677
|
},
|
|
@@ -1387,12 +1681,11 @@ export const DETECTORS = [
|
|
|
1387
1681
|
// re-provisioning, not normal scale-down). Reuses utilization's add/remove matching, but
|
|
1388
1682
|
// measures lifetime against a threshold instead of aggregate active-time.
|
|
1389
1683
|
type: 'autoscalingChurn', scope: 'app', order: 103, fixEffort: 'config', version: 1,
|
|
1390
|
-
|
|
1684
|
+
docAnchor: '#bottleneck-autoscaling-churn',
|
|
1391
1685
|
thresholds: { shortLivedMs: 120_000, warningPct: 0.30, criticalPct: 0.60, minExecutors: 5 },
|
|
1392
1686
|
detect(
|
|
1393
1687
|
|
|
1394
1688
|
|
|
1395
|
-
|
|
1396
1689
|
|
|
1397
1690
|
ctx ,
|
|
1398
1691
|
) {
|
|
@@ -1421,7 +1714,7 @@ export const DETECTORS = [
|
|
|
1421
1714
|
metric: 'shortLivedExecutorPct', value: pct,
|
|
1422
1715
|
// Raw count behind the percentage, for the impact estimator's startup-overhead figure.
|
|
1423
1716
|
shortLivedExecutorCount: shortLivedCount,
|
|
1424
|
-
confidence: this.
|
|
1717
|
+
confidence: autoscalingChurnConfidence(shortLivedPct, this.thresholds.warningPct, this.thresholds.criticalPct),
|
|
1425
1718
|
recommendation: `${pct}% of executors ran for under 2 minutes before being removed. This looks like wasteful re-provisioning rather than normal scale-down; consider raising spark.dynamicAllocation.executorIdleTimeout or widening the minExecutors/maxExecutors bounds to reduce flapping.`,
|
|
1426
1719
|
};
|
|
1427
1720
|
},
|
|
@@ -1430,7 +1723,7 @@ export const DETECTORS = [
|
|
|
1430
1723
|
// Cross-execution relation reuse: flags an input relation scanned by two or more SQL
|
|
1431
1724
|
// executions in one run, firing on real relation names (parquet:..., jdbc:...).
|
|
1432
1725
|
type: 'cachingOpportunity', scope: 'app', order: 105, fixEffort: 'code', version: 1,
|
|
1433
|
-
docAnchor: '#bottleneck-
|
|
1726
|
+
docAnchor: '#bottleneck-caching-opportunity',
|
|
1434
1727
|
thresholds: { minExecutions: 2 },
|
|
1435
1728
|
detect( ctx ) {
|
|
1436
1729
|
const sql = ctx.sql;
|
|
@@ -1457,7 +1750,7 @@ export const DETECTORS = [
|
|
|
1457
1750
|
// Dedupe relations within one execution (self-joins count once), summing read bytes per relation.
|
|
1458
1751
|
const perExec = new Map ();
|
|
1459
1752
|
walkPlanTree(exec.planTree, (node) => {
|
|
1460
|
-
const rid =
|
|
1753
|
+
const rid = relationIdOf(node);
|
|
1461
1754
|
if (!rid) return;
|
|
1462
1755
|
const bytesMetric = (node.metrics ?? []).find(m => m.name === FILES_READ_BYTES);
|
|
1463
1756
|
perExec.set(rid, (perExec.get(rid) ?? 0) + (bytesMetric ? bytesMetric.value : 0));
|
|
@@ -1562,7 +1855,7 @@ export const DETECTORS = [
|
|
|
1562
1855
|
metric: 'executionReuse', value,
|
|
1563
1856
|
format: 'derived', relations, operator: agg.operator, relation: relationDisplay,
|
|
1564
1857
|
executionIds: finalExecutionIds, totalReadBytes,
|
|
1565
|
-
confidence:
|
|
1858
|
+
confidence: cachingReuseConfidence(value, this.thresholds.minExecutions),
|
|
1566
1859
|
validationRequired:
|
|
1567
1860
|
'Composite reuse is inferred from a structural plan-shape match (operator + normalized ' +
|
|
1568
1861
|
'join/filter condition + child shapes) across SQL executions; confirm these executions ' +
|
|
@@ -1593,7 +1886,7 @@ export const DETECTORS = [
|
|
|
1593
1886
|
relation: agg.relation, format: agg.format,
|
|
1594
1887
|
executionIds: residualExecutionIds.sort((a, b) => a - b),
|
|
1595
1888
|
totalReadBytes,
|
|
1596
|
-
confidence:
|
|
1889
|
+
confidence: cachingReuseConfidence(value, this.thresholds.minExecutions),
|
|
1597
1890
|
validationRequired:
|
|
1598
1891
|
'Relation-reuse is inferred from the pre-AQE plan scan identity across SQL ' +
|
|
1599
1892
|
'executions; confirm the reads are the same data and cacheable within one ' +
|
|
@@ -1721,9 +2014,13 @@ export const DETECTORS = [
|
|
|
1721
2014
|
{
|
|
1722
2015
|
type: 'duplicatePlanSubtree', scope: 'sql', order: 130, fixEffort: 'code', version: 2,
|
|
1723
2016
|
docAnchor: '#bottleneck-duplicate-plan-subtree',
|
|
1724
|
-
|
|
2017
|
+
// stageFloorPct: the 0.5% runtime floor. The claim counts at most each linked stage's own
|
|
2018
|
+
// task-active time, so a repeat whose stages together lasted less than this share of the run
|
|
2019
|
+
// graded info: 340 of 545 findings on the 14 real logs. The repeat is still in the plan; the
|
|
2020
|
+
// floor is why they're dropped. A repeat with no linked stage time is kept.
|
|
2021
|
+
thresholds: { minSubtreeSize: 3, minOccurrences: 2, stageFloorPct: 0.005 },
|
|
1725
2022
|
detect(
|
|
1726
|
-
|
|
2023
|
+
|
|
1727
2024
|
sqlExec ,
|
|
1728
2025
|
ctx ,
|
|
1729
2026
|
) {
|
|
@@ -1731,26 +2028,44 @@ export const DETECTORS = [
|
|
|
1731
2028
|
const groups = findDuplicateSubtrees(sqlExec.planTree, this.thresholds);
|
|
1732
2029
|
if (groups.length === 0) return null;
|
|
1733
2030
|
const fallbackStageIds = stageIdsForSqlExec(sqlExec.id, ctx.stages);
|
|
1734
|
-
|
|
2031
|
+
const executionNodes = [];
|
|
2032
|
+
walkPlanTree(sqlExec.planTree, (node) => executionNodes.push(node));
|
|
2033
|
+
const operatorsByStage = operatorCountByStage(executionNodes);
|
|
2034
|
+
const appDurationMs = computeAppDurationMs(ctx);
|
|
2035
|
+
const findings = groups.map((g) => {
|
|
1735
2036
|
const nodes = [];
|
|
1736
2037
|
for (const n of g.nodes) walkPlanTree(n, (node) => nodes.push(node));
|
|
1737
2038
|
const stageIds = unionStageIds(nodes, fallbackStageIds);
|
|
2039
|
+
let stagesMs = 0;
|
|
2040
|
+
for (const id of stageIds) {
|
|
2041
|
+
const stage = ctx.stages.get(id);
|
|
2042
|
+
if (stage) stagesMs += Math.max(0, (stage.completedAt ?? 0) - (stage.submittedAt ?? 0));
|
|
2043
|
+
}
|
|
2044
|
+
if (appDurationMs != null && stagesMs > 0 && stagesMs < appDurationMs * this.thresholds.stageFloorPct) return null;
|
|
2045
|
+
const occurrencesIdentical = occurrencesHaveIdenticalDetails(g.nodes);
|
|
2046
|
+
const stageShares = stageOperatorShares(nodes, operatorsByStage);
|
|
1738
2047
|
// resolvePlanTree always sets id; safe downstream of it.
|
|
1739
2048
|
const planNodeIds = nodes.map((n) => n.id ).filter(Boolean);
|
|
1740
2049
|
const touching = g.sampleRelation ? ` (touching ${g.sampleRelation})` : '';
|
|
2050
|
+
const differing = occurrencesIdentical ? '' : ' Their filters, columns or scanned tables differ, so the repeats may compute different data.';
|
|
1741
2051
|
return {
|
|
1742
2052
|
type: 'duplicatePlanSubtree', executionId: sqlExec.id, stageIds, planNodeIds,
|
|
2053
|
+
stageShares, occurrencesIdentical,
|
|
1743
2054
|
// Fixed fallback: overwritten by deriveImpactBand when this finding gets a real
|
|
1744
|
-
// wallClock estimate
|
|
1745
|
-
|
|
2055
|
+
// wallClock estimate. Repeats with differing details get no estimate (nothing is
|
|
2056
|
+
// known to be recomputed), and neither does a subtree with no stage of its own.
|
|
2057
|
+
impactBand: occurrencesIdentical && Object.keys(stageShares).length > 0 ? 'warning' : 'info',
|
|
2058
|
+
metric: 'subtreeOccurrences', value: g.occurrences,
|
|
1746
2059
|
rootName: g.rootName, subtreeSize: g.subtreeSize, sampleRelation: g.sampleRelation,
|
|
1747
|
-
groupIndex: g.groupIndex,
|
|
2060
|
+
groupIndex: g.groupIndex,
|
|
2061
|
+
confidence: occurrencesIdentical ? duplicateSubtreeConfidence(g.subtreeSize, g.occurrences, this.thresholds) : 'low',
|
|
1748
2062
|
validationRequired: 'Duplicate-subtree matching compares operator names and metric names only, not literal values or expr IDs: confirm the repeated work is real in the Spark SQL plan tab before acting.',
|
|
1749
|
-
recommendation: g.isExchangeRoot
|
|
2063
|
+
recommendation: (g.isExchangeRoot
|
|
1750
2064
|
? `A ${g.subtreeSize}-node subtree rooted at ${pathBasename(g.rootName)} repeats ${g.occurrences}x in this plan${touching}: this looks like a possible missed exchange reuse; check whether the same shuffle could be computed once and reused.`
|
|
1751
|
-
: `A ${g.subtreeSize}-node subtree rooted at ${pathBasename(g.rootName)} repeats ${g.occurrences}x in this plan${touching}: consider caching/persisting the shared computation or check for a duplicated query branch
|
|
2065
|
+
: `A ${g.subtreeSize}-node subtree rooted at ${pathBasename(g.rootName)} repeats ${g.occurrences}x in this plan${touching}: consider caching/persisting the shared computation or check for a duplicated query branch.`) + differing,
|
|
1752
2066
|
};
|
|
1753
|
-
});
|
|
2067
|
+
}).filter((f) => f !== null);
|
|
2068
|
+
return findings.length > 0 ? findings : null;
|
|
1754
2069
|
},
|
|
1755
2070
|
},
|
|
1756
2071
|
{
|