sparkforensics-cli 0.2.0 → 0.2.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/export-template/docs/404.html +1 -1
- package/export-template/docs/assets/{app.CndaAS6v.js → app.DQTZyGL1.js} +1 -1
- package/export-template/docs/assets/chunks/@localSearchIndexroot.DppnXnDE.js +1 -0
- package/export-template/docs/assets/chunks/{VPLocalSearchBox.yJbZbsEo.js → VPLocalSearchBox.BkBIPFs6.js} +1 -1
- package/export-template/docs/assets/chunks/theme.DP0u1AUq.js +2 -0
- package/export-template/docs/assets/contributor-guide_architecture_board-widgets.md.CWpj01WU.js +1 -0
- package/export-template/docs/assets/contributor-guide_architecture_board-widgets.md.CWpj01WU.lean.js +1 -0
- package/export-template/docs/assets/contributor-guide_architecture_detector-contract.md.CgzUsQ6W.js +1 -0
- package/export-template/docs/assets/{contributor-guide_architecture_impact-estimation.md.DYCDPgkh.js → contributor-guide_architecture_impact-estimation.md.CooslVJt.js} +1 -1
- package/export-template/docs/assets/contributor-guide_architecture_state-and-history.md.C-xxn0q7.js +1 -0
- package/export-template/docs/assets/contributor-guide_architecture_state-and-history.md.C-xxn0q7.lean.js +1 -0
- package/export-template/docs/assets/contributor-guide_architecture_widget-rendering.md.R27gQrgY.js +1 -0
- package/export-template/docs/assets/contributor-guide_architecture_worker-protocol.md.IbnfNrV3.js +6 -0
- package/export-template/docs/assets/{style.DXOMCXxn.css → style.DSixAiZE.css} +1 -1
- package/export-template/docs/assets/{user-guide_alternative-log-retrieval.md.sU3KGarf.js → user-guide_alternative-log-retrieval.md.B4tPGIal.js} +1 -1
- package/export-template/docs/assets/{user-guide_alternative-log-retrieval.md.sU3KGarf.lean.js → user-guide_alternative-log-retrieval.md.B4tPGIal.lean.js} +1 -1
- package/export-template/docs/assets/user-guide_getting-started.md.BJvwLEIM.js +3 -0
- package/export-template/docs/assets/user-guide_getting-started.md.BJvwLEIM.lean.js +1 -0
- package/export-template/docs/assets/{user-guide_mcp-tools.md.C8MiIu7F.js → user-guide_mcp-tools.md.Vi3RoflJ.js} +3 -3
- package/export-template/docs/assets/{user-guide_mcp-tools.md.C8MiIu7F.lean.js → user-guide_mcp-tools.md.Vi3RoflJ.lean.js} +1 -1
- package/export-template/docs/assets/user-guide_run-comparison.md.CQc1aoU8.js +1 -0
- package/export-template/docs/assets/user-guide_run-comparison.md.CQc1aoU8.lean.js +1 -0
- package/export-template/docs/assets/user-guide_understanding-findings.md.DL1UDhvR.js +1 -0
- package/export-template/docs/assets/user-guide_understanding-findings.md.DL1UDhvR.lean.js +1 -0
- package/export-template/docs/contributor-guide/architecture/board-widgets.html +2 -2
- package/export-template/docs/contributor-guide/architecture/detector-contract.html +2 -2
- package/export-template/docs/contributor-guide/architecture/drill-down.html +1 -1
- package/export-template/docs/contributor-guide/architecture/impact-estimation.html +2 -2
- package/export-template/docs/contributor-guide/architecture/index.html +1 -1
- package/export-template/docs/contributor-guide/architecture/overview.html +1 -1
- package/export-template/docs/contributor-guide/architecture/state-and-history.html +2 -2
- package/export-template/docs/contributor-guide/architecture/widget-rendering.html +2 -2
- package/export-template/docs/contributor-guide/architecture/worker-protocol.html +2 -2
- package/export-template/docs/contributor-guide/contributing.html +1 -1
- package/export-template/docs/contributor-guide/development-setup.html +1 -1
- package/export-template/docs/contributor-guide/testing.html +1 -1
- package/export-template/docs/index.html +1 -1
- package/export-template/docs/tuning-reference/anti-patterns.html +1 -1
- package/export-template/docs/tuning-reference/aqe.html +1 -1
- package/export-template/docs/tuning-reference/bottleneck-broadcast-sizing.html +1 -1
- package/export-template/docs/tuning-reference/bottleneck-cold-start.html +1 -1
- package/export-template/docs/tuning-reference/bottleneck-duplicate-plan-subtree.html +1 -1
- package/export-template/docs/tuning-reference/bottleneck-failures.html +1 -1
- package/export-template/docs/tuning-reference/bottleneck-gc.html +1 -1
- package/export-template/docs/tuning-reference/bottleneck-job-failure-rate.html +1 -1
- package/export-template/docs/tuning-reference/bottleneck-memory-utilization.html +1 -1
- package/export-template/docs/tuning-reference/bottleneck-retry-waste.html +1 -1
- package/export-template/docs/tuning-reference/bottleneck-shuffle.html +1 -1
- package/export-template/docs/tuning-reference/bottleneck-skew.html +1 -1
- package/export-template/docs/tuning-reference/bottleneck-slow-host.html +1 -1
- package/export-template/docs/tuning-reference/bottleneck-small-files.html +1 -1
- package/export-template/docs/tuning-reference/bottleneck-spill.html +1 -1
- package/export-template/docs/tuning-reference/bottleneck-straggler.html +1 -1
- package/export-template/docs/tuning-reference/bottleneck-tiny-tasks.html +1 -1
- package/export-template/docs/tuning-reference/bottleneck-utilization.html +1 -1
- package/export-template/docs/tuning-reference/caching.html +1 -1
- package/export-template/docs/tuning-reference/cluster-config.html +1 -1
- package/export-template/docs/tuning-reference/config.html +1 -1
- package/export-template/docs/tuning-reference/data-formats.html +1 -1
- package/export-template/docs/tuning-reference/index.html +1 -1
- package/export-template/docs/tuning-reference/intro.html +1 -1
- package/export-template/docs/tuning-reference/joins.html +1 -1
- package/export-template/docs/tuning-reference/memory-model.html +1 -1
- package/export-template/docs/tuning-reference/metrics.html +1 -1
- package/export-template/docs/tuning-reference/partitioning.html +1 -1
- package/export-template/docs/tuning-reference/pyspark.html +1 -1
- package/export-template/docs/tuning-reference/shuffle.html +1 -1
- package/export-template/docs/tuning-reference/spark-architecture.html +1 -1
- package/export-template/docs/tuning-reference/table-formats.html +1 -1
- package/export-template/docs/user-guide/alternative-log-retrieval.html +2 -2
- package/export-template/docs/user-guide/getting-started.html +3 -3
- package/export-template/docs/user-guide/mcp-tools.html +4 -4
- package/export-template/docs/user-guide/run-comparison.html +2 -2
- package/export-template/docs/user-guide/understanding-findings.html +2 -2
- package/export-template/index.html +75 -79
- package/export-template/parser-worker-DyjiQvfP.js +112 -0
- package/export-template/sample-runs/sample-run.ndjson.gz +0 -0
- package/package.json +5 -4
- package/vendor-core/detectors.js +149 -24
- package/vendor-core/docs-content/detection/cache.md +4 -3
- package/vendor-core/docs-content/detection/chrn.md +4 -2
- package/vendor-core/docs-content/detection/local.md +2 -3
- package/vendor-core/docs-content/detection/mem.md +3 -3
- package/vendor-core/docs-content/detection/spec.md +4 -3
- package/vendor-core/evidence-report.js +1 -1
- package/vendor-core/parser-worker.js +2 -2
- package/vendor-core/run-comparison.js +21 -2
- package/export-template/docs/assets/chunks/@localSearchIndexroot.DNY8bVcl.js +0 -1
- package/export-template/docs/assets/chunks/theme.Df2VAG9w.js +0 -2
- package/export-template/docs/assets/contributor-guide_architecture_board-widgets.md.B-OsL91z.js +0 -1
- package/export-template/docs/assets/contributor-guide_architecture_board-widgets.md.B-OsL91z.lean.js +0 -1
- package/export-template/docs/assets/contributor-guide_architecture_detector-contract.md.BOeH4d1J.js +0 -1
- package/export-template/docs/assets/contributor-guide_architecture_state-and-history.md.m3S3UdMk.js +0 -1
- package/export-template/docs/assets/contributor-guide_architecture_state-and-history.md.m3S3UdMk.lean.js +0 -1
- package/export-template/docs/assets/contributor-guide_architecture_widget-rendering.md.DbqPf2OT.js +0 -1
- package/export-template/docs/assets/contributor-guide_architecture_worker-protocol.md.B93qJ_tT.js +0 -6
- package/export-template/docs/assets/user-guide_getting-started.md.DtEM37MK.js +0 -3
- package/export-template/docs/assets/user-guide_getting-started.md.DtEM37MK.lean.js +0 -1
- package/export-template/docs/assets/user-guide_run-comparison.md.S0TWWmLY.js +0 -1
- package/export-template/docs/assets/user-guide_run-comparison.md.S0TWWmLY.lean.js +0 -1
- package/export-template/docs/assets/user-guide_understanding-findings.md.D0R_Y-R2.js +0 -1
- package/export-template/docs/assets/user-guide_understanding-findings.md.D0R_Y-R2.lean.js +0 -1
- package/export-template/parser-worker-QqyEE4m9.js +0 -64
- /package/export-template/docs/assets/{contributor-guide_architecture_detector-contract.md.BOeH4d1J.lean.js → contributor-guide_architecture_detector-contract.md.CgzUsQ6W.lean.js} +0 -0
- /package/export-template/docs/assets/{contributor-guide_architecture_impact-estimation.md.DYCDPgkh.lean.js → contributor-guide_architecture_impact-estimation.md.CooslVJt.lean.js} +0 -0
- /package/export-template/docs/assets/{contributor-guide_architecture_widget-rendering.md.DbqPf2OT.lean.js → contributor-guide_architecture_widget-rendering.md.R27gQrgY.lean.js} +0 -0
- /package/export-template/docs/assets/{contributor-guide_architecture_worker-protocol.md.B93qJ_tT.lean.js → contributor-guide_architecture_worker-protocol.md.IbnfNrV3.lean.js} +0 -0
package/vendor-core/detectors.js
CHANGED
|
@@ -411,6 +411,22 @@ export function findDuplicateSubtrees(
|
|
|
411
411
|
return results;
|
|
412
412
|
}
|
|
413
413
|
|
|
414
|
+
// Fingerprint matching compares operator + metric names only, not literal values or expr IDs
|
|
415
|
+
// (see the finding's validationRequired text), so a small pattern repeated the bare minimum
|
|
416
|
+
// number of times is the case most likely to be coincidental rather than real duplicated work.
|
|
417
|
+
// A bigger matched subtree, or more repeats, are each on their own strong corroborating evidence
|
|
418
|
+
// that the match is real: the odds of two semantically-different query branches producing an
|
|
419
|
+
// identical operator-name sequence shrink fast as the sequence grows or repeats.
|
|
420
|
+
function duplicateSubtreeConfidence(
|
|
421
|
+
subtreeSize ,
|
|
422
|
+
occurrences ,
|
|
423
|
+
thresholds ,
|
|
424
|
+
) {
|
|
425
|
+
if (subtreeSize <= thresholds.minSubtreeSize && occurrences <= thresholds.minOccurrences) return 'low';
|
|
426
|
+
if (subtreeSize >= thresholds.minSubtreeSize * 2 || occurrences >= thresholds.minOccurrences + 2) return 'high';
|
|
427
|
+
return 'medium';
|
|
428
|
+
}
|
|
429
|
+
|
|
414
430
|
// Exact metric names Spark emits, verified against a real SQLExecutionStart's sparkPlanInfo.
|
|
415
431
|
// The write-side byte metric is "written output", not "size of written files".
|
|
416
432
|
const FILES_READ_COUNT = 'number of files read';
|
|
@@ -519,6 +535,19 @@ function clippedWasteMs(wasteMs , stageId , ctx )
|
|
|
519
535
|
const CACHE_UTILIZATION_VALIDATION =
|
|
520
536
|
"This ratio is a point-in-time storage snapshot from stage-submission events, not a runtime read-count. Confirm against the Spark UI's Storage tab before acting.";
|
|
521
537
|
|
|
538
|
+
// Both cachedRatio and diskRatio are percentages of an RDD's partition/byte count: with few
|
|
539
|
+
// partitions, one partition flipping cached/evicted (or disk-resident/memory-resident) swings the
|
|
540
|
+
// reported percentage by a large amount, so the point estimate is noisy. More partitions average
|
|
541
|
+
// that noise out into a stable ratio. numPartitions is the only sample-size signal
|
|
542
|
+
// DetectorRddInfo carries, so it drives confidence for both variants rather than a flat guess.
|
|
543
|
+
// 10/50 mirror this file's other "trust the sample" cutoffs (skew's minTasksForP95: 20,
|
|
544
|
+
// coreLocality's minTasks: 50).
|
|
545
|
+
function cacheSampleConfidence(numPartitions ) {
|
|
546
|
+
if (numPartitions < 10) return 'low';
|
|
547
|
+
if (numPartitions >= 50) return 'high';
|
|
548
|
+
return 'medium';
|
|
549
|
+
}
|
|
550
|
+
|
|
522
551
|
function partialCacheFinding(rdd , cachedRatio , impactBand ) {
|
|
523
552
|
const rddName = rdd.name || `RDD ${rdd.id}`;
|
|
524
553
|
const cachedPct = Math.round(cachedRatio * 100);
|
|
@@ -527,7 +556,7 @@ function partialCacheFinding(rdd , cachedRatio , impactBa
|
|
|
527
556
|
type: 'cacheUtilization', variant: 'partialCache', stageId: null,
|
|
528
557
|
rddId: rdd.id, rddName,
|
|
529
558
|
impactBand, metric: 'cachedRatio', value: cachedPct,
|
|
530
|
-
confidence:
|
|
559
|
+
confidence: cacheSampleConfidence(rdd.numPartitions),
|
|
531
560
|
validationRequired: CACHE_UTILIZATION_VALIDATION,
|
|
532
561
|
memorySize: rdd.memorySize, diskSize: rdd.diskSize,
|
|
533
562
|
numCachedPartitions: rdd.numCachedPartitions, numPartitions: rdd.numPartitions,
|
|
@@ -542,7 +571,7 @@ function diskSpilloverFinding(rdd , diskRatio , impactBan
|
|
|
542
571
|
type: 'cacheUtilization', variant: 'diskSpillover', stageId: null,
|
|
543
572
|
rddId: rdd.id, rddName,
|
|
544
573
|
impactBand, metric: 'diskRatio', value: diskPct,
|
|
545
|
-
confidence:
|
|
574
|
+
confidence: cacheSampleConfidence(rdd.numPartitions),
|
|
546
575
|
validationRequired: CACHE_UTILIZATION_VALIDATION,
|
|
547
576
|
memorySize: rdd.memorySize, diskSize: rdd.diskSize,
|
|
548
577
|
numCachedPartitions: rdd.numCachedPartitions, numPartitions: rdd.numPartitions,
|
|
@@ -584,6 +613,103 @@ function diskSpilloverFinding(rdd , diskRatio , impactBan
|
|
|
584
613
|
export const STRAGGLER_FLOOR_PCT_WARN = 0.005;
|
|
585
614
|
export const STRAGGLER_FLOOR_PCT_CRIT = 0.02;
|
|
586
615
|
|
|
616
|
+
// A ratio just past ratioWarn is the case most likely to be ordinary task-duration variance
|
|
617
|
+
// rather than real skew; a ratio many multiples past it (a 50x P95/median vs. a 3.1x one) is
|
|
618
|
+
// unambiguous. 1.5x/5x mirror this file's other "just past the floor vs. clearly past it" splits
|
|
619
|
+
// (duplicateSubtreeConfidence's 2x, cacheSampleConfidence's 10/50-partition cutoffs).
|
|
620
|
+
function skewConfidence(ratio , ratioWarn ) {
|
|
621
|
+
if (ratio <= ratioWarn * 1.5) return 'low';
|
|
622
|
+
if (ratio >= ratioWarn * 5) return 'high';
|
|
623
|
+
return 'medium';
|
|
624
|
+
}
|
|
625
|
+
|
|
626
|
+
// warnPct100/lowInfoPct100 are this detector's own two thresholds; scale confidence as a multiple
|
|
627
|
+
// of whichever one gates the branch that fired, the same way skewConfidence scales off ratioWarn.
|
|
628
|
+
// High-GC: a pct just past warnPct100 (10%) is likely normal variance, 3x past it is unambiguous.
|
|
629
|
+
// Low-GC ("cost", over-provisioned): a pct just under lowInfoPct100 (5%) is borderline, a pct near
|
|
630
|
+
// zero is unambiguous idle GC.
|
|
631
|
+
function gcConfidence(
|
|
632
|
+
pct ,
|
|
633
|
+
thresholds ,
|
|
634
|
+
direction ,
|
|
635
|
+
) {
|
|
636
|
+
if (direction === 'high') {
|
|
637
|
+
if (pct <= thresholds.warnPct100 * 1.5) return 'low';
|
|
638
|
+
if (pct >= thresholds.warnPct100 * 3) return 'high';
|
|
639
|
+
return 'medium';
|
|
640
|
+
}
|
|
641
|
+
if (pct >= thresholds.lowInfoPct100 * 0.66) return 'low';
|
|
642
|
+
if (pct <= thresholds.lowInfoPct100 * 0.2) return 'high';
|
|
643
|
+
return 'medium';
|
|
644
|
+
}
|
|
645
|
+
|
|
646
|
+
// warnFloor gates the finding, so a value just past it is the weakest evidence this detector can
|
|
647
|
+
// produce. highFloor is critPct: straggler's own second (speculative-share) tier, 2x warnPct
|
|
648
|
+
// (0.10 -> 0.20); reused as the high-confidence bar for the stragglerShare path too since that
|
|
649
|
+
// metric has no dedicated critical tier of its own (see the "no dedicated critical tier" comment
|
|
650
|
+
// on the straggler detector) but is the same 0-1 task-share magnitude.
|
|
651
|
+
function stragglerConfidence(shareValue , warnFloor , highFloor ) {
|
|
652
|
+
if (shareValue < warnFloor * 1.5) return 'low';
|
|
653
|
+
if (shareValue >= highFloor) return 'high';
|
|
654
|
+
return 'medium';
|
|
655
|
+
}
|
|
656
|
+
|
|
657
|
+
// minWasteMs is speculationWaste's own floor; a run that barely clears it (under 1.5x) is the
|
|
658
|
+
// weakest case, one that clears it several times over (4x, i.e. 4 minutes against a 1-minute
|
|
659
|
+
// floor) is unambiguous.
|
|
660
|
+
function speculationWasteConfidence(wastedMs , minWasteMs ) {
|
|
661
|
+
if (wastedMs <= minWasteMs * 1.5) return 'low';
|
|
662
|
+
if (wastedMs >= minWasteMs * 4) return 'high';
|
|
663
|
+
return 'medium';
|
|
664
|
+
}
|
|
665
|
+
|
|
666
|
+
// The finding already gates on wastedMBSeconds > wasteBufferMultiplier * usedMBSeconds, i.e. a
|
|
667
|
+
// ratio of 1 at the floor; scale confidence off that same ratio the way skewConfidence scales off
|
|
668
|
+
// ratioWarn, instead of introducing a second, unrelated multiplier.
|
|
669
|
+
function memoryWasteConfidence(wastedMBSeconds , usedMBSeconds , wasteBufferMultiplier ) {
|
|
670
|
+
const ratio = usedMBSeconds > 0 ? wastedMBSeconds / (wasteBufferMultiplier * usedMBSeconds) : Infinity;
|
|
671
|
+
if (ratio <= 1.5) return 'low';
|
|
672
|
+
if (ratio >= 3) return 'high';
|
|
673
|
+
return 'medium';
|
|
674
|
+
}
|
|
675
|
+
|
|
676
|
+
// Two independent weak spots can each undercut this finding: a ratio just past warnRatio (could
|
|
677
|
+
// be one bad stage), or too few sampled tasks (mirrors cacheSampleConfidence's use of
|
|
678
|
+
// numPartitions as a sample-size signal). Report whichever signal is weaker rather than
|
|
679
|
+
// averaging them away. critRatio is this detector's own existing second tier, reused directly as
|
|
680
|
+
// the ratio high-bar; minTasks*2/*4 mirror the same "2x floor is still weak, 4x is strong" spread
|
|
681
|
+
// used elsewhere in this file.
|
|
682
|
+
function coreLocalityConfidence(
|
|
683
|
+
ratio ,
|
|
684
|
+
totalTasks ,
|
|
685
|
+
thresholds ,
|
|
686
|
+
) {
|
|
687
|
+
const rank = { low: 0, medium: 1, high: 2 } ;
|
|
688
|
+
const ratioTier = ratio < thresholds.warnRatio * 1.5 ? 'low' : ratio >= thresholds.critRatio ? 'high' : 'medium';
|
|
689
|
+
const sampleTier = totalTasks < thresholds.minTasks * 2 ? 'low' : totalTasks >= thresholds.minTasks * 4 ? 'high' : 'medium';
|
|
690
|
+
return rank[ratioTier] <= rank[sampleTier] ? ratioTier : sampleTier;
|
|
691
|
+
}
|
|
692
|
+
|
|
693
|
+
// warningPct/criticalPct are this detector's own two tiers (0.30/0.60); a churn rate just past
|
|
694
|
+
// warningPct is the borderline call the impact-band split already treats as the weaker tier, so
|
|
695
|
+
// reuse criticalPct directly as the high-confidence bar instead of inventing a third figure.
|
|
696
|
+
function autoscalingChurnConfidence(shortLivedPct , warningPct , criticalPct ) {
|
|
697
|
+
if (shortLivedPct <= warningPct * 1.5) return 'low';
|
|
698
|
+
if (shortLivedPct >= criticalPct) return 'high';
|
|
699
|
+
return 'medium';
|
|
700
|
+
}
|
|
701
|
+
|
|
702
|
+
// minExecutions is the bare minimum occurrence count this detector will even emit a finding for;
|
|
703
|
+
// a match at exactly that count is the weakest reuse signal (as likely to be coincidental overlap
|
|
704
|
+
// as real shared work), while 3x the floor is several independent executions all hitting the same
|
|
705
|
+
// relation/composite shape, unambiguous. Mirrors duplicateSubtreeConfidence's occurrences handling
|
|
706
|
+
// for the same reason: repetition count is the strength signal for a structural-match detector.
|
|
707
|
+
function cachingReuseConfidence(occurrences , minExecutions ) {
|
|
708
|
+
if (occurrences <= minExecutions) return 'low';
|
|
709
|
+
if (occurrences >= minExecutions * 3) return 'high';
|
|
710
|
+
return 'medium';
|
|
711
|
+
}
|
|
712
|
+
|
|
587
713
|
export const DETECTORS = [
|
|
588
714
|
{
|
|
589
715
|
type: 'skew', scope: 'stage', order: 30, fixEffort: 'code', version: 1,
|
|
@@ -611,8 +737,8 @@ export const DETECTORS = [
|
|
|
611
737
|
type: 'skew', stageId: stage.id,
|
|
612
738
|
impactBand: 'warning',
|
|
613
739
|
metric, value,
|
|
614
|
-
confidence:
|
|
615
|
-
validationRequired: '
|
|
740
|
+
confidence: skewConfidence(ratio, this.thresholds.ratioWarn),
|
|
741
|
+
validationRequired: 'This finding is gated by a 0.5% runtime-floor threshold, our own noise floor for this metric.',
|
|
616
742
|
recommendation: `Task duration ratio (${metric}) is ${value}×: for join-driven skew, enable AQE skew-join handling (spark.sql.adaptive.skewJoin.enabled); otherwise salt the key or repartition on a better key to reduce task skew.`,
|
|
617
743
|
};
|
|
618
744
|
},
|
|
@@ -766,8 +892,7 @@ export const DETECTORS = [
|
|
|
766
892
|
{
|
|
767
893
|
type: 'gc', scope: 'stage', order: 50, fixEffort: 'config', version: 1,
|
|
768
894
|
docAnchor: '#bottleneck-gc',
|
|
769
|
-
|
|
770
|
-
validationRequired: 'The 10-second minimum-runtime floor that gates this finding is our own noise floor, unvalidated: no external tool publishes an equivalent metric to calibrate against. Confirm against known-good/known-bad real logs before trusting the impact-band split.',
|
|
895
|
+
validationRequired: 'This finding is gated by a 10-second minimum-runtime floor, our own noise floor for this metric.',
|
|
771
896
|
thresholds: {
|
|
772
897
|
warnPct100: 10,
|
|
773
898
|
// Descending tier: ExecutorGcHeuristic, ported as-is.
|
|
@@ -778,7 +903,7 @@ export const DETECTORS = [
|
|
|
778
903
|
detect(
|
|
779
904
|
|
|
780
905
|
|
|
781
|
-
|
|
906
|
+
|
|
782
907
|
|
|
783
908
|
stage ,
|
|
784
909
|
) {
|
|
@@ -790,7 +915,7 @@ export const DETECTORS = [
|
|
|
790
915
|
type: 'gc', stageId: stage.id,
|
|
791
916
|
impactBand: 'warning',
|
|
792
917
|
metric: 'gcPct', value,
|
|
793
|
-
confidence: this.
|
|
918
|
+
confidence: gcConfidence(pct, this.thresholds, 'high'), validationRequired: this.validationRequired,
|
|
794
919
|
recommendation: `GC consumed ${value}% of executor run time: reduce object creation, use primitive types, avoid UDFs, increase executor memory.`,
|
|
795
920
|
};
|
|
796
921
|
}
|
|
@@ -802,7 +927,7 @@ export const DETECTORS = [
|
|
|
802
927
|
type: 'gc', stageId: stage.id, direction: 'low',
|
|
803
928
|
impactBand: 'info',
|
|
804
929
|
metric: 'gcPct', value,
|
|
805
|
-
confidence: this.
|
|
930
|
+
confidence: gcConfidence(pct, this.thresholds, 'low'), validationRequired: this.validationRequired,
|
|
806
931
|
recommendation: `GC consumed only ${value}% of executor run time: memory may be over-provisioned; consider reducing spark.executor.memory for cost savings.`,
|
|
807
932
|
};
|
|
808
933
|
}
|
|
@@ -1028,20 +1153,21 @@ export const DETECTORS = [
|
|
|
1028
1153
|
unit: useSpeculativeMetric ? 'count' : 'pct',
|
|
1029
1154
|
speculativeTasks: stage.speculativeTasks ?? 0,
|
|
1030
1155
|
stragglerCount: stage.stragglerCount ?? 0,
|
|
1031
|
-
confidence:
|
|
1032
|
-
|
|
1156
|
+
confidence: useSpeculativeMetric
|
|
1157
|
+
? stragglerConfidence(speculativeShare, this.thresholds.warnPct, this.thresholds.critPct)
|
|
1158
|
+
: stragglerConfidence(stragglerShare, this.thresholds.shareWarn, this.thresholds.critPct),
|
|
1159
|
+
validationRequired: 'This finding is gated by 0.5%/2% runtime-floor thresholds, our own noise floor for this metric.',
|
|
1033
1160
|
recommendation: `${detail}: rule out a GC pause or a slow shuffle fetch before assuming a hardware issue; if a skewed key is the real cause, that's a candidate for AQE's skew-join handling.`,
|
|
1034
1161
|
};
|
|
1035
1162
|
},
|
|
1036
1163
|
},
|
|
1037
1164
|
{
|
|
1038
1165
|
type: 'speculationWaste', scope: 'stage', order: 71, fixEffort: 'config', version: 1,
|
|
1039
|
-
docAnchor: '#bottleneck-straggler',
|
|
1166
|
+
docAnchor: '#bottleneck-straggler',
|
|
1040
1167
|
thresholds: { minWasted: 5, minWasteMs: 60000 },
|
|
1041
1168
|
detect(
|
|
1042
1169
|
|
|
1043
1170
|
|
|
1044
|
-
|
|
1045
1171
|
|
|
1046
1172
|
stage ,
|
|
1047
1173
|
) {
|
|
@@ -1052,7 +1178,7 @@ export const DETECTORS = [
|
|
|
1052
1178
|
type: 'speculationWaste', stageId: stage.id,
|
|
1053
1179
|
impactBand: 'warning',
|
|
1054
1180
|
metric: 'speculationWasteMs', value: wastedMs,
|
|
1055
|
-
confidence: this.
|
|
1181
|
+
confidence: speculationWasteConfidence(wastedMs, this.thresholds.minWasteMs),
|
|
1056
1182
|
recommendation: `Speculative execution discarded ${Math.round(wastedMs / 1000)}s of executor time in this stage; if task durations are naturally variable rather than genuine stragglers, consider tuning spark.speculation.multiplier/quantile.`,
|
|
1057
1183
|
};
|
|
1058
1184
|
},
|
|
@@ -1298,8 +1424,8 @@ export const DETECTORS = [
|
|
|
1298
1424
|
out.push({
|
|
1299
1425
|
type: 'memoryUtilization', variant: 'wasteModel', stageId: null,
|
|
1300
1426
|
impactBand: 'info', metric: 'wastedMBSeconds', value,
|
|
1301
|
-
confidence:
|
|
1302
|
-
validationRequired: 'Memory-waste estimate uses allocated-vs-used memory-time and
|
|
1427
|
+
confidence: memoryWasteConfidence(wastedMBSeconds, usedMBSeconds, this.thresholds.wasteBufferMultiplier),
|
|
1428
|
+
validationRequired: 'Memory-waste estimate uses allocated-vs-used memory-time and a 1.5x buffer: confirm against the Spark UI before acting.',
|
|
1303
1429
|
recommendation: `Allocated executor memory sat largely idle over the run (~${value.toLocaleString('en-US')} MB-seconds wasted): review spark.executor.memory and executor count.`,
|
|
1304
1430
|
});
|
|
1305
1431
|
}
|
|
@@ -1376,8 +1502,8 @@ export const DETECTORS = [
|
|
|
1376
1502
|
metric: 'nonLocalRatio', value,
|
|
1377
1503
|
// Raw count behind the ratio, for the impact estimator. Non-null whenever totalTasks is.
|
|
1378
1504
|
nonLocalTaskCount: nonLocalTasks ,
|
|
1379
|
-
confidence:
|
|
1380
|
-
validationRequired: '
|
|
1505
|
+
confidence: coreLocalityConfidence(ratio , totalTasks, this.thresholds),
|
|
1506
|
+
validationRequired: 'This finding is gated by 15%/35% non-local-ratio thresholds (and a 50-task minimum), our own noise floor for this metric.',
|
|
1381
1507
|
recommendation: `${value}% of tasks (${nonLocalTasks }) ran without process- or node-local data placement: check spark.locality.wait settings and executor/data colocation.`,
|
|
1382
1508
|
};
|
|
1383
1509
|
},
|
|
@@ -1387,12 +1513,10 @@ export const DETECTORS = [
|
|
|
1387
1513
|
// re-provisioning, not normal scale-down). Reuses utilization's add/remove matching, but
|
|
1388
1514
|
// measures lifetime against a threshold instead of aggregate active-time.
|
|
1389
1515
|
type: 'autoscalingChurn', scope: 'app', order: 103, fixEffort: 'config', version: 1,
|
|
1390
|
-
confidence: 'low',
|
|
1391
1516
|
thresholds: { shortLivedMs: 120_000, warningPct: 0.30, criticalPct: 0.60, minExecutors: 5 },
|
|
1392
1517
|
detect(
|
|
1393
1518
|
|
|
1394
1519
|
|
|
1395
|
-
|
|
1396
1520
|
|
|
1397
1521
|
ctx ,
|
|
1398
1522
|
) {
|
|
@@ -1421,7 +1545,7 @@ export const DETECTORS = [
|
|
|
1421
1545
|
metric: 'shortLivedExecutorPct', value: pct,
|
|
1422
1546
|
// Raw count behind the percentage, for the impact estimator's startup-overhead figure.
|
|
1423
1547
|
shortLivedExecutorCount: shortLivedCount,
|
|
1424
|
-
confidence: this.
|
|
1548
|
+
confidence: autoscalingChurnConfidence(shortLivedPct, this.thresholds.warningPct, this.thresholds.criticalPct),
|
|
1425
1549
|
recommendation: `${pct}% of executors ran for under 2 minutes before being removed. This looks like wasteful re-provisioning rather than normal scale-down; consider raising spark.dynamicAllocation.executorIdleTimeout or widening the minExecutors/maxExecutors bounds to reduce flapping.`,
|
|
1426
1550
|
};
|
|
1427
1551
|
},
|
|
@@ -1562,7 +1686,7 @@ export const DETECTORS = [
|
|
|
1562
1686
|
metric: 'executionReuse', value,
|
|
1563
1687
|
format: 'derived', relations, operator: agg.operator, relation: relationDisplay,
|
|
1564
1688
|
executionIds: finalExecutionIds, totalReadBytes,
|
|
1565
|
-
confidence:
|
|
1689
|
+
confidence: cachingReuseConfidence(value, this.thresholds.minExecutions),
|
|
1566
1690
|
validationRequired:
|
|
1567
1691
|
'Composite reuse is inferred from a structural plan-shape match (operator + normalized ' +
|
|
1568
1692
|
'join/filter condition + child shapes) across SQL executions; confirm these executions ' +
|
|
@@ -1593,7 +1717,7 @@ export const DETECTORS = [
|
|
|
1593
1717
|
relation: agg.relation, format: agg.format,
|
|
1594
1718
|
executionIds: residualExecutionIds.sort((a, b) => a - b),
|
|
1595
1719
|
totalReadBytes,
|
|
1596
|
-
confidence:
|
|
1720
|
+
confidence: cachingReuseConfidence(value, this.thresholds.minExecutions),
|
|
1597
1721
|
validationRequired:
|
|
1598
1722
|
'Relation-reuse is inferred from the pre-AQE plan scan identity across SQL ' +
|
|
1599
1723
|
'executions; confirm the reads are the same data and cacheable within one ' +
|
|
@@ -1744,7 +1868,8 @@ export const DETECTORS = [
|
|
|
1744
1868
|
// wallClock estimate (the common case). Only surfaces on the rare occupancy-sweep miss.
|
|
1745
1869
|
impactBand: 'warning', metric: 'subtreeOccurrences', value: g.occurrences,
|
|
1746
1870
|
rootName: g.rootName, subtreeSize: g.subtreeSize, sampleRelation: g.sampleRelation,
|
|
1747
|
-
groupIndex: g.groupIndex,
|
|
1871
|
+
groupIndex: g.groupIndex,
|
|
1872
|
+
confidence: duplicateSubtreeConfidence(g.subtreeSize, g.occurrences, this.thresholds),
|
|
1748
1873
|
validationRequired: 'Duplicate-subtree matching compares operator names and metric names only, not literal values or expr IDs: confirm the repeated work is real in the Spark SQL plan tab before acting.',
|
|
1749
1874
|
recommendation: g.isExchangeRoot
|
|
1750
1875
|
? `A ${g.subtreeSize}-node subtree rooted at ${pathBasename(g.rootName)} repeats ${g.occurrences}x in this plan${touching}: this looks like a possible missed exchange reuse; check whether the same shuffle could be computed once and reused.`
|
|
@@ -1,6 +1,7 @@
|
|
|
1
1
|
### `CACHE`: Caching opportunity {#cache}
|
|
2
2
|
|
|
3
3
|
A reusable dataset (re-read via the same SQL relation more than once) may be
|
|
4
|
-
worth persisting between stages. Self-
|
|
5
|
-
|
|
6
|
-
|
|
4
|
+
worth persisting between stages. Self-flags a confidence that scales with
|
|
5
|
+
how many executions reuse the same relation: reuse is only inferred, from
|
|
6
|
+
plan-scan identity across SQL executions, so confirm the reads really do
|
|
7
|
+
hit the same data before you cache anything.
|
|
@@ -3,5 +3,7 @@
|
|
|
3
3
|
Executors are stood up and torn down again before they can do useful work:
|
|
4
4
|
re-provisioning churn rather than normal scale-down. Raise
|
|
5
5
|
`spark.dynamicAllocation.executorIdleTimeout`, or widen the
|
|
6
|
-
`minExecutors`/`maxExecutors` bounds to reduce flapping. Self-
|
|
7
|
-
|
|
6
|
+
`minExecutors`/`maxExecutors` bounds to reduce flapping. Self-flags a
|
|
7
|
+
confidence that scales with how far the short-lived-executor share sits
|
|
8
|
+
past the threshold: these thresholds are still a design spike, not yet
|
|
9
|
+
validated against real-world runs.
|
|
@@ -2,6 +2,5 @@
|
|
|
2
2
|
|
|
3
3
|
Tasks run without process- or node-local data placement more often than
|
|
4
4
|
expected. Check `spark.locality.wait` settings and executor/data colocation.
|
|
5
|
-
Self-
|
|
6
|
-
|
|
7
|
-
calibrate them against.
|
|
5
|
+
Self-flags a confidence that scales with the non-local ratio and sample
|
|
6
|
+
size: the thresholds are our own noise floor for this metric.
|
|
@@ -4,7 +4,7 @@ Executor memory or core capacity may be over- or under-provisioned. Some
|
|
|
4
4
|
detail here needs `spark.eventLog.logStageExecutorMetrics=true` on the run
|
|
5
5
|
being analyzed; without it, per-executor memory usage can't be broken down.
|
|
6
6
|
Review `spark.executor.memory` and executor count if allocated memory sat
|
|
7
|
-
largely idle over the run. That idle-memory variant
|
|
8
|
-
|
|
9
|
-
|
|
7
|
+
largely idle over the run. That idle-memory variant self-flags a confidence
|
|
8
|
+
that scales with how far the estimated waste sits past a 1.5x buffer: it
|
|
9
|
+
estimates waste from allocated-versus-used memory-time. Check it against
|
|
10
10
|
the Spark UI before resizing anything.
|
|
@@ -1,7 +1,8 @@
|
|
|
1
1
|
### `SPEC`: Speculation waste {#spec}
|
|
2
2
|
|
|
3
3
|
Speculative task attempts used a lot of executor time without confirming a
|
|
4
|
-
genuine straggler. Self-
|
|
5
|
-
|
|
6
|
-
|
|
4
|
+
genuine straggler. Self-flags a confidence that scales with how far the
|
|
5
|
+
wasted time sits past the threshold: these thresholds are still a design
|
|
6
|
+
spike, not yet validated against real-world runs. If task durations are
|
|
7
|
+
just naturally variable rather than genuine stragglers, tune
|
|
7
8
|
`spark.speculation.multiplier`/`spark.speculation.quantile`.
|
|
@@ -269,7 +269,7 @@ function renderEvidenceValue(key , value ) {
|
|
|
269
269
|
|
|
270
270
|
function formatWallClockRange(low , high ) {
|
|
271
271
|
const fmtMs = (ms ) => (ms === 0 ? '0s' : formatDuration(ms));
|
|
272
|
-
return low === high ? `
|
|
272
|
+
return low === high ? `Estimated ${fmtMs(high)}` : `Estimated ${fmtMs(low)}-${fmtMs(high)}`;
|
|
273
273
|
}
|
|
274
274
|
|
|
275
275
|
function formatRawWaste(rawWaste ) {
|
|
@@ -147,7 +147,7 @@ export async function runParse(
|
|
|
147
147
|
}
|
|
148
148
|
|
|
149
149
|
if (!state.app) {
|
|
150
|
-
emit({ type: 'error', message: 'Not a Spark event log:
|
|
150
|
+
emit({ type: 'error', message: 'Not a Spark event log: no application-start event found. Choose a Spark event log file, or check the docs for supported formats.' });
|
|
151
151
|
return;
|
|
152
152
|
}
|
|
153
153
|
|
|
@@ -203,7 +203,7 @@ export async function runParseFiles(
|
|
|
203
203
|
}
|
|
204
204
|
|
|
205
205
|
if (!state.app) {
|
|
206
|
-
emit({ type: 'error', message: 'Not a Spark event log:
|
|
206
|
+
emit({ type: 'error', message: 'Not a Spark event log: no application-start event found. Choose a Spark event log file, or check the docs for supported formats.' });
|
|
207
207
|
return;
|
|
208
208
|
}
|
|
209
209
|
|
|
@@ -395,6 +395,14 @@ export function buildComparison(
|
|
|
395
395
|
);
|
|
396
396
|
}
|
|
397
397
|
|
|
398
|
+
// matchStages' coverage is a Dice coefficient: (2 * pairs.length) / (baseCount
|
|
399
|
+
// + candCount). Below 0.5, more than half of each run's stages went unpaired,
|
|
400
|
+
// so the stage-level rows (stageSkew, baseStages/candStages) mostly show
|
|
401
|
+
// unrelated work side by side rather than the same stage before/after -- the
|
|
402
|
+
// comparison is dominated by guesswork, not genuine pairing. 0.5 is thus the
|
|
403
|
+
// natural midpoint for "more matched than not," not an arbitrary tuning knob.
|
|
404
|
+
const LOW_COVERAGE_THRESHOLD = 0.5;
|
|
405
|
+
|
|
398
406
|
export function compareRuns(
|
|
399
407
|
baseline ,
|
|
400
408
|
candidate ,
|
|
@@ -405,10 +413,21 @@ export function compareRuns(
|
|
|
405
413
|
// is the normal way to label an A/B experiment, so a mismatch must not block
|
|
406
414
|
// the (matching-free, name-independent) deltas. Surface it as `low` instead.
|
|
407
415
|
const namesDiffer = namesConflict(baseSnap.app, candSnap.app);
|
|
416
|
+
// Coverage is the other half of the signal: identical names on two runs that
|
|
417
|
+
// barely share any stages are just as misleading as differing names on two
|
|
418
|
+
// runs that match well, so either condition alone drops confidence to `low`.
|
|
419
|
+
const lowCoverage = match.coverage < LOW_COVERAGE_THRESHOLD;
|
|
420
|
+
const reason = namesDiffer && lowCoverage
|
|
421
|
+
? `Run names differ and only ${(match.coverage * 100).toFixed(0)}% of stages matched, so deltas may compare different work.`
|
|
422
|
+
: namesDiffer
|
|
423
|
+
? 'Run names differ, so deltas may compare different work.'
|
|
424
|
+
: lowCoverage
|
|
425
|
+
? `Only ${(match.coverage * 100).toFixed(0)}% of stages matched between runs, so per-stage rows mostly compare unrelated work.`
|
|
426
|
+
: null;
|
|
408
427
|
return {
|
|
409
428
|
baselineLabel: baseline.label, candidateLabel: candidate.label,
|
|
410
|
-
confidence: namesDiffer ? 'low' : 'ok',
|
|
411
|
-
reason
|
|
429
|
+
confidence: namesDiffer || lowCoverage ? 'low' : 'ok',
|
|
430
|
+
reason,
|
|
412
431
|
matchedCoverage: match.coverage,
|
|
413
432
|
metrics: metricDeltas(baseSnap, candSnap),
|
|
414
433
|
findings: findingsDelta(baseSnap, candSnap),
|