sparkforensics-mcp 0.4.0 → 0.5.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (64) hide show
  1. package/package.json +1 -1
  2. package/vendor-core/allocation.js +106 -0
  3. package/vendor-core/analyzer.js +13 -13
  4. package/vendor-core/cli/budgets.js +23 -9
  5. package/vendor-core/cli/collect-run.js +11 -4
  6. package/vendor-core/cli/regression-budgets.js +83 -0
  7. package/vendor-core/comparison-verdict.js +22 -22
  8. package/vendor-core/core-source-hash.txt +1 -1
  9. package/vendor-core/detectors.js +202 -68
  10. package/vendor-core/docs-content/detection/cache.md +3 -2
  11. package/vendor-core/docs-content/detection/cfg.md +9 -8
  12. package/vendor-core/docs-content/detection/chrn.md +1 -2
  13. package/vendor-core/docs-content/detection/cold.md +4 -2
  14. package/vendor-core/docs-content/detection/fail.md +3 -2
  15. package/vendor-core/docs-content/detection/gc.md +3 -2
  16. package/vendor-core/docs-content/detection/host.md +2 -1
  17. package/vendor-core/docs-content/detection/local.md +1 -1
  18. package/vendor-core/docs-content/detection/mem.md +5 -2
  19. package/vendor-core/docs-content/detection/plan.md +2 -1
  20. package/vendor-core/docs-content/detection/sfail.md +2 -1
  21. package/vendor-core/docs-content/detection/shape.md +5 -4
  22. package/vendor-core/docs-content/detection/skew.md +3 -1
  23. package/vendor-core/docs-content/detection/slow.md +2 -2
  24. package/vendor-core/docs-content/detection/spec.md +2 -3
  25. package/vendor-core/docs-content/detection/spill.md +1 -1
  26. package/vendor-core/docs-site-config.js +1 -1
  27. package/vendor-core/effective-conf.js +107 -0
  28. package/vendor-core/efficiency-model.js +8 -6
  29. package/vendor-core/event-handlers.js +160 -47
  30. package/vendor-core/event-schemas.js +2 -0
  31. package/vendor-core/evidence-report.js +18 -10
  32. package/vendor-core/finding-generic-recommendation.js +20 -1
  33. package/vendor-core/finding-names.js +7 -0
  34. package/vendor-core/finding-presentation.js +61 -26
  35. package/vendor-core/finding-tag-help.js +1 -1
  36. package/vendor-core/finding-types.js +12 -0
  37. package/vendor-core/format-utils.js +4 -3
  38. package/vendor-core/impact-estimator.js +20 -2
  39. package/vendor-core/impact-format.js +14 -13
  40. package/vendor-core/impact-model.js +27 -5
  41. package/vendor-core/ingest.js +4 -2
  42. package/vendor-core/list-runs.js +5 -2
  43. package/vendor-core/mcp-tools.js +1 -1
  44. package/vendor-core/model-assembler.js +23 -1
  45. package/vendor-core/parser-worker.js +1 -1
  46. package/vendor-core/proxy.js +3 -1
  47. package/vendor-core/python-stage.js +25 -0
  48. package/vendor-core/recommendation-rollup.js +16 -9
  49. package/vendor-core/redact.js +51 -10
  50. package/vendor-core/remediation.js +20 -0
  51. package/vendor-core/run-comparison.js +43 -24
  52. package/vendor-core/run-interpretation.js +2 -1
  53. package/vendor-core/run-metrics.js +198 -0
  54. package/vendor-core/run-totals.js +24 -0
  55. package/vendor-core/run-verdict.js +3 -4
  56. package/vendor-core/scorecard-estimates.js +1 -0
  57. package/vendor-core/session-snapshot.js +7 -0
  58. package/vendor-core/shs-schemas.js +2 -2
  59. package/vendor-core/spark-memory.js +17 -0
  60. package/vendor-core/stage-plan-nodes.js +18 -0
  61. package/vendor-core/stage-quantiles.js +4 -0
  62. package/vendor-core/types.js +49 -1
  63. package/vendor-core/wasted-core-hours.js +10 -7
  64. package/vendor-core/write-targets.js +312 -0
@@ -1,4 +1,5 @@
1
- import { pathBasename, formatBytes, nsToMs, IMPACT_BAND_ORDER } from './format-utils.js';
1
+ import { pathBasename, formatBytes, IMPACT_BAND_ORDER } from './format-utils.js';
2
+ import { shareLabel } from './finding-presentation.js';
2
3
  import { scanRelationId } from './plan-summary.js';
3
4
  import { computePeakConcurrentCores, computePeakConcurrentExecutorCount } from './core-count.js';
4
5
  import { walkPlanTree } from './plan-tree-walk.js';
@@ -13,11 +14,14 @@ import {
13
14
 
14
15
  } from './impact-model.js';
15
16
  import { isExchangeNode, isBroadcastExchangeNode } from './plan-node-detail.js';
17
+ import { totalExecutorCpuMs } from './run-totals.js';
18
+ import { DUPLICATE_SUBTREE_DIFFERING_NOTE, duplicateSubtreeDetail, SLOW_HOST_DIMENSION_LABEL } from './finding-generic-recommendation.js';
16
19
  import { stageIdsForSqlExec } from './sql-stages.js';
17
20
  import { cyrb53 } from './string-hash.js';
21
+ import { decreaseConf, increaseConf, setConf } from './remediation.js';
18
22
  import { MAX_FAILURE_GROUPS, describeTaskFailure, } from './task-failure.js';
19
23
 
20
-
24
+
21
25
 
22
26
  const MB = 1024 * 1024;
23
27
  const GB = 1024 * MB;
@@ -637,10 +641,12 @@ function stragglerTailClaim(stage ) {
637
641
  return tailClaim(stage, Math.max(0, (stage.taskDurationMax ?? 0) - longestTaskAfterFixMs), longestTaskAfterFixMs);
638
642
  }
639
643
 
640
- // A tail claim shortens the stage's longest task, hence TAIL_CLAIM (see occupancy.ts).
644
+ // A tail claim shortens the stage's longest task, hence TAIL_CLAIM (see occupancy.ts). Its core
645
+ // time is the measured task time the fix removes (removedCoreWorkMs).
641
646
  function tailClaimImpact(claim , stageId , ctx ) {
642
- return singleStageImpact(claim.wasteMs, stageId, ctx, 'measured', { value: claim.wasteMs, unit: 'ms' },
647
+ const estimate = singleStageImpact(claim.wasteMs, stageId, ctx, 'measured', { value: claim.wasteMs, unit: 'ms' },
643
648
  { ...TAIL_CLAIM, removedCoreWorkMs: claim.removedCoreWorkMs, longestTaskAfterFixMs: claim.longestTaskAfterFixMs });
649
+ return { ...estimate, coreTimeMs: { low: claim.removedCoreWorkMs, high: claim.removedCoreWorkMs } };
644
650
  }
645
651
 
646
652
  // The figure a runtime floor checks: the claim's recoverable wall-clock, not a delta a physical
@@ -649,6 +655,76 @@ function tailClaimFloorMs(claim , stageId , ctx )
649
655
  return tailClaimImpact(claim, stageId, ctx.impact).wallClock?.high ?? claim.wasteMs;
650
656
  }
651
657
 
658
+ // A setting is only a fix when the run's logged conf doesn't already have it: a run that set it
659
+ // needs another remedy. Only explicitly logged properties count; Spark's unlogged version
660
+ // defaults are not modeled. Booleans compare case-insensitively, as Spark parses them.
661
+ function loggedAs(app , key , suggested ) {
662
+ const logged = app?.config?.[key]?.trim();
663
+ return typeof suggested === 'boolean'
664
+ ? logged?.toLowerCase() === String(suggested)
665
+ : logged === suggested;
666
+ }
667
+
668
+ function setConfUnlessLogged(app , key , suggested ) {
669
+ return loggedAs(app, key, suggested) ? [] : [setConf(key, suggested)];
670
+ }
671
+
672
+ // A switch's fix worded for the run's logged conf: `recommend` names the property while the run
673
+ // doesn't have it, `alreadyOn` points at the remedy left once it does, with no remediation.
674
+ function switchFix(on , key , suggested , recommend , alreadyOn ) {
675
+ return on ? { text: alreadyOn, remediation: [] } : { text: recommend, remediation: [setConf(key, suggested)] };
676
+ }
677
+
678
+ function skewJoinFix(app ) {
679
+ const key = 'spark.sql.adaptive.skewJoin.enabled';
680
+ const remedy = 'salt the key or repartition on a better key';
681
+ if (loggedAs(app, 'spark.sql.adaptive.enabled', false)) {
682
+ return {
683
+ text: `AQE is off, so enable it (spark.sql.adaptive.enabled) for skew-join handling to apply; otherwise ${remedy}`,
684
+ remediation: [setConf('spark.sql.adaptive.enabled', true), ...setConfUnlessLogged(app, key, true)],
685
+ };
686
+ }
687
+ return switchFix(loggedAs(app, key, true), key, true,
688
+ `for join-driven skew, enable AQE skew-join handling (${key}); otherwise ${remedy}`,
689
+ `AQE skew-join handling is already on, so ${remedy}`);
690
+ }
691
+
692
+ // The resources flag is read from the same property as the logged conf.
693
+ function dynamicAllocationFix(app , recommend , alreadyOn ) {
694
+ const key = 'spark.dynamicAllocation.enabled';
695
+ return switchFix(app.resources?.dynamicAllocationEnabled === true || loggedAs(app, key, true), key, true, recommend, alreadyOn);
696
+ }
697
+
698
+ // The run's logged spark.sql.shuffle.partitions as a count, or null when unlogged or not a count.
699
+ function loggedShufflePartitions(app ) {
700
+ const logged = app?.config?.['spark.sql.shuffle.partitions']?.trim();
701
+ return logged != null && /^\d+$/.test(logged) ? Number(logged) : null;
702
+ }
703
+
704
+ // lowShuffleParallelism's fix. The partition count that brings each shuffle partition down to the
705
+ // ideal size, as the estimate models it, is per stage; the property is job-wide. A logged value at
706
+ // or above it means the property is not what limits that stage (a repartition(n) or an RDD
707
+ // shuffle is), so the text points at the stage's own partitioning and no property is suggested.
708
+ // Unlogged, the count is not a safe value: null.
709
+ function lowShuffleParallelismFix(app , needed ) {
710
+ const logged = loggedShufflePartitions(app);
711
+ if (logged != null && logged >= needed) {
712
+ return {
713
+ text: `spark.sql.shuffle.partitions is already ${logged}, so raise this stage's own partition count (its repartition(n) or RDD parallelism) so each partition is smaller`,
714
+ remediation: [],
715
+ };
716
+ }
717
+ return {
718
+ text: 'raise spark.sql.shuffle.partitions so each partition is smaller',
719
+ remediation: [increaseConf('spark.sql.shuffle.partitions', logged == null ? null : needed)],
720
+ };
721
+ }
722
+
723
+ // No dynamic-allocation property has an effect on a run whose logged conf turns it off.
724
+ function dynamicAllocationOff(app ) {
725
+ return app?.resources?.dynamicAllocationEnabled === false || app?.config?.['spark.dynamicAllocation.enabled']?.trim().toLowerCase() === 'false';
726
+ }
727
+
652
728
  // Shared by cacheUtilization's two variants, worded per storage source: neither is a runtime
653
729
  // block-access read-count.
654
730
  const CACHE_UTILIZATION_VALIDATION = {
@@ -669,46 +745,53 @@ function cacheSampleConfidence(numPartitions )
669
745
  return 'medium';
670
746
  }
671
747
 
748
+ /** A sentence names a cached RDD by its first 40 characters, since RDD names are often a whole
749
+ * plan string (the Cache Storage card shows it in full); an unnamed RDD reads "RDD <id>". */
750
+ function rddLabel({ id, name } ) {
751
+ return `RDD ${!name ? id : name.length > 40 ? `${name.slice(0, 40)}...` : name}`;
752
+ }
753
+
672
754
  function partialCacheFinding(rdd , cachedRatio , impactBand ) {
673
- const rddName = rdd.name || `RDD ${rdd.id}`;
674
755
  const cachedPct = Math.round(cachedRatio * 100);
675
756
  const evictedPct = 100 - cachedPct;
676
757
  return {
677
758
  type: 'cacheUtilization', variant: 'partialCache', stageId: null,
678
- rddId: rdd.id, rddName,
759
+ rddId: rdd.id, rddName: rdd.name || `RDD ${rdd.id}`,
679
760
  impactBand, metric: 'cachedRatio', value: cachedPct,
680
761
  confidence: cacheSampleConfidence(rdd.numPartitions),
681
762
  validationRequired: CACHE_UTILIZATION_VALIDATION[rdd.storageSource ?? 'rddInfo'],
682
763
  memorySize: rdd.memorySize, diskSize: rdd.diskSize,
683
764
  numCachedPartitions: rdd.numCachedPartitions, numPartitions: rdd.numPartitions,
684
- recommendation: `RDD ${rddName} is ${evictedPct}% evicted from cache (${cachedPct}% of partitions cached). Increase executor memory or reduce the cached dataset size.`,
765
+ recommendation: `${rddLabel(rdd)} is ${evictedPct}% evicted from cache (${cachedPct}% of partitions cached): increase executor memory or reduce the cached dataset size.`,
766
+ remediation: [increaseConf('spark.executor.memory')],
685
767
  };
686
768
  }
687
769
 
688
770
  function diskSpilloverFinding(rdd , diskRatio , impactBand ) {
689
- const rddName = rdd.name || `RDD ${rdd.id}`;
690
771
  const diskPct = Math.round(diskRatio * 100);
691
772
  return {
692
773
  type: 'cacheUtilization', variant: 'diskSpillover', stageId: null,
693
- rddId: rdd.id, rddName,
774
+ rddId: rdd.id, rddName: rdd.name || `RDD ${rdd.id}`,
694
775
  impactBand, metric: 'diskRatio', value: diskPct,
695
776
  confidence: cacheSampleConfidence(rdd.numPartitions),
696
777
  validationRequired: CACHE_UTILIZATION_VALIDATION[rdd.storageSource ?? 'rddInfo'],
697
778
  memorySize: rdd.memorySize, diskSize: rdd.diskSize,
698
779
  numCachedPartitions: rdd.numCachedPartitions, numPartitions: rdd.numPartitions,
699
- recommendation: `RDD ${rddName} is ${diskPct}% spilled to disk despite requesting MEMORY_AND_DISK. Executor memory may be too small for this cached dataset.`,
780
+ recommendation: `${rddLabel(rdd)} is ${diskPct}% spilled to disk despite requesting MEMORY_AND_DISK: executor memory may be too small for it.`,
781
+ remediation: [increaseConf('spark.executor.memory')],
700
782
  };
701
783
  }
702
784
 
703
785
  // Persisted RDDs with no storage evidence at all: no block updates in the log, and RDD Info's
704
786
  // sizes are the 0 that Spark 2.3+ always writes. Reports the gap instead of a clean result, the
705
787
  // same missing-evidence shape as memoryUtilization's dataUnavailable caveat.
706
- function storageUnobservedFinding(persistedRddCount ) {
788
+ function storageUnobservedFinding(persistedRddCount , app ) {
707
789
  const rdds = persistedRddCount === 1 ? '1 persisted RDD has' : `${persistedRddCount} persisted RDDs have`;
708
790
  return {
709
791
  type: 'cacheUtilization', variant: 'storageUnobserved', stageId: null,
710
792
  impactBand: 'info', metric: 'persistedRdds', value: persistedRddCount, dataUnavailable: true,
711
793
  recommendation: `${rdds} no cache-storage evidence in this log, so eviction and disk spillover can't be checked: Spark 2.3+ records cached sizes only as block updates, which need spark.eventLog.logBlockUpdates.enabled=true.`,
794
+ remediation: setConfUnlessLogged(app, 'spark.eventLog.logBlockUpdates.enabled', true),
712
795
  };
713
796
  }
714
797
 
@@ -955,19 +1038,13 @@ function cachingReuseConfidence(occurrences , minExecutions )
955
1038
  return 'medium';
956
1039
  }
957
1040
 
958
- // A share threshold as caveat text states it: 0.005 -> "0.5%". Rounded to 4 decimals of a percent
959
- // so float noise (0.07 * 100) never prints.
960
- function shareLabel(share ) {
961
- return `${Math.round(share * 1e6) / 1e4}%`;
962
- }
963
-
964
1041
  // Caveats that name a threshold read it from the thresholds the detector ran with, so a tuned run
965
1042
  // states the floor it actually used.
966
1043
  function gcValidation(minRunTimeMs ) {
967
- return `This finding is gated by a ${minRunTimeMs / 1000}-second minimum-runtime floor, our own noise floor for this metric.`;
1044
+ return `Checked only on stages with at least ${minRunTimeMs / 1000}s of executor run time.`;
968
1045
  }
969
1046
 
970
- const INCOMPLETE_RUN_RECOMMENDATION = 'This event log never recorded an ApplicationEnd event: the capture stopped before the run finished (an in-flight job, a rotated-away log, or a cut-short capture). Findings and metrics elsewhere on this board reflect only what was captured up to that point, not the full run.';
1047
+ const INCOMPLETE_RUN_RECOMMENDATION = 'This event log never recorded an ApplicationEnd event: the capture stopped before the run finished (a job still running, a rotated log, or a cut-short capture), so every figure on this board covers only what was captured.';
971
1048
 
972
1049
  export const DETECTORS = [
973
1050
  defineStageDetector({
@@ -984,13 +1061,15 @@ export const DETECTORS = [
984
1061
  const floorWasteMs = tailClaimFloorMs(skewTailClaim(stage, metric === 'P95/median'), stage.id, ctx);
985
1062
  if (!meetsRuntimeFloor(floorWasteMs, appDurationMs(ctx.app), thresholds.floorPctWarn)) return null;
986
1063
  const value = Math.round(ratio * 10) / 10;
1064
+ const fix = skewJoinFix(ctx.app);
987
1065
  return {
988
1066
  type: 'skew', stageId: stage.id,
989
1067
  impactBand: 'warning',
990
1068
  metric, value,
991
1069
  confidence: skewConfidence(ratio, thresholds.ratioWarn),
992
- validationRequired: `This finding is gated by a ${shareLabel(thresholds.floorPctWarn)} runtime-floor threshold, our own noise floor for this metric.`,
993
- recommendation: `Task duration ratio (${metric}) is ${value}×: for join-driven skew, enable AQE skew-join handling (spark.sql.adaptive.skewJoin.enabled); otherwise salt the key or repartition on a better key to reduce task skew.`,
1070
+ validationRequired: `Flagged only when it costs at least ${shareLabel(thresholds.floorPctWarn)} of run time.`,
1071
+ recommendation: `Task duration ratio (${metric}) is ${value}×: ${fix.text}.`,
1072
+ remediation: fix.remediation,
994
1073
  };
995
1074
  },
996
1075
  estimate(finding, ctx) {
@@ -1002,14 +1081,22 @@ export const DETECTORS = [
1002
1081
  },
1003
1082
  }),
1004
1083
  defineStageDetector({
1005
- type: 'stageShape', order: 35, fixEffort: 'code', version: 1,
1084
+ type: 'stageShape', order: 35, fixEffort: 'code', version: 2,
1006
1085
  emits: ['stageShape'],
1007
1086
  docAnchor: '#bottleneck-stage-shape',
1008
1087
  // lowParallelismFloorPct: the same 0.5% runtime floor the tiered detectors use. Parallelizing a
1009
1088
  // stage can't save more than the stage's own duration, so a shorter stage can't clear it; on
1010
1089
  // the 14 real logs that was 2839 of 3005 lowParallelism findings (2168 on sub-second stages).
1011
1090
  // App-wide idle capacity stays covered by utilization.
1012
- thresholds: { pRatioMax: 0.5, oiRatioMax: 10, skewWarn: 3, lowParallelismFloorPct: 0.005 },
1091
+ // taskStageSkew: stageShareMin is the share of the stage's wall-clock the longest task must
1092
+ // span, and skewWarn how far past the median task it must run (skew's own max/median 3×).
1093
+ // The share alone can't tell a straggler apart: on a single wave (tasks <= cores) the longest
1094
+ // task spans nearly the whole stage however even the tasks are. taskStageSkewFloorPct is the
1095
+ // same 0.5% runtime floor as lowParallelismFloorPct.
1096
+ thresholds: {
1097
+ pRatioMax: 0.5, oiRatioMax: 10, skewWarn: 3, stageShareMin: 0.5,
1098
+ lowParallelismFloorPct: 0.005, taskStageSkewFloorPct: 0.005,
1099
+ },
1013
1100
  detect(stage, ctx, thresholds) {
1014
1101
  const out = [];
1015
1102
  const execCount = (stage.executorStats ?? []).length;
@@ -1025,7 +1112,7 @@ export const DETECTORS = [
1025
1112
  rule: 'lowParallelism', metric: 'pRatio', value: Math.round(pRatio * 100) / 100,
1026
1113
  // Absolute core count behind pRatio, for the impact estimator's idle-core-ms figure.
1027
1114
  totalCores,
1028
- recommendation: `This stage runs ${stage.taskCount} ${stage.taskCount === 1 ? 'task' : 'tasks'} across ~${totalCores} cores, so it is under-parallelized and leaves cluster capacity idle.`,
1115
+ recommendation: `This stage runs ${stage.taskCount} ${stage.taskCount === 1 ? 'task' : 'tasks'} across ~${totalCores} cores: it is under-parallelized and leaves cluster capacity idle.`,
1029
1116
  });
1030
1117
  }
1031
1118
  }
@@ -1040,18 +1127,21 @@ export const DETECTORS = [
1040
1127
  });
1041
1128
  }
1042
1129
  }
1043
- // TaskStageSkew: straggler cost vs stage wall-clock. Skip near-zero duration. Always info
1044
- // like its siblings: this trigger forces the occupancy-clipped estimate to exactly zero on
1045
- // every firing, so there's no wall-clock-backed tier left to gate on.
1046
- if (stageDurationMs > 0) {
1047
- const ratio = stage.taskDurationMax / stageDurationMs;
1048
- if (ratio > thresholds.skewWarn) {
1130
+ // TaskStageSkew: one straggler sets when the stage ends. A task runs inside its stage's
1131
+ // window, so the longest task's share of the stage's wall-clock is at most 1. Skip a
1132
+ // zero-length or single-task stage. Always info like its siblings: skew and straggler
1133
+ // already make the wall-clock claim for the same tail, so this one reports idle core-time.
1134
+ if (stageDurationMs > 0 && stage.taskCount > 1 && stage.taskDurationP50 > 0
1135
+ && !stageBelowRuntimeFloor(stage, ctx, thresholds.taskStageSkewFloorPct)) {
1136
+ const share = stage.taskDurationMax / stageDurationMs;
1137
+ const vsMedian = stage.taskDurationMax / stage.taskDurationP50;
1138
+ if (share > thresholds.stageShareMin && vsMedian > thresholds.skewWarn) {
1049
1139
  out.push({
1050
1140
  type: 'stageShape', stageId: stage.id, impactBand: 'info',
1051
- rule: 'taskStageSkew', metric: 'taskStageSkew', value: Math.round(ratio * 10) / 10,
1141
+ rule: 'taskStageSkew', metric: 'taskStageSkew', value: Math.round(share * 100) / 100,
1052
1142
  // Absolute core count, for the impact estimator's idle-core-ms figure.
1053
1143
  totalCores,
1054
- recommendation: `One task takes ${Math.round(ratio * 10) / 10}× this stage's wall-clock duration; a single straggler is gating the whole stage.`,
1144
+ recommendation: `The longest task ran for ${Math.round(share * 100)}% of this stage's wall-clock, ${Math.round(vsMedian * 10) / 10}× the median task: a single straggler is gating the whole stage.`,
1055
1145
  });
1056
1146
  }
1057
1147
  }
@@ -1065,7 +1155,7 @@ export const DETECTORS = [
1065
1155
  const idleCoreMs =
1066
1156
  Math.max(0, ((finding.totalCores ) ?? 0) - (stage.taskCount ?? 0)) * stageDurationMs;
1067
1157
  // Real per-stage data (cores, task count, duration), no assumed constant.
1068
- return costOnly('measured', { value: idleCoreMs, unit: 'coreMs' });
1158
+ return costOnly('measured', { value: idleCoreMs, unit: 'coreMs', idle: true });
1069
1159
  }
1070
1160
  if (finding.rule === 'dataExplosion') {
1071
1161
  const excessBytes = Math.max(0, (stage.outputBytes ?? 0) - (stage.inputBytes ?? 0));
@@ -1076,12 +1166,12 @@ export const DETECTORS = [
1076
1166
  const totalCores = (finding.totalCores ) ?? 0;
1077
1167
  const taskCount = stage.taskCount ?? 0;
1078
1168
  // Cores idle during the straggler's tail, at achieved concurrency (not full cluster
1079
- // capacity, which is lowParallelism's territory): this rule's trigger forces the
1080
- // occupancy-clipped estimate to zero on every firing, so it's resourceOnly, not a wall-clock claim.
1169
+ // capacity, which is lowParallelism's territory): resourceOnly, since skew and straggler
1170
+ // already claim that tail's wall-clock time.
1081
1171
  const idleCoreMs =
1082
1172
  Math.max(0, Math.min(totalCores, taskCount) - 1) *
1083
1173
  Math.max(0, (stage.taskDurationMax ?? 0) - (stage.taskDurationP50 ?? 0));
1084
- return costOnly('measured', { value: idleCoreMs, unit: 'coreMs' });
1174
+ return costOnly('measured', { value: idleCoreMs, unit: 'coreMs', idle: true });
1085
1175
  }
1086
1176
  return null;
1087
1177
  },
@@ -1103,6 +1193,7 @@ export const DETECTORS = [
1103
1193
  impactBand: 'info',
1104
1194
  metric: 'shuffleReadBytes', value: bytes,
1105
1195
  recommendation: `${formatBytes(bytes)} shuffled in this stage: consider increasing spark.sql.shuffle.partitions or adding a broadcast join.`,
1196
+ remediation: [increaseConf('spark.sql.shuffle.partitions')],
1106
1197
  };
1107
1198
  },
1108
1199
  estimate(finding, ctx) {
@@ -1125,7 +1216,7 @@ export const DETECTORS = [
1125
1216
  emits: ['partitionSizing'],
1126
1217
  docAnchor: '#bottleneck-partition-sizing',
1127
1218
  thresholds: { skewRatio: 5, skewFloorBytes: 256 * MB, lowParTotalBytes: GB, lowParMaxTasks: 7, maxPartBytes: 5 * GB },
1128
- detect(stage, _ctx, thresholds) {
1219
+ detect(stage, ctx, thresholds) {
1129
1220
  const out = [];
1130
1221
  const { shuffleReadP50: p50, shuffleReadMax: max, shuffleReadBytes: total, taskCount } = stage;
1131
1222
  if (max > thresholds.skewRatio * p50 && max > thresholds.skewFloorBytes) {
@@ -1133,18 +1224,22 @@ export const DETECTORS = [
1133
1224
  // "Infinity×", so fall back to median-free phrasing.
1134
1225
  const ratioText = p50 > 0
1135
1226
  ? `${Math.round(max / p50 * 10) / 10}× the median (${formatBytes(p50)})`
1136
- : `far larger than the median (${formatBytes(p50)}, effectively empty)`;
1227
+ : 'far larger than the median, which is effectively empty';
1228
+ const fix = skewJoinFix(ctx.app);
1137
1229
  out.push({
1138
1230
  type: 'partitionSizing', stageId: stage.id, impactBand: 'warning',
1139
1231
  rule: 'shufflePartitionSkew', metric: 'shuffleReadMax', value: max,
1140
- recommendation: `The largest shuffle partition (${formatBytes(max)}) is ${ratioText}: for join skew, enable AQE skew-join handling (spark.sql.adaptive.skewJoin.enabled); otherwise salt the key or repartition on a better key.`,
1232
+ recommendation: `The largest shuffle partition (${formatBytes(max)}) is ${ratioText}: ${fix.text}.`,
1233
+ remediation: fix.remediation,
1141
1234
  });
1142
1235
  }
1143
1236
  if (total >= thresholds.lowParTotalBytes && taskCount <= thresholds.lowParMaxTasks) {
1237
+ const fix = lowShuffleParallelismFix(ctx.app, Math.ceil(total / IDEAL_BYTES_PER_PARTITION_TASK));
1144
1238
  out.push({
1145
1239
  type: 'partitionSizing', stageId: stage.id, impactBand: 'warning',
1146
1240
  rule: 'lowShuffleParallelism', metric: 'taskCount', value: taskCount,
1147
- recommendation: `${Math.round(total / GB * 10) / 10} GB of shuffle spread over only ${taskCount} tasks: raise spark.sql.shuffle.partitions so each partition is smaller.`,
1241
+ recommendation: `${Math.round(total / GB * 10) / 10} GB of shuffle spread over only ${taskCount} tasks: ${fix.text}.`,
1242
+ remediation: fix.remediation,
1148
1243
  });
1149
1244
  }
1150
1245
  if (max >= thresholds.maxPartBytes) {
@@ -1212,6 +1307,7 @@ export const DETECTORS = [
1212
1307
  recommendation: cls === 'skew'
1213
1308
  ? `${formatBytes(stage.memoryBytesSpilled)} spilled, skew-driven: fix task skew first; adding memory will not help.`
1214
1309
  : `${formatBytes(stage.memoryBytesSpilled)} spilled: raise spark.sql.shuffle.partitions or increase executor memory.`,
1310
+ remediation: cls === 'skew' ? [] : [increaseConf('spark.sql.shuffle.partitions'), increaseConf('spark.executor.memory')],
1215
1311
  };
1216
1312
  },
1217
1313
  estimate(finding, ctx) {
@@ -1251,6 +1347,7 @@ export const DETECTORS = [
1251
1347
  metric: 'gcPct', value,
1252
1348
  confidence: gcConfidence(pct, thresholds, 'high'), validationRequired: gcValidation(thresholds.minRunTimeMs),
1253
1349
  recommendation: `GC consumed ${value}% of executor run time: reduce object creation, use primitive types, avoid UDFs, increase executor memory.`,
1350
+ remediation: [increaseConf('spark.executor.memory')],
1254
1351
  };
1255
1352
  }
1256
1353
  // Low-GC (cost) branch: only for stages that ran long enough to be meaningful.
@@ -1264,6 +1361,7 @@ export const DETECTORS = [
1264
1361
  metric: 'gcPct', value,
1265
1362
  confidence: gcConfidence(pct, thresholds, 'low'), validationRequired: gcValidation(thresholds.minRunTimeMs),
1266
1363
  recommendation: `GC consumed only ${value}% of executor run time: memory may be over-provisioned; consider reducing spark.executor.memory for cost savings.`,
1364
+ remediation: [decreaseConf('spark.executor.memory')],
1267
1365
  };
1268
1366
  }
1269
1367
  return null;
@@ -1315,6 +1413,9 @@ export const DETECTORS = [
1315
1413
  if ((hosts.length < thresholds.minHosts && execs0.length < thresholds.minHosts) || stage.taskCount < thresholds.minTasks) return null;
1316
1414
  if (stageBelowRuntimeFloor(stage, ctx, thresholds.stageFloorPct)) return null;
1317
1415
  const out = [];
1416
+ const speculation = switchFix(loggedAs(ctx.app, 'spark.speculation', true), 'spark.speculation', true,
1417
+ 'check what it was running, and consider enabling spark.speculation to relaunch a lagging task automatically',
1418
+ 'check what it was running; speculation is already on, so a lagging task there is already relaunched');
1318
1419
  if (hosts.length >= thresholds.minHosts) {
1319
1420
  const means = hosts.map(h => ({ host: h.host, taskCount: h.taskCount, mean: h.totalDuration / h.taskCount }));
1320
1421
  const sorted = [...means].map(h => h.mean).sort((a, b) => a - b);
@@ -1331,7 +1432,8 @@ export const DETECTORS = [
1331
1432
  // `value` is a ratio; the estimator needs the absolute per-host mean.
1332
1433
  hostMeanMs: h.mean,
1333
1434
  host: h.host, hostTaskShare: Math.round(share * 100) / 100,
1334
- recommendation: `${h.host} may just hold data locality for its tasks or carry one heavy stage, not necessarily a hardware fault: check what it was running, and consider enabling spark.speculation to relaunch a lagging task automatically.`,
1435
+ recommendation: `${h.host} may just hold data locality for its tasks or carry one heavy stage, not necessarily a hardware fault: ${speculation.text}.`,
1436
+ remediation: speculation.remediation,
1335
1437
  });
1336
1438
  }
1337
1439
  }
@@ -1387,7 +1489,7 @@ export const DETECTORS = [
1387
1489
  // in this dimension's own unit (ms for taskTime, bytes for the rest).
1388
1490
  execMaxValue: r.value,
1389
1491
  executorId: r.key,
1390
- recommendation: `Executor ${r.key} deviates ${Math.round(r.ratio * 10) / 10}× from the median on ${d.dimension}: investigate uneven partition assignment or a degraded executor.`,
1492
+ recommendation: `Executor ${r.key}'s ${SLOW_HOST_DIMENSION_LABEL[d.dimension]} is ${Math.round(r.ratio * 10) / 10}× the median: investigate uneven partition assignment or a degraded executor.`,
1391
1493
  });
1392
1494
  }
1393
1495
  return out;
@@ -1434,6 +1536,7 @@ export const DETECTORS = [
1434
1536
  type: 'stageSlowness', stageId: stage.id, impactBand,
1435
1537
  metric: 'stageDurationMinutes', value,
1436
1538
  recommendation: `This stage ran ${value} minutes with no more specific cause flagged: often a partition-count problem, raise parallelism via spark.sql.shuffle.partitions or spark.default.parallelism, or check for a large per-task data volume driving heavy shuffle and spill.`,
1539
+ remediation: [increaseConf('spark.sql.shuffle.partitions'), increaseConf('spark.default.parallelism')],
1437
1540
  };
1438
1541
  },
1439
1542
  estimate(finding, ctx) {
@@ -1454,7 +1557,7 @@ export const DETECTORS = [
1454
1557
  ? stage.taskActiveMs
1455
1558
  : Math.max(0, (stage.completedAt ?? 0) - (stage.submittedAt ?? 0));
1456
1559
  const taskCount = stage.taskCount ?? 0;
1457
- const wasteMs = readBytes <= 0 && tasksMostlyIdle(stage) ? 0 : activeMs * Math.max(0, 1 - taskCount / ctx.totalCores);
1560
+ const wasteMs = readBytes <= 0 && tasksMostlyIdle(stage, ctx.sql) ? 0 : activeMs * Math.max(0, 1 - taskCount / ctx.totalCores);
1458
1561
  return singleStageImpact(wasteMs, finding.stageId, ctx, 'modeled', { value: wasteMs, unit: 'ms' }, TAIL_CLAIM);
1459
1562
  },
1460
1563
  }),
@@ -1571,7 +1674,7 @@ export const DETECTORS = [
1571
1674
  confidence: useSpeculativeMetric
1572
1675
  ? stragglerConfidence(speculativeShare, thresholds.warnPct, thresholds.critPct)
1573
1676
  : stragglerConfidence(stragglerShare, thresholds.shareWarn, thresholds.critPct),
1574
- validationRequired: `This finding is gated by ${shareLabel(thresholds.floorPctWarn)}/${shareLabel(thresholds.floorPctCrit)} runtime-floor thresholds, our own noise floor for this metric.`,
1677
+ validationRequired: `Warning needs at least ${shareLabel(thresholds.floorPctWarn)} of run time at stake, critical ${shareLabel(thresholds.floorPctCrit)}.`,
1575
1678
  recommendation: `${detail}: rule out a GC pause or a slow shuffle fetch before assuming a hardware issue; if a skewed key is the real cause, that's a candidate for AQE's skew-join handling.`,
1576
1679
  };
1577
1680
  },
@@ -1596,7 +1699,8 @@ export const DETECTORS = [
1596
1699
  impactBand: 'warning',
1597
1700
  metric: 'speculationWasteMs', value: wastedMs,
1598
1701
  confidence: speculationWasteConfidence(wastedMs, thresholds.minWasteMs),
1599
- recommendation: `Speculative execution discarded ${Math.round(wastedMs / 1000)}s of executor time in this stage; if task durations are naturally variable rather than genuine stragglers, consider tuning spark.speculation.multiplier/quantile.`,
1702
+ recommendation: `Speculative execution discarded ${Math.round(wastedMs / 1000)}s of executor time in this stage: if task durations are naturally variable rather than genuine stragglers, consider tuning spark.speculation.multiplier/quantile.`,
1703
+ remediation: [increaseConf('spark.speculation.multiplier'), increaseConf('spark.speculation.quantile')],
1600
1704
  };
1601
1705
  },
1602
1706
  estimate(finding, ctx) {
@@ -1658,6 +1762,7 @@ export const DETECTORS = [
1658
1762
  type: 'tinyTask', stageId: stage.id, impactBand: 'info',
1659
1763
  metric: 'taskDurationP50', value: Math.round(stage.taskDurationP50),
1660
1764
  recommendation: `Many small tasks (${stage.taskCount}, P50 ${Math.round(stage.taskDurationP50)}ms): scheduler overhead may dominate. Try ${fix}.`,
1765
+ remediation: stage.shuffleReadBytes > 0 ? [decreaseConf('spark.sql.shuffle.partitions')] : [],
1661
1766
  };
1662
1767
  },
1663
1768
  estimate(finding, ctx) {
@@ -1736,6 +1841,7 @@ export const DETECTORS = [
1736
1841
  type: 'coldStart', stageId: null, impactBand: 'warning',
1737
1842
  metric: 'startupGapSeconds', value,
1738
1843
  recommendation: `The first stage waited ${value}s for an executor to become available: keep a warm pool of idle executors, or if using dynamic allocation, raise the minimum/initial executor count so it doesn't scale up from zero.`,
1844
+ remediation: dynamicAllocationOff(ctx.app) ? [] : [increaseConf('spark.dynamicAllocation.minExecutors'), increaseConf('spark.dynamicAllocation.initialExecutors')],
1739
1845
  };
1740
1846
  },
1741
1847
  estimate(finding) {
@@ -1776,13 +1882,14 @@ export const DETECTORS = [
1776
1882
  // CPU-time-based utilization (sparkMeasure): metric only, no threshold.
1777
1883
  let cpuUtilizationPct = null;
1778
1884
  if (totalCores > 0) {
1779
- let cpuMs = 0;
1780
- // executorCpuTime is reported by Spark in nanoseconds.
1781
- for (const s of ctx.stages.values()) cpuMs += nsToMs(s.executorCpuTime ?? 0);
1782
- cpuUtilizationPct = Math.round((cpuMs / (appDuration * totalCores)) * 100);
1885
+ // Null when no stage recorded CPU time (older Spark), the same rule as the CLI metrics block.
1886
+ const cpuMs = totalExecutorCpuMs(ctx.stages.values());
1887
+ cpuUtilizationPct = cpuMs == null ? null : Math.round((cpuMs / (appDuration * totalCores)) * 100);
1783
1888
  }
1784
1889
 
1785
1890
  const value = Math.round(utilization * 100);
1891
+ const fix = dynamicAllocationFix(app, 'consider reducing cluster size or enabling dynamic allocation',
1892
+ 'dynamic allocation is already on, so consider reducing cluster size');
1786
1893
  return {
1787
1894
  type: 'utilization', stageId: null, impactBand: 'info',
1788
1895
  metric: 'avgUtilization', value,
@@ -1790,7 +1897,8 @@ export const DETECTORS = [
1790
1897
  appDurationMs: appDuration,
1791
1898
  totalCores,
1792
1899
  cpuUtilizationPct,
1793
- recommendation: `Average executor utilization was only ${value}%: consider reducing cluster size or enabling dynamic allocation.`,
1900
+ recommendation: `Average executor utilization was only ${value}%: ${fix.text}.`,
1901
+ remediation: fix.remediation,
1794
1902
  };
1795
1903
  },
1796
1904
  estimate(finding) {
@@ -1801,7 +1909,7 @@ export const DETECTORS = [
1801
1909
  return costOnly('measured');
1802
1910
  }
1803
1911
  const idleCoreHours = (1 - fraction) * appDurationMs * totalCores / 3.6e6;
1804
- return costOnly('measured', { value: idleCoreHours, unit: 'coreHours' });
1912
+ return costOnly('measured', { value: idleCoreHours, unit: 'coreHours', idle: true });
1805
1913
  },
1806
1914
  }),
1807
1915
  defineAppDetector({
@@ -1838,12 +1946,15 @@ export const DETECTORS = [
1838
1946
  const idleRate = capacityCoreMs > 0 ? 1 - (busyCoreMs / capacityCoreMs) : 0;
1839
1947
  if (idleRate > thresholds.idleCoreWarn) {
1840
1948
  const value = Math.round(idleRate * 100);
1949
+ const fix = dynamicAllocationFix(app, 'reduce cluster size or enable dynamic allocation',
1950
+ 'dynamic allocation is already on, so reduce cluster size');
1841
1951
  out.push({
1842
1952
  type: 'memoryUtilization', variant: 'idleCores', stageId: null,
1843
1953
  impactBand: 'warning', metric: 'idleCoreRate', value,
1844
1954
  // Raw (unrounded) rate plus sizing inputs for the impact estimator: `value` is rounded pct.
1845
1955
  idleRateFraction: idleRate, allocatedMB, peakExecutors, appDurationMs,
1846
- recommendation: `${value}% of allocated core-time ran no task: reduce cluster size or enable dynamic allocation.`,
1956
+ recommendation: `${value}% of available core-time ran no task: ${fix.text}.`,
1957
+ remediation: fix.remediation,
1847
1958
  });
1848
1959
  }
1849
1960
  }
@@ -1860,10 +1971,15 @@ export const DETECTORS = [
1860
1971
  }
1861
1972
  }
1862
1973
  if (peakHeapByExec.size === 0) {
1974
+ const key = 'spark.eventLog.logStageExecutorMetrics';
1975
+ const fix = switchFix(loggedAs(app, key, true), key, true,
1976
+ `Per-executor memory usage requires ${key}=true: not enabled for this run.`,
1977
+ 'Per-executor memory usage is missing from this log even though executor metrics logging is on for this run.');
1863
1978
  out.push({
1864
1979
  type: 'memoryUtilization', variant: 'memoryBand', stageId: null,
1865
1980
  impactBand: 'info', metric: 'memoryBand', dataUnavailable: true,
1866
- recommendation: 'Per-executor memory usage requires spark.eventLog.logStageExecutorMetrics=true: not enabled for this run.',
1981
+ recommendation: fix.text,
1982
+ remediation: fix.remediation,
1867
1983
  });
1868
1984
  } else if (allocatedMB != null && allocatedMB > 0) {
1869
1985
  const allocatedBytes = allocatedMB * 1024 * 1024;
@@ -1877,6 +1993,7 @@ export const DETECTORS = [
1877
1993
  stageId: null, executorId: execId,
1878
1994
  impactBand: 'warning', metric: 'heapUsedRatio', value: Math.round(ratio * 100),
1879
1995
  recommendation: `Executor ${execId} peaked at ${Math.round(ratio * 100)}% of allocated heap: memory may be too small; raise spark.executor.memory to avoid OOM/spill.`,
1996
+ remediation: [increaseConf('spark.executor.memory')],
1880
1997
  });
1881
1998
  } else if (ratio < thresholds.bandTooHigh) {
1882
1999
  out.push({
@@ -1887,6 +2004,7 @@ export const DETECTORS = [
1887
2004
  // unused-memory-over-time model.
1888
2005
  allocatedBytes, heap, appDurationMs,
1889
2006
  recommendation: `Executor ${execId} used only ${Math.round(ratio * 100)}% of allocated heap: memory may be over-provisioned; consider reducing spark.executor.memory for cost savings.`,
2007
+ remediation: [decreaseConf('spark.executor.memory')],
1890
2008
  });
1891
2009
  }
1892
2010
  }
@@ -1907,6 +2025,7 @@ export const DETECTORS = [
1907
2025
  confidence: memoryWasteConfidence(wastedMBSeconds, usedMBSeconds, thresholds.wasteBufferMultiplier),
1908
2026
  validationRequired: `Memory-waste estimate uses allocated-vs-used memory-time and a ${thresholds.wasteBufferMultiplier}x buffer: confirm against the Spark UI before acting.`,
1909
2027
  recommendation: `Allocated executor memory sat largely idle over the run (~${value.toLocaleString('en-US')} MB-seconds wasted): review spark.executor.memory and executor count.`,
2028
+ remediation: [decreaseConf('spark.executor.memory')],
1910
2029
  });
1911
2030
  }
1912
2031
  }
@@ -1994,7 +2113,7 @@ export const DETECTORS = [
1994
2113
  }
1995
2114
  }
1996
2115
  }
1997
- if (persistedRddCount > 0 && !anyStorageEvidence) out.push(storageUnobservedFinding(persistedRddCount));
2116
+ if (persistedRddCount > 0 && !anyStorageEvidence) out.push(storageUnobservedFinding(persistedRddCount, ctx.app));
1998
2117
  return out;
1999
2118
  },
2000
2119
  estimate(finding) {
@@ -2038,8 +2157,8 @@ export const DETECTORS = [
2038
2157
  // Raw count behind the ratio, for the impact estimator. Non-null whenever totalTasks is.
2039
2158
  nonLocalTaskCount: nonLocalTasks ,
2040
2159
  confidence: coreLocalityConfidence(ratio , totalTasks, thresholds),
2041
- validationRequired: `This finding is gated by ${shareLabel(thresholds.warnRatio)}/${shareLabel(thresholds.critRatio)} non-local-ratio thresholds (and a ${thresholds.minTasks}-task minimum), our own noise floor for this metric.`,
2042
- recommendation: `${value}% of tasks (${nonLocalTasks }) ran without process- or node-local data placement: check spark.locality.wait settings and executor/data colocation.`,
2160
+ validationRequired: `Flagged when at least ${shareLabel(thresholds.warnRatio)} of tasks run non-local (critical at ${shareLabel(thresholds.critRatio)}), on runs of ${thresholds.minTasks}+ tasks.`,
2161
+ recommendation: `${value}% of tasks (${nonLocalTasks }) ran without process- or node-local data placement: check executor/data colocation.`,
2043
2162
  };
2044
2163
  },
2045
2164
  estimate(finding) {
@@ -2083,7 +2202,12 @@ export const DETECTORS = [
2083
2202
  // Raw count behind the percentage, for the impact estimator's startup-overhead figure.
2084
2203
  shortLivedExecutorCount: shortLivedCount,
2085
2204
  confidence: autoscalingChurnConfidence(shortLivedPct, thresholds.warningPct, thresholds.criticalPct),
2086
- recommendation: `${pct}% of executors ran for under 2 minutes before being removed. This looks like wasteful re-provisioning rather than normal scale-down; consider raising spark.dynamicAllocation.executorIdleTimeout or widening the minExecutors/maxExecutors bounds to reduce flapping.`,
2205
+ recommendation: `${pct}% of executors ran for under 2 minutes before being removed: this looks like wasteful re-provisioning rather than normal scale-down; consider raising spark.dynamicAllocation.executorIdleTimeout or widening the minExecutors/maxExecutors bounds to reduce flapping.`,
2206
+ remediation: [
2207
+ increaseConf('spark.dynamicAllocation.executorIdleTimeout'),
2208
+ decreaseConf('spark.dynamicAllocation.minExecutors'),
2209
+ increaseConf('spark.dynamicAllocation.maxExecutors'),
2210
+ ],
2087
2211
  };
2088
2212
  },
2089
2213
  estimate(finding) {
@@ -2221,8 +2345,8 @@ export const DETECTORS = [
2221
2345
  const relationDisplay = rids.map(relationDisplayName).join(` ${connector} `);
2222
2346
  const value = finalExecutionIds.length;
2223
2347
  const recommendation = totalReadBytes >= 128 * MB
2224
- ? `${verb[0].toUpperCase()}${verb.slice(1)} result read by ${value} queries (~${formatBytes(totalReadBytes)}). Cache/persist the ${verb} DataFrame so it is computed once.`
2225
- : `${verb[0].toUpperCase()}${verb.slice(1)} result read by ${value} queries. Cache the ${verb} DataFrame, or reconsider whether it needs to be recomputed each time.`;
2348
+ ? `${verb[0].toUpperCase()}${verb.slice(1)} result read by ${value} queries (~${formatBytes(totalReadBytes)}): cache/persist the ${verb} DataFrame so it is computed once.`
2349
+ : `${verb[0].toUpperCase()}${verb.slice(1)} result read by ${value} queries: cache the ${verb} DataFrame, or reconsider whether it needs to be recomputed each time.`;
2226
2350
 
2227
2351
  out.push({
2228
2352
  type: 'cachingOpportunity', variant: 'composite', stageId: null, impactBand: 'info',
@@ -2252,8 +2376,8 @@ export const DETECTORS = [
2252
2376
  const value = residualExecutionIds.length;
2253
2377
  const totalReadBytes = residualExecutionIds.reduce((sum, id) => sum + (agg.executionBytes.get(id) ?? 0), 0);
2254
2378
  const recommendation = totalReadBytes >= 128 * MB
2255
- ? `Read by ${value} queries (~${formatBytes(totalReadBytes)}). Cache/persist the shared DataFrame so it is scanned once.`
2256
- : `Read by ${value} queries. Cache the shared DataFrame, or broadcast it if it is a small join lookup.`;
2379
+ ? `Read by ${value} queries (~${formatBytes(totalReadBytes)}): cache/persist the shared DataFrame so it is scanned once.`
2380
+ : `Read by ${value} queries: cache the shared DataFrame, or broadcast it if it is a small join lookup.`;
2257
2381
  out.push({
2258
2382
  type: 'cachingOpportunity', stageId: null, impactBand: 'info',
2259
2383
  metric: 'executionReuse', value,
@@ -2328,6 +2452,7 @@ export const DETECTORS = [
2328
2452
  type: 'configAudit', property: 'spark.shuffle.service.enabled',
2329
2453
  impactBand: 'warning', metric: 'config', valueText: 'false',
2330
2454
  recommendation: 'Dynamic allocation is on but the external shuffle service is off: set spark.shuffle.service.enabled=true so shuffle data survives executor removal.',
2455
+ remediation: setConfUnlessLogged(target.app, 'spark.shuffle.service.enabled', true),
2331
2456
  };
2332
2457
  }
2333
2458
  return null;
@@ -2347,7 +2472,8 @@ export const DETECTORS = [
2347
2472
  return {
2348
2473
  type: 'configAudit', property: 'spark.dynamicAllocation.minExecutors',
2349
2474
  impactBand: 'critical', metric: 'config', valueText: `${minN} > ${maxN}`,
2350
- recommendation: `Autoscaling bounds are inverted: spark.dynamicAllocation.minExecutors (${minN}) exceeds maxExecutors (${maxN}). Set min ≤ max.`,
2475
+ recommendation: `spark.dynamicAllocation.minExecutors (${minN}) exceeds maxExecutors (${maxN}): set min ≤ max.`,
2476
+ remediation: [decreaseConf('spark.dynamicAllocation.minExecutors', maxN)],
2351
2477
  };
2352
2478
  }
2353
2479
  if (maxN == null) {
@@ -2355,6 +2481,7 @@ export const DETECTORS = [
2355
2481
  type: 'configAudit', property: 'spark.dynamicAllocation.maxExecutors',
2356
2482
  impactBand: 'info', metric: 'config', valueText: '(unset)',
2357
2483
  recommendation: 'Dynamic allocation is on with no upper bound: set spark.dynamicAllocation.maxExecutors to cap cluster growth.',
2484
+ remediation: [setConf('spark.dynamicAllocation.maxExecutors')],
2358
2485
  };
2359
2486
  }
2360
2487
  return null;
@@ -2375,6 +2502,7 @@ export const DETECTORS = [
2375
2502
  type: 'configAudit', property: 'spark.serializer',
2376
2503
  impactBand: 'info', metric: 'config', valueText: ser ?? '(default JavaSerializer)',
2377
2504
  recommendation: `Current serializer is ${ser ?? 'the default JavaSerializer'}: consider spark.serializer=org.apache.spark.serializer.KryoSerializer for faster, smaller buffers.`,
2505
+ remediation: setConfUnlessLogged(app, 'spark.serializer', 'org.apache.spark.serializer.KryoSerializer'),
2378
2506
  };
2379
2507
  },
2380
2508
  estimate: noWasteModel,
@@ -2394,6 +2522,7 @@ export const DETECTORS = [
2394
2522
  type: 'configAudit', property: 'spark.executor.memoryOverhead',
2395
2523
  impactBand: 'info', metric: 'config', valueText: `${ovMB} MiB`,
2396
2524
  recommendation: `Executor memoryOverhead (${ovMB} MiB) is below Spark's default floor of ${floor} MiB (max of 384 MiB or 10% of executor memory): raise it to avoid off-heap OOM-kills.`,
2525
+ remediation: [increaseConf('spark.executor.memoryOverhead', `${floor}m`)],
2397
2526
  };
2398
2527
  },
2399
2528
  estimate: noWasteModel,
@@ -2431,8 +2560,8 @@ export const DETECTORS = [
2431
2560
  const stageShares = stageOperatorShares(nodes, operatorsByStage);
2432
2561
  // resolvePlanTree always sets id; safe downstream of it.
2433
2562
  const planNodeIds = nodes.map((n) => n.id ).filter(Boolean);
2434
- const touching = g.sampleRelation ? ` (touching ${g.sampleRelation})` : '';
2435
- const differing = occurrencesIdentical ? '' : ' Their filters, columns or scanned tables differ, so the repeats may compute different data.';
2563
+ const detail = duplicateSubtreeDetail({ ...g, value: g.occurrences });
2564
+ const differing = occurrencesIdentical ? '' : ` ${DUPLICATE_SUBTREE_DIFFERING_NOTE}`;
2436
2565
  return {
2437
2566
  type: 'duplicatePlanSubtree', executionId: sqlExec.id, stageIds, planNodeIds,
2438
2567
  stageShares, occurrencesIdentical,
@@ -2444,10 +2573,10 @@ export const DETECTORS = [
2444
2573
  rootName: g.rootName, subtreeSize: g.subtreeSize, sampleRelation: g.sampleRelation,
2445
2574
  groupIndex: g.groupIndex,
2446
2575
  confidence: occurrencesIdentical ? duplicateSubtreeConfidence(g.subtreeSize, g.occurrences, thresholds) : 'low',
2447
- validationRequired: 'Duplicate-subtree matching compares operator names and metric names only, not literal values or expr IDs: confirm the repeated work is real in the Spark SQL plan tab before acting.',
2576
+ validationRequired: 'Matching compares operator and metric names only, not literals or expression IDs: confirm the repeat in the Spark UI SQL tab before acting.',
2448
2577
  recommendation: (g.isExchangeRoot
2449
- ? `A ${g.subtreeSize}-node subtree rooted at ${pathBasename(g.rootName)} repeats ${g.occurrences}x in this plan${touching}: this looks like a possible missed exchange reuse; check whether the same shuffle could be computed once and reused.`
2450
- : `A ${g.subtreeSize}-node subtree rooted at ${pathBasename(g.rootName)} repeats ${g.occurrences}x in this plan${touching}: consider caching/persisting the shared computation or check for a duplicated query branch.`) + differing,
2578
+ ? `${detail}: this looks like a possible missed exchange reuse; check whether the same shuffle could be computed once and reused.`
2579
+ : `${detail}: consider caching/persisting the shared computation or check for a duplicated query branch.`) + differing,
2451
2580
  };
2452
2581
  }).filter((f) => f !== null);
2453
2582
  return findings.length > 0 ? findings : null;
@@ -2583,6 +2712,7 @@ export const DETECTORS = [
2583
2712
  impactBand: 'info', metric: 'smallerSideBytes', value: smaller,
2584
2713
  largerSideBytes: larger,
2585
2714
  recommendation: `The smaller input to this Sort Merge Join (${formatBytes(smaller)}) is well under the broadcast threshold relative to the larger side (${formatBytes(larger)}): this could have been a broadcast join. Consider a broadcast() hint or raising spark.sql.autoBroadcastJoinThreshold.`,
2715
+ remediation: [increaseConf('spark.sql.autoBroadcastJoinThreshold')],
2586
2716
  });
2587
2717
  }
2588
2718
  }
@@ -2594,13 +2724,17 @@ export const DETECTORS = [
2594
2724
  // (node.stageIds always empty in real data); its child carries the executor-side
2595
2725
  // metrics, so only the child unions in.
2596
2726
  const child = (node.children ?? [])[0];
2727
+ const autoBroadcastOff = ctx.app?.config?.['spark.sql.autoBroadcastJoinThreshold']?.trim() === '-1';
2597
2728
  out.push({
2598
2729
  type: 'overBroadcast', executionId: sqlExec.id,
2599
2730
  stageIds: unionStageIds(child ? [child] : [], fallbackStageIds),
2600
2731
  // resolvePlanTree always sets id; safe downstream of it.
2601
2732
  planNodeIds: [node.id ].filter(Boolean),
2602
2733
  impactBand: 'warning', metric: 'broadcastBytes', value: m.value,
2603
- recommendation: `This broadcast (${formatBytes(m.value)}) exceeds the ${binaryThresholdLabel(overBroadcastBytes)} threshold: check for a misapplied broadcast hint or a misconfigured spark.sql.autoBroadcastJoinThreshold.`,
2734
+ recommendation: autoBroadcastOff
2735
+ ? `This broadcast (${formatBytes(m.value)}) exceeds the ${binaryThresholdLabel(overBroadcastBytes)} threshold: automatic broadcast is already disabled, so remove the broadcast() hint that forced it.`
2736
+ : `This broadcast (${formatBytes(m.value)}) exceeds the ${binaryThresholdLabel(overBroadcastBytes)} threshold: check for a misapplied broadcast hint or a misconfigured spark.sql.autoBroadcastJoinThreshold.`,
2737
+ remediation: autoBroadcastOff ? [] : [decreaseConf('spark.sql.autoBroadcastJoinThreshold')],
2604
2738
  });
2605
2739
  }
2606
2740
  }