sparkforensics-mcp 0.3.0 → 0.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (62) hide show
  1. package/bin/sparkforensics-mcp.mjs +41 -10
  2. package/package.json +1 -1
  3. package/vendor-core/analyzer.js +156 -48
  4. package/vendor-core/check-coverage.js +88 -0
  5. package/vendor-core/cli/budgets.js +31 -18
  6. package/vendor-core/cli/collect-run.js +76 -31
  7. package/vendor-core/cli/threshold-config.js +28 -0
  8. package/vendor-core/comparison-verdict.js +177 -0
  9. package/vendor-core/core-source-hash.txt +1 -0
  10. package/vendor-core/core-usage-locality.js +56 -2
  11. package/vendor-core/detector-docs.js +58 -0
  12. package/vendor-core/detectors.js +918 -458
  13. package/vendor-core/docs-config.js +0 -36
  14. package/vendor-core/docs-content/chapters/nav-index.json +31 -0
  15. package/vendor-core/docs-content/detection/cstor.md +9 -0
  16. package/vendor-core/docs-site-config.js +3 -0
  17. package/vendor-core/event-handlers.js +170 -6
  18. package/vendor-core/event-schemas.js +21 -0
  19. package/vendor-core/evidence-report.js +421 -112
  20. package/vendor-core/export-data.js +79 -6
  21. package/vendor-core/finding-action-label.js +9 -88
  22. package/vendor-core/finding-filter-predicate.js +9 -0
  23. package/vendor-core/finding-generic-recommendation.js +6 -104
  24. package/vendor-core/finding-names.js +21 -45
  25. package/vendor-core/finding-presentation.js +333 -0
  26. package/vendor-core/finding-tag-help.js +110 -0
  27. package/vendor-core/finding-types.js +361 -0
  28. package/vendor-core/findings-of-type.js +11 -0
  29. package/vendor-core/format-utils.js +92 -27
  30. package/vendor-core/html-export.js +51 -0
  31. package/vendor-core/impact-band.js +21 -8
  32. package/vendor-core/impact-estimator.js +8 -521
  33. package/vendor-core/impact-format.js +114 -0
  34. package/vendor-core/impact-model.js +175 -0
  35. package/vendor-core/ingest.js +2 -0
  36. package/vendor-core/intervals.js +13 -0
  37. package/vendor-core/list-runs.js +2 -3
  38. package/vendor-core/load-vendored.js +70 -5
  39. package/vendor-core/mcp-server-factory.js +14 -10
  40. package/vendor-core/mcp-tools.js +105 -45
  41. package/vendor-core/model-assembler.js +12 -0
  42. package/vendor-core/occupancy.js +1 -1
  43. package/vendor-core/parser-worker.js +1 -1
  44. package/vendor-core/plan-graph-model.js +3 -2
  45. package/vendor-core/plan-node-detail.js +1 -1
  46. package/vendor-core/recommendation-rollup.js +63 -3
  47. package/vendor-core/redact.js +45 -27
  48. package/vendor-core/run-comparison.js +32 -7
  49. package/vendor-core/run-interpretation.js +290 -0
  50. package/vendor-core/run-outcome.js +74 -0
  51. package/vendor-core/run-payload.js +17 -0
  52. package/vendor-core/run-shape.js +40 -0
  53. package/vendor-core/run-verdict.js +353 -0
  54. package/vendor-core/scaling-sim.js +4 -5
  55. package/vendor-core/scorecard-estimates.js +62 -0
  56. package/vendor-core/sql-stages.js +11 -0
  57. package/vendor-core/stage-quantiles.js +2 -0
  58. package/vendor-core/threshold-overrides.js +160 -0
  59. package/vendor-core/threshold-summary.js +11 -33
  60. package/vendor-core/types.js +6 -42
  61. package/vendor-core/wall-clock.js +1 -12
  62. package/vendor-core/wasted-core-hours.js +2 -2
@@ -3,16 +3,32 @@ import { scanRelationId } from './plan-summary.js';
3
3
  import { computePeakConcurrentCores, computePeakConcurrentExecutorCount } from './core-count.js';
4
4
  import { walkPlanTree } from './plan-tree-walk.js';
5
5
  import { computeCoreLocalityRatio } from './core-locality-ratio.js';
6
- import { estimateSingleStage, tailRecoveryMs, tailRemovedWorkMs, stragglerFixLongestTaskMs, } from './occupancy.js';
6
+ import { tailRecoveryMs, tailRemovedWorkMs, stragglerFixLongestTaskMs, } from './occupancy.js';
7
+ import { IMPACT_FLOOR_PCT_WARN, IMPACT_FLOOR_PCT_CRIT, appDurationMs } from './impact-band.js';
8
+ import {
9
+ BROADCAST_BANDWIDTH_BPS, EXECUTOR_STARTUP_OVERHEAD_MS, FILE_OPEN_OVERHEAD_MS, IDEAL_BYTES_PER_PARTITION_TASK,
10
+ NETWORK_FETCH_PENALTY_MS, RE_READ_THROUGHPUT_BPS, SHUFFLE_THROUGHPUT_BPS, SPILL_IO_THROUGHPUT_BPS, TAIL_CLAIM,
11
+ TASK_SCHEDULING_OVERHEAD_MS, costOnly, fetchWaitWallClockMs, measuredTaskOverhead, multiStageImpact, noWasteModel,
12
+ retryWallClockMs, singleStageImpact, stageIoParallelism, stageMappableWasteOrCostOnly, tasksMostlyIdle,
13
+
14
+ } from './impact-model.js';
7
15
  import { isExchangeNode, isBroadcastExchangeNode } from './plan-node-detail.js';
16
+ import { stageIdsForSqlExec } from './sql-stages.js';
8
17
  import { cyrb53 } from './string-hash.js';
9
18
  import { MAX_FAILURE_GROUPS, describeTaskFailure, } from './task-failure.js';
10
-
19
+
20
+
11
21
 
12
22
  const MB = 1024 * 1024;
13
23
  const GB = 1024 * MB;
14
24
  const TB = 1024 * GB;
15
25
 
26
+ // Labels a byte threshold in this file's binary units, so a 1 GiB default reads '1 GB'.
27
+ function binaryThresholdLabel(bytes ) {
28
+ const [unit, size] = ([['GB', GB], ['MB', MB], ['KB', 1024]] ).find(([, u]) => bytes >= u) ?? ['bytes', 1];
29
+ return `${Math.round(bytes / size * 10) / 10} ${unit}`;
30
+ }
31
+
16
32
  // Local runtime shapes.
17
33
  //
18
34
  // types.ts's Stage/SqlExecution/SparkAppInfo describe the posted AppModel surface for the view
@@ -27,16 +43,7 @@ const TB = 1024 * GB;
27
43
 
28
44
 
29
45
 
30
-
31
-
32
-
33
-
34
-
35
-
36
-
37
-
38
-
39
-
46
+
40
47
  // Per-executor snapshot from a StageExecutorMetrics event: a loose bag of Spark's
41
48
  // ExecutorMetrics field names, only a few of which any detector reads.
42
49
 
@@ -103,6 +110,7 @@ const TB = 1024 * GB;
103
110
 
104
111
 
105
112
 
113
+
106
114
 
107
115
 
108
116
 
@@ -115,7 +123,10 @@ const TB = 1024 * GB;
115
123
 
116
124
 
117
125
 
126
+
118
127
 
128
+
129
+
119
130
 
120
131
 
121
132
 
@@ -128,25 +139,22 @@ const TB = 1024 * GB;
128
139
 
129
140
 
130
141
 
131
- // The full context analyze() passes as every stage/sql detect()'s second arg, and as the sole
132
- // arg for 'app' scope. 'config' scope gets a narrower `{ app }` (auditConfig), typed per-entry.
142
+ // The full context analyze() passes as every stage/sql detect()'s second arg, and as the first
143
+ // arg for 'app' scope. 'config' scope gets a narrower `{ app }` (auditConfig) instead.
133
144
  //
134
- // `app` is typed as non-nullable here (unlike AppModel.app), but incompleteRun, coldStart,
135
- // utilization, and autoscalingChurn defensively guard against null at runtime to tolerate
136
- // malformed/incomplete logs. Despite the type annotation, app may be null in edge cases, and
137
- // these detectors handle it gracefully. auditConfig's config-scope target keeps `app` nullable
138
- // instead.
145
+ // `app` is nullable as on AppModel.app: a malformed or cut-short log may have no app record, so
146
+ // every app-reading detector guards it. `runAggregates.busyCoreMs` is optional for the same reason.
139
147
 
140
-
148
+
141
149
 
142
150
 
143
151
 
144
152
 
145
153
 
146
-
147
-
154
+
148
155
 
149
-
156
+
157
+
150
158
 
151
159
 
152
160
  // auditConfig(app) calls every config-scope detect({ app }) with whatever appModel.app is
@@ -190,18 +198,6 @@ function pickDominantReason(reasons )
190
198
  return [...reasons].sort((a, b) => b.count - a.count)[0].reason;
191
199
  }
192
200
 
193
- // Shared by every scope:'sql' detector. sql.get(id).stageIds is always empty (parser-worker
194
- // never populates it), so stage linkage is derived from each stage's own sqlExecutionId.
195
- // Minimal param shape (not DetectorStage): external callers pass Map<StageId, Stage>.
196
- export function stageIdsForSqlExec(
197
- executionId ,
198
- stages ,
199
- ) {
200
- const out = [];
201
- for (const s of stages.values()) if (s.sqlExecutionId === executionId) out.push(s.id);
202
- return out;
203
- }
204
-
205
201
  // Shared by the three Plan Advisor detectors: union the given nodes' own stageIds, or fall back
206
202
  // to the whole execution's stage set when none have coverage (never a partial blend). `fallback`
207
203
  // is a param so callers compute stageIdsForSqlExec once per detect() and reuse it per finding.
@@ -550,15 +546,19 @@ function maxMedianRatio(
550
546
  }
551
547
 
552
548
 
553
-
549
+
554
550
 
555
-
556
-
551
+
552
+
553
+
557
554
 
555
+
556
+
558
557
 
559
558
 
560
559
  // Machine-readable detector metadata for the evidence report (no `detect` closure), so a
561
- // portable report records which detector + thresholds produced each finding.
560
+ // portable report records which detector + thresholds produced each finding. These are the
561
+ // defaults; tunedDetectorCatalog() (threshold-overrides.ts) is the same rows under overrides.
562
562
  export function detectorCatalog() {
563
563
  return DETECTORS.map((d) => ({
564
564
  type: d.type,
@@ -587,49 +587,74 @@ export function computeSkewRatio(
587
587
 
588
588
  // Absolute-magnitude floor as a % of app runtime, not a fixed ms constant: a skew/straggler
589
589
  // ratio on a few ms is noise in an hours-long run but real in a seconds-long one; a fixed-ms
590
- // floor can't scale. Used by skew/straggler, gated via clippedWasteMs against the same
591
- // occupancy-clipped figure impact-estimator.ts displays as savings.
590
+ // floor can't scale. Used by skew/straggler, gated on the same occupancy-clipped tail claim their
591
+ // estimate() displays as savings (tailClaimImpact).
592
592
  // NOT SOURCED: floor percentages are our own noise floor, unvalidated.
593
- function computeAppDurationMs(ctx ) {
594
- const app = ctx?.app;
595
- if (app?.startTime == null || app?.endTime == null) return null;
596
- const durationMs = app.endTime - app.startTime;
597
- return durationMs > 0 ? durationMs : null;
598
- }
599
-
600
593
  // Unknown app timing never suppresses a finding; it just skips the floor gate.
601
- function meetsRuntimeFloor(wasteMs , appDurationMs , floorPct ) {
602
- return appDurationMs == null || wasteMs >= appDurationMs * floorPct;
594
+ function meetsRuntimeFloor(wasteMs , runMs , floorPct ) {
595
+ return runMs == null || wasteMs >= runMs * floorPct;
603
596
  }
604
597
 
605
598
  // A stage that ran for less than floorPct of the run (a known duration): an estimate clipped to
606
599
  // the stage can't reach floorPct, so every finding there grades info. A zero-length stage (no
607
600
  // submission time on older Spark) is not skipped: it gets no estimate and keeps its own band.
608
- function stageBelowRuntimeFloor(stage , ctx , floorPct ) {
601
+ function stageBelowRuntimeFloor(stage , ctx , floorPct ) {
609
602
  const stageDurationMs = (stage.completedAt ?? 0) - (stage.submittedAt ?? 0);
610
- const appDurationMs = computeAppDurationMs(ctx);
611
- return appDurationMs != null && stageDurationMs > 0 && stageDurationMs < appDurationMs * floorPct;
603
+ const runMs = appDurationMs(ctx.app);
604
+ return runMs != null && stageDurationMs > 0 && stageDurationMs < runMs * floorPct;
612
605
  }
613
606
 
614
- // Runs a raw waste delta through the same occupancy clip impact-estimator.ts applies before
615
- // display, so the runtime floor checks recoverable wall-clock, not a delta a physical floor
616
- // leaves unrecoverable. Falls back to the raw delta when occupancy data is unavailable.
617
- // skew/straggler claims shorten the stage's longest task, hence shortensLongestTask (see occupancy.ts).
618
- function clippedWasteMs(
619
- wasteMs , stageId , ctx , removedCoreWorkMs , longestTaskAfterFixMs = 0,
620
- ) {
621
- if (!ctx) return wasteMs;
622
- const est = estimateSingleStage(
623
- wasteMs, stageId, ctx.stages , ctx.occupancy,
624
- { shortensLongestTask: true, removedCoreWorkMs, longestTaskAfterFixMs },
625
- );
626
- return est ? est.wallClock.high : wasteMs;
607
+ // What a skew or straggler fix claims off its stage: the wall-clock its slow tail costs, the task
608
+ // time the fix removes, and the longest task it leaves. One figure, read twice: detect() gates its
609
+ // runtime floor on the claim's clipped estimate and estimate() reports that same estimate as the
610
+ // savings, so the firing floor and the displayed figure can't disagree.
611
+
612
+
613
+
614
+
615
+
616
+
617
+ // singleDelta: the slowest task's own excess, the fallback when the stage has no task replay.
618
+ function tailClaim(stage , singleDelta , longestTaskAfterFixMs ) {
619
+ return {
620
+ wasteMs: tailRecoveryMs(stage, singleDelta),
621
+ removedCoreWorkMs: tailRemovedWorkMs(stage, singleDelta),
622
+ longestTaskAfterFixMs,
623
+ };
624
+ }
625
+
626
+ // skew's delta is the task computeSkewRatio's metric sampled (P95 or max) over the median. Fixing
627
+ // the skew still waits on the longest task it leaves, as for straggler.
628
+ function skewTailClaim(stage , usesP95Branch ) {
629
+ const p50 = stage.taskDurationP50 ?? 0;
630
+ const singleDelta = Math.max(0, usesP95Branch ? (stage.taskDurationP95 ?? 0) - p50 : (stage.taskDurationMax ?? 0) - p50);
631
+ return tailClaim(stage, singleDelta, stragglerFixLongestTaskMs(stage));
632
+ }
633
+
634
+ // straggler's delta is the slowest task over the longest one the fix leaves.
635
+ function stragglerTailClaim(stage ) {
636
+ const longestTaskAfterFixMs = stragglerFixLongestTaskMs(stage);
637
+ return tailClaim(stage, Math.max(0, (stage.taskDurationMax ?? 0) - longestTaskAfterFixMs), longestTaskAfterFixMs);
627
638
  }
628
639
 
629
- // Shared by cacheUtilization's two variants: the ratio is a point-in-time storage snapshot from
630
- // stage-submission events, not a runtime block-access read-count.
631
- const CACHE_UTILIZATION_VALIDATION =
632
- "This ratio is a point-in-time storage snapshot from stage-submission events, not a runtime read-count. Confirm against the Spark UI's Storage tab before acting.";
640
+ // A tail claim shortens the stage's longest task, hence TAIL_CLAIM (see occupancy.ts).
641
+ function tailClaimImpact(claim , stageId , ctx ) {
642
+ return singleStageImpact(claim.wasteMs, stageId, ctx, 'measured', { value: claim.wasteMs, unit: 'ms' },
643
+ { ...TAIL_CLAIM, removedCoreWorkMs: claim.removedCoreWorkMs, longestTaskAfterFixMs: claim.longestTaskAfterFixMs });
644
+ }
645
+
646
+ // The figure a runtime floor checks: the claim's recoverable wall-clock, not a delta a physical
647
+ // floor leaves unrecoverable. Falls back to the raw claim when occupancy data is unavailable.
648
+ function tailClaimFloorMs(claim , stageId , ctx ) {
649
+ return tailClaimImpact(claim, stageId, ctx.impact).wallClock?.high ?? claim.wasteMs;
650
+ }
651
+
652
+ // Shared by cacheUtilization's two variants, worded per storage source: neither is a runtime
653
+ // block-access read-count.
654
+ const CACHE_UTILIZATION_VALIDATION = {
655
+ rddInfo: "This ratio is a point-in-time storage snapshot from stage-submission events, not a runtime read-count. Confirm against the Spark UI's Storage tab before acting.",
656
+ blockUpdates: "This ratio is the RDD's peak cache residency rebuilt from block-update events, not a runtime read-count. Confirm against the Spark UI's Storage tab before acting.",
657
+ } ;
633
658
 
634
659
  // Both cachedRatio and diskRatio are percentages of an RDD's partition/byte count: with few
635
660
  // partitions, one partition flipping cached/evicted (or disk-resident/memory-resident) swings the
@@ -653,7 +678,7 @@ function partialCacheFinding(rdd , cachedRatio , impactBa
653
678
  rddId: rdd.id, rddName,
654
679
  impactBand, metric: 'cachedRatio', value: cachedPct,
655
680
  confidence: cacheSampleConfidence(rdd.numPartitions),
656
- validationRequired: CACHE_UTILIZATION_VALIDATION,
681
+ validationRequired: CACHE_UTILIZATION_VALIDATION[rdd.storageSource ?? 'rddInfo'],
657
682
  memorySize: rdd.memorySize, diskSize: rdd.diskSize,
658
683
  numCachedPartitions: rdd.numCachedPartitions, numPartitions: rdd.numPartitions,
659
684
  recommendation: `RDD ${rddName} is ${evictedPct}% evicted from cache (${cachedPct}% of partitions cached). Increase executor memory or reduce the cached dataset size.`,
@@ -668,35 +693,164 @@ function diskSpilloverFinding(rdd , diskRatio , impactBan
668
693
  rddId: rdd.id, rddName,
669
694
  impactBand, metric: 'diskRatio', value: diskPct,
670
695
  confidence: cacheSampleConfidence(rdd.numPartitions),
671
- validationRequired: CACHE_UTILIZATION_VALIDATION,
696
+ validationRequired: CACHE_UTILIZATION_VALIDATION[rdd.storageSource ?? 'rddInfo'],
672
697
  memorySize: rdd.memorySize, diskSize: rdd.diskSize,
673
698
  numCachedPartitions: rdd.numCachedPartitions, numPartitions: rdd.numPartitions,
674
699
  recommendation: `RDD ${rddName} is ${diskPct}% spilled to disk despite requesting MEMORY_AND_DISK. Executor memory may be too small for this cached dataset.`,
675
700
  };
676
701
  }
677
702
 
678
- // Entry shape for every DETECTORS item. TTarget stays `unknown` at the array level: detect's
679
- // real first-arg varies by scope (DetectorStage/DetectorSqlExec/DetectorCtx/{ app }), and
680
- // unifying them would need an unsound cast or a discriminated-union redesign. Each entry gets a
681
- // precise detect by annotating its own params: object-literal method params are checked
682
- // bivariantly, so a narrower annotation here doesn't conflict with the `unknown` declaration.
683
-
684
-
685
-
703
+ // Persisted RDDs with no storage evidence at all: no block updates in the log, and RDD Info's
704
+ // sizes are the 0 that Spark 2.3+ always writes. Reports the gap instead of a clean result, the
705
+ // same missing-evidence shape as memoryUtilization's dataUnavailable caveat.
706
+ function storageUnobservedFinding(persistedRddCount ) {
707
+ const rdds = persistedRddCount === 1 ? '1 persisted RDD has' : `${persistedRddCount} persisted RDDs have`;
708
+ return {
709
+ type: 'cacheUtilization', variant: 'storageUnobserved', stageId: null,
710
+ impactBand: 'info', metric: 'persistedRdds', value: persistedRddCount, dataUnavailable: true,
711
+ recommendation: `${rdds} no cache-storage evidence in this log, so eviction and disk spillover can't be checked: Spark 2.3+ records cached sizes only as block updates, which need spark.eventLog.logBlockUpdates.enabled=true.`,
712
+ };
713
+ }
714
+
715
+ /** A detector entry's threshold set. number[] too: slowHost's ratioTiers and broadcastSizing's
716
+ * tiers are tier tables its detect() indexes by position. */
717
+
718
+
719
+
720
+
721
+ // What each scope's detect() is handed. `thresholds` is the entry's own set, or the caller's
722
+ // overrides merged over it (analyze()'s `thresholds` option). 'config' has no DetectorCtx:
723
+ // auditConfig() runs those entries on the app alone.
724
+
725
+
726
+
727
+
728
+
729
+
730
+
731
+
732
+
733
+ // The same calls with the thresholds already bound: what a runner holds after withThresholds().
734
+
735
+
736
+
737
+
738
+
739
+
740
+
741
+ // The fields a define*Detector() call spells out. `detect` is a property, not a method, so its
742
+ // parameters are checked contravariantly: a detect() that reads a threshold the entry doesn't
743
+ // declare, or expects a different target, fails to compile.
744
+
745
+
746
+
747
+
748
+
749
+
750
+
751
+
752
+
753
+
686
754
 
687
755
 
688
756
 
689
757
 
690
-
691
-
692
-
693
-
694
-
758
+
695
759
 
696
760
 
697
-
761
+
698
762
 
763
+
764
+
765
+
766
+
767
+
768
+
699
769
 
770
+
771
+
772
+
773
+
774
+
775
+
776
+
777
+
778
+
779
+
780
+
781
+
782
+ /** How runners (analyze(), auditConfig(), the catalog helpers) see any DETECTORS entry: its
783
+ * thresholds type erased and detect() left off, so they reach it only through withThresholds().
784
+ * estimate() takes any Finding here: estimateImpact() only hands an entry the types it emits. */
785
+
786
+
787
+
788
+
789
+
790
+
791
+ // Same-shape overrides merged over the defaults. analyze()'s callers validate overrides at their
792
+ // own boundary (threshold-overrides.ts); this re-checks so a programmatic caller can't hand a
793
+ // detector a threshold of the wrong shape, which is what makes the cast below sound.
794
+ function mergeThresholds (type , defaults , overrides ) {
795
+ for (const [name, value] of Object.entries(overrides)) {
796
+ // Own keys only: an inherited name such as `constructor` or `toString` is no threshold.
797
+ if (!Object.hasOwn(defaults, name)) throw new Error(`Detector ${type} has no threshold "${name}".`);
798
+ const fallback = defaults[name];
799
+ const sameShape = Array.isArray(fallback)
800
+ ? Array.isArray(value) && value.length === fallback.length
801
+ : typeof value === 'number';
802
+ if (!sameShape) throw new Error(`Threshold ${type}.${name} must have the same shape as its default.`);
803
+ }
804
+ return Object.freeze({ ...defaults, ...overrides }) ;
805
+ }
806
+
807
+ function defineDetector
808
+
809
+
810
+ (
811
+ scope , spec ,
812
+ bind ,
813
+ ) {
814
+ // Frozen: the defaults are the specification, never a knob to mutate in place.
815
+ const defaults = Object.freeze({ ...spec.thresholds }) ;
816
+ const boundDefaults = bind(defaults);
817
+ return {
818
+ ...spec, scope, thresholds: defaults,
819
+ withThresholds: (overrides) => (overrides && Object.keys(overrides).length > 0
820
+ ? bind(mergeThresholds(spec.type, defaults, overrides))
821
+ : boundDefaults),
822
+ };
823
+ }
824
+
825
+ // One helper per scope: each infers the entry's thresholds type from its `thresholds` literal
826
+ // and hands detect() exactly that type, with the scope's target and a required context.
827
+ export function defineStageDetector
828
+
829
+
830
+ (spec ) {
831
+ return defineDetector('stage', spec, (t) => (stage, ctx) => spec.detect(stage, ctx, t));
832
+ }
833
+
834
+ export function defineSqlDetector
835
+
836
+
837
+ (spec ) {
838
+ return defineDetector('sql', spec, (t) => (sqlExec, ctx) => spec.detect(sqlExec, ctx, t));
839
+ }
840
+
841
+ export function defineAppDetector
842
+
843
+
844
+ (spec ) {
845
+ return defineDetector('app', spec, (t) => (ctx) => spec.detect(ctx, t));
846
+ }
847
+
848
+ export function defineConfigDetector
849
+
850
+
851
+ (spec ) {
852
+ return defineDetector('config', spec, (t) => (target) => spec.detect(target, t));
853
+ }
700
854
 
701
855
  // Threshold field naming convention:
702
856
  // *Pct = 0–1 fraction (normalized)
@@ -704,11 +858,6 @@ function diskSpilloverFinding(rdd , diskRatio , impactBan
704
858
  // *Ratio = multiplicative factor
705
859
  // *Share/*Rate/*Util = 0–1 fraction (normalized)
706
860
 
707
- // straggler's noise-floor thresholds (NOT SOURCED: unvalidated), exported so impact-band.ts
708
- // reuses the same figures instead of hand-copying.
709
- export const STRAGGLER_FLOOR_PCT_WARN = 0.005;
710
- export const STRAGGLER_FLOOR_PCT_CRIT = 0.02;
711
-
712
861
  // A ratio just past ratioWarn is the case most likely to be ordinary task-duration variance
713
862
  // rather than real skew; a ratio many multiples past it (a 50x P95/median vs. a 3.1x one) is
714
863
  // unambiguous. 1.5x/5x mirror this file's other "just past the floor vs. clearly past it" splits
@@ -806,62 +955,71 @@ function cachingReuseConfidence(occurrences , minExecutions )
806
955
  return 'medium';
807
956
  }
808
957
 
809
- export const DETECTORS = [
810
- {
811
- type: 'skew', scope: 'stage', order: 30, fixEffort: 'code', version: 1,
958
+ // A share threshold as caveat text states it: 0.005 -> "0.5%". Rounded to 4 decimals of a percent
959
+ // so float noise (0.07 * 100) never prints.
960
+ function shareLabel(share ) {
961
+ return `${Math.round(share * 1e6) / 1e4}%`;
962
+ }
963
+
964
+ // Caveats that name a threshold read it from the thresholds the detector ran with, so a tuned run
965
+ // states the floor it actually used.
966
+ function gcValidation(minRunTimeMs ) {
967
+ return `This finding is gated by a ${minRunTimeMs / 1000}-second minimum-runtime floor, our own noise floor for this metric.`;
968
+ }
969
+
970
+ const INCOMPLETE_RUN_RECOMMENDATION = 'This event log never recorded an ApplicationEnd event: the capture stopped before the run finished (an in-flight job, a rotated-away log, or a cut-short capture). Findings and metrics elsewhere on this board reflect only what was captured up to that point, not the full run.';
971
+
972
+ export const DETECTORS = [
973
+ defineStageDetector({
974
+ type: 'skew', order: 30, fixEffort: 'code', version: 1,
975
+ emits: ['skew'],
812
976
  docAnchor: '#bottleneck-skew',
813
- thresholds: { ratioWarn: 3, minTasksForP95: 20, floorPctWarn: 0.005 },
814
- detect(
815
-
816
-
817
-
818
- stage ,
819
- ctx ,
820
- ) {
821
- const result = computeSkewRatio(stage, this.thresholds.minTasksForP95);
977
+ thresholds: { ratioWarn: 3, minTasksForP95: 20, floorPctWarn: IMPACT_FLOOR_PCT_WARN },
978
+ detect(stage, ctx, thresholds) {
979
+ const result = computeSkewRatio(stage, thresholds.minTasksForP95);
822
980
  if (result === null) return null;
823
981
  const { ratio, metric } = result;
824
- if (ratio <= this.thresholds.ratioWarn) return null;
825
- // Same absolute delta impact-estimator.ts's 'skew' case reports as savings; clipped the
826
- // same way before the floor check so the gate agrees with what's displayed.
827
- const singleDelta = Math.max(0, metric === 'P95/median' ? stage.taskDurationP95 - stage.taskDurationP50 : stage.taskDurationMax - stage.taskDurationP50);
828
- const wasteMs = tailRecoveryMs(stage, singleDelta);
829
- const appDurationMs = computeAppDurationMs(ctx);
830
- const floorWasteMs = clippedWasteMs(wasteMs, stage.id, ctx, tailRemovedWorkMs(stage, singleDelta), stragglerFixLongestTaskMs(stage));
831
- if (!meetsRuntimeFloor(floorWasteMs, appDurationMs, this.thresholds.floorPctWarn)) return null;
982
+ if (ratio <= thresholds.ratioWarn) return null;
983
+ // The claim estimate() reports as savings, clipped the same way, so the gate agrees with it.
984
+ const floorWasteMs = tailClaimFloorMs(skewTailClaim(stage, metric === 'P95/median'), stage.id, ctx);
985
+ if (!meetsRuntimeFloor(floorWasteMs, appDurationMs(ctx.app), thresholds.floorPctWarn)) return null;
832
986
  const value = Math.round(ratio * 10) / 10;
833
987
  return {
834
988
  type: 'skew', stageId: stage.id,
835
989
  impactBand: 'warning',
836
990
  metric, value,
837
- confidence: skewConfidence(ratio, this.thresholds.ratioWarn),
838
- validationRequired: 'This finding is gated by a 0.5% runtime-floor threshold, our own noise floor for this metric.',
991
+ confidence: skewConfidence(ratio, thresholds.ratioWarn),
992
+ validationRequired: `This finding is gated by a ${shareLabel(thresholds.floorPctWarn)} runtime-floor threshold, our own noise floor for this metric.`,
839
993
  recommendation: `Task duration ratio (${metric}) is ${value}×: for join-driven skew, enable AQE skew-join handling (spark.sql.adaptive.skewJoin.enabled); otherwise salt the key or repartition on a better key to reduce task skew.`,
840
994
  };
841
995
  },
842
- },
843
- {
844
- type: 'stageShape', scope: 'stage', order: 35, fixEffort: 'code', version: 1,
996
+ estimate(finding, ctx) {
997
+ if (finding.stageId == null) return null;
998
+ const stage = ctx.stages.get(finding.stageId);
999
+ if (!stage) return null;
1000
+ // computeSkewRatio's own metric labels: 'P95/median' or 'max/median'.
1001
+ return tailClaimImpact(skewTailClaim(stage, finding.metric === 'P95/median'), finding.stageId, ctx);
1002
+ },
1003
+ }),
1004
+ defineStageDetector({
1005
+ type: 'stageShape', order: 35, fixEffort: 'code', version: 1,
1006
+ emits: ['stageShape'],
845
1007
  docAnchor: '#bottleneck-stage-shape',
846
1008
  // lowParallelismFloorPct: the same 0.5% runtime floor the tiered detectors use. Parallelizing a
847
1009
  // stage can't save more than the stage's own duration, so a shorter stage can't clear it; on
848
1010
  // the 14 real logs that was 2839 of 3005 lowParallelism findings (2168 on sub-second stages).
849
1011
  // App-wide idle capacity stays covered by utilization.
850
1012
  thresholds: { pRatioMax: 0.5, oiRatioMax: 10, skewWarn: 3, lowParallelismFloorPct: 0.005 },
851
- detect(
852
-
853
- stage ,
854
- ctx ,
855
- ) {
1013
+ detect(stage, ctx, thresholds) {
856
1014
  const out = [];
857
1015
  const execCount = (stage.executorStats ?? []).length;
858
- const cores = ctx?.app?.resources?.executor?.cores ?? 1;
1016
+ const cores = ctx.app?.resources?.executor?.cores ?? 1;
859
1017
  const totalCores = execCount * cores;
860
1018
  const stageDurationMs = (stage.completedAt ?? 0) - (stage.submittedAt ?? 0);
861
1019
  // PRatio: under-parallelization.
862
- if (totalCores > 0 && !stageBelowRuntimeFloor(stage, ctx, this.thresholds.lowParallelismFloorPct)) {
1020
+ if (totalCores > 0 && !stageBelowRuntimeFloor(stage, ctx, thresholds.lowParallelismFloorPct)) {
863
1021
  const pRatio = stage.taskCount / totalCores;
864
- if (pRatio < this.thresholds.pRatioMax) {
1022
+ if (pRatio < thresholds.pRatioMax) {
865
1023
  out.push({
866
1024
  type: 'stageShape', stageId: stage.id, impactBand: 'info',
867
1025
  rule: 'lowParallelism', metric: 'pRatio', value: Math.round(pRatio * 100) / 100,
@@ -874,7 +1032,7 @@ export const DETECTORS = [
874
1032
  // OIRatio: data explosion. Skip when inputBytes is 0 (Infinity guard).
875
1033
  if (stage.inputBytes > 0) {
876
1034
  const oiRatio = stage.outputBytes / stage.inputBytes;
877
- if (oiRatio > this.thresholds.oiRatioMax) {
1035
+ if (oiRatio > thresholds.oiRatioMax) {
878
1036
  out.push({
879
1037
  type: 'stageShape', stageId: stage.id, impactBand: 'info',
880
1038
  rule: 'dataExplosion', metric: 'oiRatio', value: Math.round(oiRatio * 10) / 10,
@@ -887,7 +1045,7 @@ export const DETECTORS = [
887
1045
  // every firing, so there's no wall-clock-backed tier left to gate on.
888
1046
  if (stageDurationMs > 0) {
889
1047
  const ratio = stage.taskDurationMax / stageDurationMs;
890
- if (ratio > this.thresholds.skewWarn) {
1048
+ if (ratio > thresholds.skewWarn) {
891
1049
  out.push({
892
1050
  type: 'stageShape', stageId: stage.id, impactBand: 'info',
893
1051
  rule: 'taskStageSkew', metric: 'taskStageSkew', value: Math.round(ratio * 10) / 10,
@@ -899,22 +1057,47 @@ export const DETECTORS = [
899
1057
  }
900
1058
  return out;
901
1059
  },
902
- },
903
- {
904
- type: 'shuffle', scope: 'stage', order: 20, fixEffort: 'config', version: 1,
1060
+ estimate(finding, ctx) {
1061
+ const stage = ctx.stages.get(finding.stageId );
1062
+ if (!stage) return null;
1063
+ if (finding.rule === 'lowParallelism') {
1064
+ const stageDurationMs = (stage.completedAt ?? 0) - (stage.submittedAt ?? 0);
1065
+ const idleCoreMs =
1066
+ Math.max(0, ((finding.totalCores ) ?? 0) - (stage.taskCount ?? 0)) * stageDurationMs;
1067
+ // Real per-stage data (cores, task count, duration), no assumed constant.
1068
+ return costOnly('measured', { value: idleCoreMs, unit: 'coreMs' });
1069
+ }
1070
+ if (finding.rule === 'dataExplosion') {
1071
+ const excessBytes = Math.max(0, (stage.outputBytes ?? 0) - (stage.inputBytes ?? 0));
1072
+ // Measured input/output byte counts, no assumed constant.
1073
+ return costOnly('measured', { value: excessBytes, unit: 'bytes' });
1074
+ }
1075
+ if (finding.rule === 'taskStageSkew') {
1076
+ const totalCores = (finding.totalCores ) ?? 0;
1077
+ const taskCount = stage.taskCount ?? 0;
1078
+ // Cores idle during the straggler's tail, at achieved concurrency (not full cluster
1079
+ // capacity, which is lowParallelism's territory): this rule's trigger forces the
1080
+ // occupancy-clipped estimate to zero on every firing, so it's resourceOnly, not a wall-clock claim.
1081
+ const idleCoreMs =
1082
+ Math.max(0, Math.min(totalCores, taskCount) - 1) *
1083
+ Math.max(0, (stage.taskDurationMax ?? 0) - (stage.taskDurationP50 ?? 0));
1084
+ return costOnly('measured', { value: idleCoreMs, unit: 'coreMs' });
1085
+ }
1086
+ return null;
1087
+ },
1088
+ }),
1089
+ defineStageDetector({
1090
+ type: 'shuffle', order: 20, fixEffort: 'config', version: 1,
1091
+ emits: ['shuffle'],
905
1092
  docAnchor: '#bottleneck-shuffle',
906
1093
  // stageFloorPct: the 0.5% runtime floor. The shuffle claim is clipped to the stage, so on a
907
1094
  // shorter stage it graded info: 182 of 284 shuffle findings on the 14 real logs. The shuffle
908
1095
  // is still there on those stages; the floor is why they're dropped.
909
1096
  thresholds: { minBytes: 50 * MB, stageFloorPct: 0.005 },
910
- detect(
911
-
912
- stage ,
913
- ctx ,
914
- ) {
1097
+ detect(stage, ctx, thresholds) {
915
1098
  const bytes = stage.shuffleReadBytes;
916
- if (bytes <= this.thresholds.minBytes) return null;
917
- if (stageBelowRuntimeFloor(stage, ctx, this.thresholds.stageFloorPct)) return null;
1099
+ if (bytes <= thresholds.minBytes) return null;
1100
+ if (stageBelowRuntimeFloor(stage, ctx, thresholds.stageFloorPct)) return null;
918
1101
  return {
919
1102
  type: 'shuffle', stageId: stage.id,
920
1103
  impactBand: 'info',
@@ -922,23 +1105,30 @@ export const DETECTORS = [
922
1105
  recommendation: `${formatBytes(bytes)} shuffled in this stage: consider increasing spark.sql.shuffle.partitions or adding a broadcast join.`,
923
1106
  };
924
1107
  },
925
- },
926
- {
927
- type: 'partitionSizing', scope: 'stage', order: 22, fixEffort: 'config', version: 1,
1108
+ estimate(finding, ctx) {
1109
+ if (finding.stageId == null) return null;
1110
+ const stage = ctx.stages.get(finding.stageId);
1111
+ if (!stage) return null;
1112
+ const shuffleReadBytes = stage.shuffleReadBytes ?? 0;
1113
+ const modeledMs = (shuffleReadBytes / (SHUFFLE_THROUGHPUT_BPS * stageIoParallelism(stage))) * 1000;
1114
+ // The link model can't see whether the reads stalled the tasks: capped at the fetch wait the
1115
+ // tasks measured, the claim never exceeds what the stage spent blocked on the network.
1116
+ const measuredMs = fetchWaitWallClockMs(stage);
1117
+ const wasteMs = measuredMs == null ? modeledMs : Math.min(modeledMs, measuredMs);
1118
+ // rawWaste: the measured byte volume behind the modeled figure.
1119
+ return singleStageImpact(wasteMs, finding.stageId, ctx,
1120
+ measuredMs != null && measuredMs < modeledMs ? 'measured' : 'modeled', { value: shuffleReadBytes, unit: 'bytes' });
1121
+ },
1122
+ }),
1123
+ defineStageDetector({
1124
+ type: 'partitionSizing', order: 22, fixEffort: 'config', version: 1,
1125
+ emits: ['partitionSizing'],
928
1126
  docAnchor: '#bottleneck-partition-sizing',
929
1127
  thresholds: { skewRatio: 5, skewFloorBytes: 256 * MB, lowParTotalBytes: GB, lowParMaxTasks: 7, maxPartBytes: 5 * GB },
930
- detect(
931
-
932
-
933
-
934
-
935
-
936
-
937
- stage ,
938
- ) {
1128
+ detect(stage, _ctx, thresholds) {
939
1129
  const out = [];
940
1130
  const { shuffleReadP50: p50, shuffleReadMax: max, shuffleReadBytes: total, taskCount } = stage;
941
- if (max > this.thresholds.skewRatio * p50 && max > this.thresholds.skewFloorBytes) {
1131
+ if (max > thresholds.skewRatio * p50 && max > thresholds.skewFloorBytes) {
942
1132
  // p50 can be 0 (over half the shuffle partitions empty): a ratio against zero renders
943
1133
  // "Infinity×", so fall back to median-free phrasing.
944
1134
  const ratioText = p50 > 0
@@ -950,14 +1140,14 @@ export const DETECTORS = [
950
1140
  recommendation: `The largest shuffle partition (${formatBytes(max)}) is ${ratioText}: for join skew, enable AQE skew-join handling (spark.sql.adaptive.skewJoin.enabled); otherwise salt the key or repartition on a better key.`,
951
1141
  });
952
1142
  }
953
- if (total >= this.thresholds.lowParTotalBytes && taskCount <= this.thresholds.lowParMaxTasks) {
1143
+ if (total >= thresholds.lowParTotalBytes && taskCount <= thresholds.lowParMaxTasks) {
954
1144
  out.push({
955
1145
  type: 'partitionSizing', stageId: stage.id, impactBand: 'warning',
956
1146
  rule: 'lowShuffleParallelism', metric: 'taskCount', value: taskCount,
957
1147
  recommendation: `${Math.round(total / GB * 10) / 10} GB of shuffle spread over only ${taskCount} tasks: raise spark.sql.shuffle.partitions so each partition is smaller.`,
958
1148
  });
959
1149
  }
960
- if (max >= this.thresholds.maxPartBytes) {
1150
+ if (max >= thresholds.maxPartBytes) {
961
1151
  // Fixed 'critical': an OOM/crash-risk safety signal, not a time-waste one. Exempted in
962
1152
  // impact-band.ts's deriveImpactBand from the wall-clock-based overwrite every other
963
1153
  // finding here gets, so a long-running job can't demote an active crash risk to 'info'
@@ -970,19 +1160,46 @@ export const DETECTORS = [
970
1160
  }
971
1161
  return out;
972
1162
  },
973
- },
974
- {
975
- type: 'spill', scope: 'stage', order: 10, fixEffort: 'code', version: 1,
1163
+ estimate(finding, ctx) {
1164
+ if (finding.stageId == null) return null;
1165
+ const stage = ctx.stages.get(finding.stageId);
1166
+ if (!stage) return null;
1167
+ let wasteMs = 0;
1168
+ if (finding.rule === 'maxPartitionTooBig') {
1169
+ wasteMs = ((stage.shuffleReadMax ?? 0) / SHUFFLE_THROUGHPUT_BPS) * 1000;
1170
+ } else if (finding.rule === 'shufflePartitionSkew') {
1171
+ const delta = Math.max(0, (stage.shuffleReadMax ?? 0) - (stage.shuffleReadP50 ?? 0));
1172
+ wasteMs = (delta / SHUFFLE_THROUGHPUT_BPS) * 1000;
1173
+ } else if (finding.rule === 'lowShuffleParallelism') {
1174
+ const targetTaskCount = Math.ceil((stage.shuffleReadBytes ?? 0) / IDEAL_BYTES_PER_PARTITION_TASK);
1175
+ const taskCount = stage.taskCount ?? 0;
1176
+ if (targetTaskCount > taskCount && taskCount > 0) {
1177
+ const stageDurationMs = Math.max(0, (stage.completedAt ?? 0) - (stage.submittedAt ?? 0));
1178
+ // Too few shuffle partitions means each task processes more than the ideal bytes,
1179
+ // serializing work more partitions would run concurrently: the waste is that serialized
1180
+ // work, not the scheduling cost of tasks you'd add (adding tasks incurs overhead, recovers
1181
+ // nothing). Model the achievable duration at target parallelism by scaling down proportionally.
1182
+ wasteMs = stageDurationMs * (1 - taskCount / targetTaskCount);
1183
+ }
1184
+ } else {
1185
+ return null;
1186
+ }
1187
+ return singleStageImpact(wasteMs, finding.stageId, ctx, 'modeled', { value: wasteMs, unit: 'ms' });
1188
+ },
1189
+ }),
1190
+ defineStageDetector({
1191
+ type: 'spill', order: 10, fixEffort: 'code', version: 1,
1192
+ emits: ['spill'],
976
1193
  docAnchor: '#bottleneck-spill',
977
1194
  // stageFloorPct: the 0.5% runtime floor, as for shuffle (12 of 38 spill findings on the 14 real
978
1195
  // logs, all info). The spill is still there on those stages; the floor is why they're dropped.
979
1196
  thresholds: { singleTaskDiskGiB: 1, singleTaskMemGiB: 4, highDiskGiB: 1, highTaskDiskMB: 512, highMemGiB: 4, medDiskMB: 256, medMemGiB: 1, skewRatio: 5, skewDiskFloorMB: 128, skewMemFloorMB: 256, skewMinTasks: 10, stageFloorPct: 0.005 },
980
- detect( stage , ctx ) {
1197
+ detect(stage, ctx, thresholds) {
981
1198
  if (stage.memoryBytesSpilled === 0) return null;
982
- if (stageBelowRuntimeFloor(stage, ctx, this.thresholds.stageFloorPct)) return null;
1199
+ if (stageBelowRuntimeFloor(stage, ctx, thresholds.stageFloorPct)) return null;
983
1200
  const cls = stage.spillClassification;
984
1201
  const classified = cls === 'skew' || cls === 'volume';
985
- const mag = computeSpillMagnitude(stage, this.thresholds);
1202
+ const mag = computeSpillMagnitude(stage, thresholds);
986
1203
  const impactBand = 'warning';
987
1204
  return {
988
1205
  type: 'spill', stageId: stage.id, impactBand,
@@ -997,11 +1214,20 @@ export const DETECTORS = [
997
1214
  : `${formatBytes(stage.memoryBytesSpilled)} spilled: raise spark.sql.shuffle.partitions or increase executor memory.`,
998
1215
  };
999
1216
  },
1000
- },
1001
- {
1002
- type: 'gc', scope: 'stage', order: 50, fixEffort: 'config', version: 1,
1217
+ estimate(finding, ctx) {
1218
+ if (finding.stageId == null) return null;
1219
+ const stage = ctx.stages.get(finding.stageId);
1220
+ if (!stage) return null;
1221
+ const diskBytesSpilled = stage.diskBytesSpilled ?? 0;
1222
+ const wasteMs = (diskBytesSpilled / (SPILL_IO_THROUGHPUT_BPS * stageIoParallelism(stage))) * 1000;
1223
+ // Surfaces the number the formula uses: the displayed metric is memoryBytesSpilled, but disk spill costs the I/O time.
1224
+ return singleStageImpact(wasteMs, finding.stageId, ctx, 'modeled', { value: diskBytesSpilled, unit: 'bytes' });
1225
+ },
1226
+ }),
1227
+ defineStageDetector({
1228
+ type: 'gc', order: 50, fixEffort: 'config', version: 1,
1229
+ emits: ['gc'],
1003
1230
  docAnchor: '#bottleneck-gc',
1004
- validationRequired: 'This finding is gated by a 10-second minimum-runtime floor, our own noise floor for this metric.',
1005
1231
  thresholds: {
1006
1232
  warnPct100: 10,
1007
1233
  // Descending tier: ExecutorGcHeuristic, ported as-is.
@@ -1014,44 +1240,60 @@ export const DETECTORS = [
1014
1240
  // findings); the low-GC pattern is still true on those stages, the floor is why they're dropped.
1015
1241
  lowInfoFloorPct: 0.005,
1016
1242
  },
1017
- detect(
1018
-
1019
-
1020
-
1021
-
1022
- stage ,
1023
- ctx ,
1024
- ) {
1243
+ detect(stage, ctx, thresholds) {
1025
1244
  const pct = stage.gcPct;
1026
- if ((stage.executorRunTime ?? 0) >= this.thresholds.minRunTimeMs
1027
- && pct > this.thresholds.warnPct100) {
1245
+ if ((stage.executorRunTime ?? 0) >= thresholds.minRunTimeMs
1246
+ && pct > thresholds.warnPct100) {
1028
1247
  const value = Math.round(pct * 10) / 10;
1029
1248
  return {
1030
1249
  type: 'gc', stageId: stage.id,
1031
1250
  impactBand: 'warning',
1032
1251
  metric: 'gcPct', value,
1033
- confidence: gcConfidence(pct, this.thresholds, 'high'), validationRequired: this.validationRequired,
1252
+ confidence: gcConfidence(pct, thresholds, 'high'), validationRequired: gcValidation(thresholds.minRunTimeMs),
1034
1253
  recommendation: `GC consumed ${value}% of executor run time: reduce object creation, use primitive types, avoid UDFs, increase executor memory.`,
1035
1254
  };
1036
1255
  }
1037
1256
  // Low-GC (cost) branch: only for stages that ran long enough to be meaningful.
1038
- if ((stage.executorRunTime ?? 0) >= this.thresholds.minRunTimeMs
1039
- && pct < this.thresholds.lowInfoPct100
1040
- && !stageBelowRuntimeFloor(stage, ctx, this.thresholds.lowInfoFloorPct)) {
1257
+ if ((stage.executorRunTime ?? 0) >= thresholds.minRunTimeMs
1258
+ && pct < thresholds.lowInfoPct100
1259
+ && !stageBelowRuntimeFloor(stage, ctx, thresholds.lowInfoFloorPct)) {
1041
1260
  const value = Math.round(pct * 10) / 10;
1042
1261
  return {
1043
1262
  type: 'gc', stageId: stage.id, direction: 'low',
1044
1263
  impactBand: 'info',
1045
1264
  metric: 'gcPct', value,
1046
- confidence: gcConfidence(pct, this.thresholds, 'low'), validationRequired: this.validationRequired,
1265
+ confidence: gcConfidence(pct, thresholds, 'low'), validationRequired: gcValidation(thresholds.minRunTimeMs),
1047
1266
  recommendation: `GC consumed only ${value}% of executor run time: memory may be over-provisioned; consider reducing spark.executor.memory for cost savings.`,
1048
1267
  };
1049
1268
  }
1050
1269
  return null;
1051
1270
  },
1052
- },
1053
- {
1054
- type: 'slowHost', scope: 'stage', order: 60, fixEffort: 'config', version: 1,
1271
+ estimate(finding, ctx) {
1272
+ // The low-GC branch is an over-provisioning signal whose fix (less executor memory) raises
1273
+ // GC rather than recovering it: the stage's GC time is no saving there, so no waste model.
1274
+ if (finding.direction === 'low') return costOnly('none');
1275
+ if (finding.stageId == null) return null;
1276
+ const stage = ctx.stages.get(finding.stageId);
1277
+ if (!stage) return null;
1278
+ const stageDurationMs = (stage.completedAt ?? 0) - (stage.submittedAt ?? 0);
1279
+ const executorRunTime = stage.executorRunTime ?? 0;
1280
+ const jvmGCTime = stage.jvmGCTime ?? 0;
1281
+ // The raw cross-task core-time sum, before any conversion: the one figure here
1282
+ // that is straight from the log rather than modeled.
1283
+ const rawWaste = { value: jvmGCTime, unit: 'coreMs' } ;
1284
+ if (executorRunTime <= 0 || stageDurationMs <= 0) {
1285
+ return costOnly('modeled', rawWaste);
1286
+ }
1287
+ const avgConcurrency = executorRunTime / stageDurationMs;
1288
+ // jvmGCTime is a cross-task core-time sum (same shape as executorRunTime); dividing by the
1289
+ // stage's average concurrency converts it to an approximate wall-clock figure. Modeled, not exact.
1290
+ const wasteMs = jvmGCTime / avgConcurrency;
1291
+ return singleStageImpact(wasteMs, finding.stageId, ctx, 'modeled', rawWaste);
1292
+ },
1293
+ }),
1294
+ defineStageDetector({
1295
+ type: 'slowHost', order: 60, fixEffort: 'config', version: 1,
1296
+ emits: ['slowHost'],
1055
1297
  docAnchor: '#bottleneck-slow-host',
1056
1298
  thresholds: {
1057
1299
  minHosts: 3, minTasks: 15, ratioWarn: 2.0, minShare: 0.20, shareWarn: 0.75, taskShareWarn: 0.50, ratioTiers: [1.33, 1.78, 3.16, 10],
@@ -1067,23 +1309,13 @@ export const DETECTORS = [
1067
1309
  // floor is why they're dropped.
1068
1310
  stageFloorPct: 0.005,
1069
1311
  },
1070
- detect(
1071
-
1072
-
1073
-
1074
-
1075
-
1076
-
1077
-
1078
- stage ,
1079
- ctx ,
1080
- ) {
1312
+ detect(stage, ctx, thresholds) {
1081
1313
  const hosts = stage.hostStats ?? [];
1082
1314
  const execs0 = stage.executorStats ?? [];
1083
- if ((hosts.length < this.thresholds.minHosts && execs0.length < this.thresholds.minHosts) || stage.taskCount < this.thresholds.minTasks) return null;
1084
- if (stageBelowRuntimeFloor(stage, ctx, this.thresholds.stageFloorPct)) return null;
1315
+ if ((hosts.length < thresholds.minHosts && execs0.length < thresholds.minHosts) || stage.taskCount < thresholds.minTasks) return null;
1316
+ if (stageBelowRuntimeFloor(stage, ctx, thresholds.stageFloorPct)) return null;
1085
1317
  const out = [];
1086
- if (hosts.length >= this.thresholds.minHosts) {
1318
+ if (hosts.length >= thresholds.minHosts) {
1087
1319
  const means = hosts.map(h => ({ host: h.host, taskCount: h.taskCount, mean: h.totalDuration / h.taskCount }));
1088
1320
  const sorted = [...means].map(h => h.mean).sort((a, b) => a - b);
1089
1321
  const overallMedian = sorted[Math.floor(sorted.length / 2)];
@@ -1091,7 +1323,7 @@ export const DETECTORS = [
1091
1323
  for (const h of means) {
1092
1324
  const ratio = h.mean / overallMedian;
1093
1325
  const share = h.taskCount / stage.taskCount;
1094
- if (ratio < this.thresholds.ratioWarn || share < this.thresholds.minShare || h.mean < this.thresholds.floorMs) continue;
1326
+ if (ratio < thresholds.ratioWarn || share < thresholds.minShare || h.mean < thresholds.floorMs) continue;
1095
1327
  out.push({
1096
1328
  type: 'slowHost', stageId: stage.id,
1097
1329
  impactBand: 'warning',
@@ -1108,7 +1340,7 @@ export const DETECTORS = [
1108
1340
  for (const h of hosts) {
1109
1341
  const durationShare = h.totalDuration / totalDuration;
1110
1342
  const taskShare = h.taskCount / stage.taskCount;
1111
- if (durationShare >= this.thresholds.shareWarn && taskShare >= this.thresholds.taskShareWarn) {
1343
+ if (durationShare >= thresholds.shareWarn && taskShare >= thresholds.taskShareWarn) {
1112
1344
  out.push({
1113
1345
  type: 'slowHost', stageId: stage.id, impactBand: 'warning',
1114
1346
  variant: 'durationShare',
@@ -1123,11 +1355,11 @@ export const DETECTORS = [
1123
1355
  }
1124
1356
  }
1125
1357
  const execs = stage.executorStats ?? [];
1126
- const tiers = this.thresholds.ratioTiers;
1358
+ const tiers = thresholds.ratioTiers;
1127
1359
  const impactBandFor = (r ) =>
1128
1360
  r >= tiers[3] ? 'critical' : (r >= tiers[1] ? 'warning' : (r >= tiers[0] ? 'info' : null));
1129
- const floorMs = this.thresholds.floorMs, floorBytes = this.thresholds.floorBytes;
1130
- const dims = [
1361
+ const floorMs = thresholds.floorMs, floorBytes = thresholds.floorBytes;
1362
+ const dims = [
1131
1363
  { dimension: 'taskTime', floor: floorMs, samples: execs.filter(e => e.taskCount > 0).map(e => ({ key: e.executorId, value: e.totalDuration / e.taskCount })) },
1132
1364
  { dimension: 'inputBytes', floor: floorBytes, samples: execs.map(e => ({ key: e.executorId, value: e.inputBytes ?? 0 })) },
1133
1365
  { dimension: 'shuffleBytes', floor: floorBytes, samples: execs.map(e => ({ key: e.executorId, value: (e.shuffleReadBytes ?? 0) + (e.shuffleWriteBytes ?? 0) })) },
@@ -1160,26 +1392,41 @@ export const DETECTORS = [
1160
1392
  }
1161
1393
  return out;
1162
1394
  },
1163
- },
1164
- {
1165
- type: 'stageSlowness', scope: 'stage', order: 65, fixEffort: 'code', version: 2,
1395
+ estimate(finding, ctx) {
1396
+ // Three duration-based shapes, each carrying its absolute-ms figure under a different field
1397
+ // (`value` is always a ratio/share, never ms): the per-host mean branch (discriminated by
1398
+ // `metric`), the duration-share branch (`variant`), and the multiDim taskTime dimension. Every
1399
+ // byte-based multiDim dimension has no absolute figure today, so it stays informational.
1400
+ const absoluteMs =
1401
+ finding.metric === 'hostMeanRatio' || finding.variant === 'durationShare'
1402
+ ? (finding.hostMeanMs )
1403
+ : finding.variant === 'multiDim' && finding.dimension === 'taskTime'
1404
+ ? (finding.execMaxValue )
1405
+ : null;
1406
+ if (absoluteMs == null) {
1407
+ return costOnly('none'); // byte-based multiDim dims: no absolute figure today, no model applied
1408
+ }
1409
+ if (finding.stageId == null) return null;
1410
+ const stage = ctx.stages.get(finding.stageId);
1411
+ if (!stage) return null;
1412
+ const wasteMs = Math.max(0, absoluteMs - (stage.taskDurationP50 ?? 0));
1413
+ return singleStageImpact(wasteMs, finding.stageId, ctx, 'measured', { value: wasteMs, unit: 'ms' });
1414
+ },
1415
+ }),
1416
+ defineStageDetector({
1417
+ type: 'stageSlowness', order: 65, fixEffort: 'code', version: 2,
1418
+ emits: ['stageSlowness'],
1166
1419
  docAnchor: '#bottleneck-stage-slowness',
1167
1420
  thresholds: { infoMin: 15 },
1168
- // Cross-detector suppression (see "Detector contract" in detector-contract.md). Requires this
1169
- // entry to be declared AFTER slowHost in DETECTORS so slowHost findings are already in `out`.
1170
- suppressWhen(finding, out) {
1171
- return out.some(o => o.type === 'slowHost' && o.stageId === finding.stageId);
1172
- },
1173
- detect(
1174
-
1175
- stage ,
1176
- ) {
1421
+ // A stage slowHost already explains needs no generic "this stage is slow" finding on top.
1422
+ suppressedBy: 'slowHost',
1423
+ detect(stage, _ctx, thresholds) {
1177
1424
  // Basis is real wall-clock stage duration, not per-executor average; the impact-estimator
1178
1425
  // formula reuses this exact stageDurationMs computation.
1179
1426
  const stageDurationMs = (stage.completedAt ?? 0) - (stage.submittedAt ?? 0);
1180
1427
  if (!(stageDurationMs > 0)) return null;
1181
1428
  const durationMinutes = stageDurationMs / 60000;
1182
- const t = this.thresholds;
1429
+ const t = thresholds;
1183
1430
  const impactBand = durationMinutes >= t.infoMin ? 'info' : null;
1184
1431
  if (!impactBand) return null;
1185
1432
  const value = Math.round(durationMinutes * 10) / 10;
@@ -1189,36 +1436,57 @@ export const DETECTORS = [
1189
1436
  recommendation: `This stage ran ${value} minutes with no more specific cause flagged: often a partition-count problem, raise parallelism via spark.sql.shuffle.partitions or spark.default.parallelism, or check for a large per-task data volume driving heavy shuffle and spill.`,
1190
1437
  };
1191
1438
  },
1192
- },
1193
- {
1194
- type: 'stageFailed', scope: 'stage', order: 42, fixEffort: 'code', version: 1,
1439
+ estimate(finding, ctx) {
1440
+ if (finding.stageId == null) return null;
1441
+ const stage = ctx.stages.get(finding.stageId);
1442
+ if (!stage) return null;
1443
+ // The recommended fix is more partitions, which only helps a stage that ran fewer tasks than
1444
+ // the cluster has cores: the time its tasks were running could then spread over up to
1445
+ // totalCores (lowShuffleParallelism's shape). Time the stage sat open with no task running
1446
+ // is queueing no partition count recovers. Splitting partitions splits the longest task
1447
+ // too, hence TAIL_CLAIM's post-fix floor. Unknown cluster size: no defensible figure.
1448
+ if (ctx.totalCores <= 0) return costOnly('modeled');
1449
+ // A stage that read no input and no shuffle, its tasks idle waiting on an external system,
1450
+ // gains nothing from more partitions (a 1-task JDBC count stage open 27 minutes on 5s of CPU
1451
+ // was claimed 99% recoverable): claim 0. Stages that read bytes keep their claim.
1452
+ const readBytes = (stage.inputBytes ?? 0) + (stage.shuffleReadBytes ?? 0);
1453
+ const activeMs = typeof stage.taskActiveMs === 'number'
1454
+ ? stage.taskActiveMs
1455
+ : Math.max(0, (stage.completedAt ?? 0) - (stage.submittedAt ?? 0));
1456
+ const taskCount = stage.taskCount ?? 0;
1457
+ const wasteMs = readBytes <= 0 && tasksMostlyIdle(stage) ? 0 : activeMs * Math.max(0, 1 - taskCount / ctx.totalCores);
1458
+ return singleStageImpact(wasteMs, finding.stageId, ctx, 'modeled', { value: wasteMs, unit: 'ms' }, TAIL_CLAIM);
1459
+ },
1460
+ }),
1461
+ defineStageDetector({
1462
+ type: 'stageFailed', order: 42, fixEffort: 'code', version: 1,
1463
+ emits: ['stageFailed'],
1195
1464
  docAnchor: '#bottleneck-failures',
1196
1465
  thresholds: {},
1197
- detect(stage ) {
1466
+ detect(stage) {
1198
1467
  if (stage.stageFailureReason == null) return null;
1199
1468
  return {
1200
1469
  type: 'stageFailed', stageId: stage.id, impactBand: 'critical',
1201
1470
  variant: 'stageFailure',
1202
- metric: 'stageFailureReason', value: stage.stageFailureReason,
1471
+ metric: 'stageFailureReason', valueText: stage.stageFailureReason,
1203
1472
  numTasks: stage.taskCount,
1204
1473
  memoryBytesSpilled: stage.memoryBytesSpilled,
1205
1474
  failedTaskDetails: stage.failedTaskSamples ?? [],
1206
1475
  recommendation: `This stage attempt failed outright. Inspect the driver log for the failure reason and the job that triggered it.`,
1207
1476
  };
1208
1477
  },
1209
- },
1210
- {
1211
- type: 'failures', scope: 'stage', order: 40, fixEffort: 'code', version: 2,
1478
+ estimate: noWasteModel,
1479
+ }),
1480
+ defineStageDetector({
1481
+ type: 'failures', order: 40, fixEffort: 'code', version: 2,
1482
+ emits: ['failures'],
1212
1483
  docAnchor: '#bottleneck-failures',
1213
1484
  thresholds: { minTasks: 10, warnRate: 0.05, critRate: 0.20 },
1214
- detect(
1215
-
1216
- stage ,
1217
- ) {
1218
- if (stage.taskCount < this.thresholds.minTasks) return null;
1485
+ detect(stage, _ctx, thresholds) {
1486
+ if (stage.taskCount < thresholds.minTasks) return null;
1219
1487
  if (!stage.failedTasks) return null;
1220
1488
  const failureRate = stage.failedTasks / stage.taskCount;
1221
- if (failureRate <= this.thresholds.warnRate) return null;
1489
+ if (failureRate <= thresholds.warnRate) return null;
1222
1490
  const value = Math.round(failureRate * 1000) / 10;
1223
1491
  const dominantReason = pickDominantReason(stage.failureReasons);
1224
1492
  // Groups arrive most frequent first. Name the dominant error from the largest group under the
@@ -1230,7 +1498,7 @@ export const DETECTORS = [
1230
1498
  const groupedTasks = failureGroups.reduce((sum, g) => sum + g.count, 0);
1231
1499
  return {
1232
1500
  type: 'failures', stageId: stage.id,
1233
- impactBand: failureRate > this.thresholds.critRate ? 'critical' : 'warning',
1501
+ impactBand: failureRate > thresholds.critRate ? 'critical' : 'warning',
1234
1502
  metric: 'failureRate', value,
1235
1503
  failedTasks: stage.failedTasks,
1236
1504
  dominantReason,
@@ -1242,12 +1510,14 @@ export const DETECTORS = [
1242
1510
  recommendation: `${value}% of tasks failed${dominantError ? ` (dominant error: ${dominantError})` : ''}: investigate driver logs for executor instability or data-driven errors.`,
1243
1511
  };
1244
1512
  },
1245
- },
1246
- {
1247
- type: 'straggler', scope: 'stage', order: 70, fixEffort: 'code', version: 1,
1513
+ estimate: noWasteModel,
1514
+ }),
1515
+ defineStageDetector({
1516
+ type: 'straggler', order: 70, fixEffort: 'code', version: 1,
1517
+ emits: ['straggler'],
1248
1518
  docAnchor: '#bottleneck-straggler',
1249
- // floorPctWarn/floorPctCrit are re-exported as STRAGGLER_FLOOR_PCT_WARN/CRIT and reused as
1250
- // impact-band.ts's global noise floor: keep the two in sync.
1519
+ // floorPctWarn/floorPctCrit default to impact-band.ts's run-wide noise floor, so a tail this
1520
+ // gate admits at its warn floor grades at least warning there too.
1251
1521
  // shareWarnAtFloor: in a large stage, the few stragglers that gate it for tens of seconds can be
1252
1522
  // only 2.5-5% of its tasks. Scored against a task-level replay of every stage on 14 real logs
1253
1523
  // (recoverable = replay with each task over 4x P50 capped at P50), admitting 2.5-5% shares
@@ -1257,40 +1527,27 @@ export const DETECTORS = [
1257
1527
  // than the stage's own duration, so every finding there graded info. On the 14 real logs that
1258
1528
  // was 671 of 753 straggler findings, none above info; the slow tail is still real on those
1259
1529
  // stages, the floor is why they're dropped.
1260
- thresholds: { minTasks: 10, shareWarn: 0.05, shareWarnAtFloor: 0.025, warnPct: 0.10, critPct: 0.20, floorPctWarn: STRAGGLER_FLOOR_PCT_WARN, floorPctCrit: STRAGGLER_FLOOR_PCT_CRIT },
1261
- detect(
1262
-
1263
-
1264
-
1265
-
1266
-
1267
-
1268
- stage ,
1269
- ctx ,
1270
- ) {
1271
- if (stage.taskCount < this.thresholds.minTasks) return null;
1272
- const appDurationMs = computeAppDurationMs(ctx);
1273
- if (stageBelowRuntimeFloor(stage, ctx, this.thresholds.floorPctWarn)) return null;
1530
+ thresholds: { minTasks: 10, shareWarn: 0.05, shareWarnAtFloor: 0.025, warnPct: 0.10, critPct: 0.20, floorPctWarn: IMPACT_FLOOR_PCT_WARN, floorPctCrit: IMPACT_FLOOR_PCT_CRIT },
1531
+ detect(stage, ctx, thresholds) {
1532
+ if (stage.taskCount < thresholds.minTasks) return null;
1533
+ const runMs = appDurationMs(ctx.app);
1534
+ if (stageBelowRuntimeFloor(stage, ctx, thresholds.floorPctWarn)) return null;
1274
1535
  const stragglerShare = (stage.stragglerCount ?? 0) / stage.taskCount;
1275
1536
  const useSpeculative = (stage.speculativeTasks ?? 0) > 0;
1276
- if (!useSpeculative && stragglerShare <= this.thresholds.shareWarnAtFloor) return null;
1537
+ if (!useSpeculative && stragglerShare <= thresholds.shareWarnAtFloor) return null;
1277
1538
  const speculativeShare = useSpeculative ? stage.speculativeTasks / stage.taskCount : 0;
1278
- // Same absolute delta impact-estimator.ts's straggler/stageShape case reports as savings: a
1279
- // high straggler/speculative share on a stage whose tasks barely vary models near-zero
1280
- // savings, so it must not outrank 'info'. Clipped the same way before the floor check.
1281
- const longestTaskAfterFixMs = stragglerFixLongestTaskMs(stage);
1282
- const singleDelta = Math.max(0, stage.taskDurationMax - longestTaskAfterFixMs);
1283
- const wasteMs = tailRecoveryMs(stage, singleDelta);
1284
- const floorWasteMs = clippedWasteMs(wasteMs, stage.id, ctx, tailRemovedWorkMs(stage, singleDelta), longestTaskAfterFixMs);
1285
- const meetsWarnFloor = meetsRuntimeFloor(floorWasteMs, appDurationMs, this.thresholds.floorPctWarn);
1286
- const meetsCritFloor = meetsRuntimeFloor(floorWasteMs, appDurationMs, this.thresholds.floorPctCrit);
1539
+ // The claim estimate() reports as savings: a high straggler/speculative share on a stage
1540
+ // whose tasks barely vary models near-zero savings, so it must not outrank 'info'.
1541
+ const floorWasteMs = tailClaimFloorMs(stragglerTailClaim(stage), stage.id, ctx);
1542
+ const meetsWarnFloor = meetsRuntimeFloor(floorWasteMs, runMs, thresholds.floorPctWarn);
1543
+ const meetsCritFloor = meetsRuntimeFloor(floorWasteMs, runMs, thresholds.floorPctCrit);
1287
1544
  // The lower gate needs positive evidence the tail matters: meetsRuntimeFloor passes by
1288
1545
  // default when the app's duration is unknown (an incomplete run), which isn't that.
1289
- const stragglerShareFires = stragglerShare > this.thresholds.shareWarn
1290
- || (stragglerShare > this.thresholds.shareWarnAtFloor && appDurationMs != null && meetsWarnFloor);
1546
+ const stragglerShareFires = stragglerShare > thresholds.shareWarn
1547
+ || (stragglerShare > thresholds.shareWarnAtFloor && runMs != null && meetsWarnFloor);
1291
1548
  if (!useSpeculative && !stragglerShareFires) return null;
1292
- const speculativeTier = speculativeShare >= this.thresholds.critPct && meetsCritFloor ? 'critical'
1293
- : speculativeShare >= this.thresholds.warnPct && meetsWarnFloor ? 'warning' : 'info';
1549
+ const speculativeTier = speculativeShare >= thresholds.critPct && meetsCritFloor ? 'critical'
1550
+ : speculativeShare >= thresholds.warnPct && meetsWarnFloor ? 'warning' : 'info';
1294
1551
  // Straggler share has no dedicated critical tier per detector-contract.md; only warning.
1295
1552
  const stragglerTier = stragglerShareFires && meetsWarnFloor ? 'warning' : 'info';
1296
1553
  // Fixed fallback: overwritten by deriveImpactBand when this finding gets a real wallClock
@@ -1312,46 +1569,53 @@ export const DETECTORS = [
1312
1569
  speculativeTasks: stage.speculativeTasks ?? 0,
1313
1570
  stragglerCount: stage.stragglerCount ?? 0,
1314
1571
  confidence: useSpeculativeMetric
1315
- ? stragglerConfidence(speculativeShare, this.thresholds.warnPct, this.thresholds.critPct)
1316
- : stragglerConfidence(stragglerShare, this.thresholds.shareWarn, this.thresholds.critPct),
1317
- validationRequired: 'This finding is gated by 0.5%/2% runtime-floor thresholds, our own noise floor for this metric.',
1572
+ ? stragglerConfidence(speculativeShare, thresholds.warnPct, thresholds.critPct)
1573
+ : stragglerConfidence(stragglerShare, thresholds.shareWarn, thresholds.critPct),
1574
+ validationRequired: `This finding is gated by ${shareLabel(thresholds.floorPctWarn)}/${shareLabel(thresholds.floorPctCrit)} runtime-floor thresholds, our own noise floor for this metric.`,
1318
1575
  recommendation: `${detail}: rule out a GC pause or a slow shuffle fetch before assuming a hardware issue; if a skewed key is the real cause, that's a candidate for AQE's skew-join handling.`,
1319
1576
  };
1320
1577
  },
1321
- },
1322
- {
1323
- type: 'speculationWaste', scope: 'stage', order: 71, fixEffort: 'config', version: 1,
1578
+ estimate(finding, ctx) {
1579
+ if (finding.stageId == null) return null;
1580
+ const stage = ctx.stages.get(finding.stageId);
1581
+ if (!stage) return null;
1582
+ return tailClaimImpact(stragglerTailClaim(stage), finding.stageId, ctx);
1583
+ },
1584
+ }),
1585
+ defineStageDetector({
1586
+ type: 'speculationWaste', order: 71, fixEffort: 'config', version: 1,
1587
+ emits: ['speculationWaste'],
1324
1588
  docAnchor: '#bottleneck-speculation-waste',
1325
1589
  thresholds: { minWasted: 5, minWasteMs: 60000 },
1326
- detect(
1327
-
1328
-
1329
-
1330
- stage ,
1331
- ) {
1590
+ detect(stage, _ctx, thresholds) {
1332
1591
  const wasted = stage.speculationWastedAttempts ?? 0;
1333
1592
  const wastedMs = stage.speculationWasteMs ?? 0;
1334
- if (wasted < this.thresholds.minWasted || wastedMs < this.thresholds.minWasteMs) return null;
1593
+ if (wasted < thresholds.minWasted || wastedMs < thresholds.minWasteMs) return null;
1335
1594
  return {
1336
1595
  type: 'speculationWaste', stageId: stage.id,
1337
1596
  impactBand: 'warning',
1338
1597
  metric: 'speculationWasteMs', value: wastedMs,
1339
- confidence: speculationWasteConfidence(wastedMs, this.thresholds.minWasteMs),
1598
+ confidence: speculationWasteConfidence(wastedMs, thresholds.minWasteMs),
1340
1599
  recommendation: `Speculative execution discarded ${Math.round(wastedMs / 1000)}s of executor time in this stage; if task durations are naturally variable rather than genuine stragglers, consider tuning spark.speculation.multiplier/quantile.`,
1341
1600
  };
1342
1601
  },
1343
- },
1344
- {
1345
- type: 'retryWaste', scope: 'stage', order: 45, fixEffort: 'code', version: 1,
1602
+ estimate(finding, ctx) {
1603
+ if (finding.stageId == null) return null;
1604
+ const stage = ctx.stages.get(finding.stageId);
1605
+ if (!stage) return null;
1606
+ const wasteMs = (stage.speculationWasteMs ) ?? 0;
1607
+ return singleStageImpact(wasteMs, finding.stageId, ctx, 'measured', { value: wasteMs, unit: 'ms' });
1608
+ },
1609
+ }),
1610
+ defineStageDetector({
1611
+ type: 'retryWaste', order: 45, fixEffort: 'code', version: 1,
1612
+ emits: ['retryWaste'],
1346
1613
  docAnchor: '#bottleneck-retry-waste',
1347
1614
  thresholds: { minWasted: 3, minWasteMs: 30000 },
1348
- detect(
1349
-
1350
- stage ,
1351
- ) {
1615
+ detect(stage, _ctx, thresholds) {
1352
1616
  const wasted = stage.wastedAttempts ?? 0;
1353
1617
  const wastedMs = stage.retryWasteMs ?? 0;
1354
- if (wasted < this.thresholds.minWasted || wastedMs < this.thresholds.minWasteMs) return null;
1618
+ if (wasted < thresholds.minWasted || wastedMs < thresholds.minWasteMs) return null;
1355
1619
  return {
1356
1620
  type: 'retryWaste', stageId: stage.id,
1357
1621
  impactBand: 'warning',
@@ -1363,22 +1627,29 @@ export const DETECTORS = [
1363
1627
  extended: `${wasted} task attempts were superseded by a later retry, wasting ${Math.round(wastedMs / 1000)}s of executor time. Common causes: executor loss (OOM-kill, node death) or shuffle FetchFailed forcing a stage-map recompute. Check driver logs for the dominant reason (see the Failures widget) even if the final failure rate looks low; retries hide the true cost.`,
1364
1628
  };
1365
1629
  },
1366
- },
1367
- {
1368
- type: 'tinyTask', scope: 'stage', order: 80, fixEffort: 'code', version: 1,
1630
+ estimate(finding, ctx) {
1631
+ // The waste figure lives on the Stage, not the Finding: detect() only re-publishes it as metric/value.
1632
+ if (finding.stageId == null) return null;
1633
+ const stage = ctx.stages.get(finding.stageId);
1634
+ if (!stage) return null;
1635
+ const wasteMs = (stage.retryWasteMs ) ?? 0;
1636
+ const wallClockMs = retryWallClockMs(stage);
1637
+ return singleStageImpact(wallClockMs, finding.stageId, ctx,
1638
+ wallClockMs === wasteMs ? 'measured' : 'modeled', { value: wasteMs, unit: 'ms' });
1639
+ },
1640
+ }),
1641
+ defineStageDetector({
1642
+ type: 'tinyTask', order: 80, fixEffort: 'code', version: 1,
1643
+ emits: ['tinyTask'],
1369
1644
  docAnchor: '#bottleneck-tiny-tasks',
1370
1645
  // stageFloorPct: the tiered detectors' 0.5% runtime floor. Coalescing can't save more than the
1371
1646
  // stage's own duration, so on a shorter stage every finding graded info: on the 14 real logs
1372
1647
  // 132 of 151 tinyTask findings. The tasks are still tiny there; the floor is why they're dropped.
1373
1648
  thresholds: { minTasks: 100, maxP50: 500, maxP95: 1000, stageFloorPct: 0.005 },
1374
- detect(
1375
-
1376
- stage ,
1377
- ctx ,
1378
- ) {
1379
- if (stage.taskCount < this.thresholds.minTasks) return null;
1380
- if (stageBelowRuntimeFloor(stage, ctx, this.thresholds.stageFloorPct)) return null;
1381
- if (stage.taskDurationP50 > this.thresholds.maxP50 || stage.taskDurationP95 > this.thresholds.maxP95) return null;
1649
+ detect(stage, ctx, thresholds) {
1650
+ if (stage.taskCount < thresholds.minTasks) return null;
1651
+ if (stageBelowRuntimeFloor(stage, ctx, thresholds.stageFloorPct)) return null;
1652
+ if (stage.taskDurationP50 > thresholds.maxP50 || stage.taskDurationP95 > thresholds.maxP95) return null;
1382
1653
  const coalesceTo = Math.max(1, Math.round(stage.taskCount / 10));
1383
1654
  const fix = stage.shuffleReadBytes > 0
1384
1655
  ? `lower spark.sql.shuffle.partitions or .coalesce(${coalesceTo})`
@@ -1389,30 +1660,46 @@ export const DETECTORS = [
1389
1660
  recommendation: `Many small tasks (${stage.taskCount}, P50 ${Math.round(stage.taskDurationP50)}ms): scheduler overhead may dominate. Try ${fix}.`,
1390
1661
  };
1391
1662
  },
1392
- },
1393
- {
1663
+ estimate(finding, ctx) {
1664
+ if (finding.stageId == null) return null;
1665
+ const stage = ctx.stages.get(finding.stageId);
1666
+ if (!stage) return null;
1667
+ const taskCount = stage.taskCount ?? 0;
1668
+ const excessTaskCount = Math.max(0, taskCount - Math.round(taskCount / 10));
1669
+ const measured = measuredTaskOverhead(stage);
1670
+ if (measured) {
1671
+ // Coalescing to a tenth of the tasks removes the excess tasks' per-task overhead: core
1672
+ // time spent in parallel, so wall-clock at the stage's achieved concurrency (floored at
1673
+ // 1: a mostly-idle stage can't save more wall-clock than the task time it removes).
1674
+ const wasteMs = (excessTaskCount * measured.perTaskMs) / Math.max(1, measured.concurrency);
1675
+ return singleStageImpact(wasteMs, finding.stageId, ctx, 'measured', { value: wasteMs, unit: 'ms' });
1676
+ }
1677
+ const wasteMs = excessTaskCount * TASK_SCHEDULING_OVERHEAD_MS;
1678
+ return singleStageImpact(wasteMs, finding.stageId, ctx, 'modeled', { value: wasteMs, unit: 'ms' });
1679
+ },
1680
+ }),
1681
+ defineAppDetector({
1394
1682
  // No docAnchor: the upstream spark-tuning-reference docs have no section for this
1395
1683
  // tool-specific "capture stopped early" signal.
1396
- type: 'incompleteRun', scope: 'app', order: 5, fixEffort: 'code', version: 1,
1684
+ type: 'incompleteRun', order: 5, fixEffort: 'code', version: 1,
1685
+ emits: ['incompleteRun'],
1397
1686
  thresholds: {},
1398
- recommendation: 'This event log never recorded an ApplicationEnd event: the capture stopped before the run finished (an in-flight job, a rotated-away log, or a cut-short capture). Findings and metrics elsewhere on this board reflect only what was captured up to that point, not the full run.',
1399
- detect( ctx ) {
1687
+ detect(ctx) {
1400
1688
  if (!ctx.app || ctx.app.startTime == null || ctx.app.endTime != null) return null;
1401
1689
  return {
1402
1690
  type: 'incompleteRun', stageId: null, impactBand: 'warning',
1403
- metric: 'applicationEnd', value: 'missing',
1404
- recommendation: this.recommendation,
1691
+ metric: 'applicationEnd', valueText: 'missing',
1692
+ recommendation: INCOMPLETE_RUN_RECOMMENDATION,
1405
1693
  };
1406
1694
  },
1407
- },
1408
- {
1409
- type: 'coldStart', scope: 'app', order: 90, fixEffort: 'code', version: 1,
1695
+ estimate: noWasteModel,
1696
+ }),
1697
+ defineAppDetector({
1698
+ type: 'coldStart', order: 90, fixEffort: 'code', version: 1,
1699
+ emits: ['coldStart'],
1410
1700
  docAnchor: '#bottleneck-cold-start',
1411
1701
  thresholds: { gapSeconds: 30 },
1412
- detect(
1413
-
1414
- ctx ,
1415
- ) {
1702
+ detect(ctx, thresholds) {
1416
1703
  const { app, stages, executorsAdded, executorsRemoved } = ctx;
1417
1704
  // Nullish (not falsy) check: a literal startTime:0 must not be treated as "missing".
1418
1705
  if (!app || app.startTime == null || stages.size === 0) return null;
@@ -1443,7 +1730,7 @@ export const DETECTORS = [
1443
1730
  }
1444
1731
  if (!Number.isFinite(firstExecutorAdded)) return null;
1445
1732
  const gapSeconds = (firstExecutorAdded - firstStageSubmitted) / 1000;
1446
- if (gapSeconds <= this.thresholds.gapSeconds) return null;
1733
+ if (gapSeconds <= thresholds.gapSeconds) return null;
1447
1734
  const value = Math.round(gapSeconds);
1448
1735
  return {
1449
1736
  type: 'coldStart', stageId: null, impactBand: 'warning',
@@ -1451,15 +1738,21 @@ export const DETECTORS = [
1451
1738
  recommendation: `The first stage waited ${value}s for an executor to become available: keep a warm pool of idle executors, or if using dynamic allocation, raise the minimum/initial executor count so it doesn't scale up from zero.`,
1452
1739
  };
1453
1740
  },
1454
- },
1455
- {
1456
- type: 'utilization', scope: 'app', order: 100, fixEffort: 'config', version: 1,
1741
+ estimate(finding) {
1742
+ // detect() reports the gap as `metric: 'startupGapSeconds', value: <seconds>`.
1743
+ if (typeof finding.value !== 'number') return null;
1744
+ const wasteMs = finding.value * 1000;
1745
+ // Time before any task starts can never overlap any stage; a genuine unclipped point estimate,
1746
+ // not tied to any stage's gate (coldStart is app-scoped, stageId: null).
1747
+ return { basis: 'serial', wallClock: { low: wasteMs, high: wasteMs }, estimateMethod: 'measured' };
1748
+ },
1749
+ }),
1750
+ defineAppDetector({
1751
+ type: 'utilization', order: 100, fixEffort: 'config', version: 1,
1752
+ emits: ['utilization'],
1457
1753
  docAnchor: '#bottleneck-utilization',
1458
1754
  thresholds: { minUtil: 0.60 },
1459
- detect(
1460
-
1461
- ctx ,
1462
- ) {
1755
+ detect(ctx, thresholds) {
1463
1756
  const { app, executorsAdded, executorsRemoved, runAggregates } = ctx;
1464
1757
  // Nullish (not falsy) check: a literal startTime:0 must not be treated as "missing".
1465
1758
  if (!app || executorsAdded.length === 0 || app.startTime == null || app.endTime == null) return null;
@@ -1478,7 +1771,7 @@ export const DETECTORS = [
1478
1771
  // lifetime-based measure this replaces.
1479
1772
  const busyCoreMs = runAggregates?.busyCoreMs ?? 0;
1480
1773
  const utilization = busyCoreMs / capacityCoreMs;
1481
- if (utilization >= this.thresholds.minUtil) return null;
1774
+ if (utilization >= thresholds.minUtil) return null;
1482
1775
 
1483
1776
  // CPU-time-based utilization (sparkMeasure): metric only, no threshold.
1484
1777
  let cpuUtilizationPct = null;
@@ -1500,9 +1793,20 @@ export const DETECTORS = [
1500
1793
  recommendation: `Average executor utilization was only ${value}%: consider reducing cluster size or enabling dynamic allocation.`,
1501
1794
  };
1502
1795
  },
1503
- },
1504
- {
1505
- type: 'memoryUtilization', scope: 'app', order: 102, fixEffort: 'config', version: 1,
1796
+ estimate(finding) {
1797
+ const fraction = finding.utilizationFraction ;
1798
+ const appDurationMs = finding.appDurationMs ;
1799
+ const totalCores = finding.totalCores ;
1800
+ if (fraction == null || appDurationMs == null || totalCores == null) {
1801
+ return costOnly('measured');
1802
+ }
1803
+ const idleCoreHours = (1 - fraction) * appDurationMs * totalCores / 3.6e6;
1804
+ return costOnly('measured', { value: idleCoreHours, unit: 'coreHours' });
1805
+ },
1806
+ }),
1807
+ defineAppDetector({
1808
+ type: 'memoryUtilization', order: 102, fixEffort: 'config', version: 1,
1809
+ emits: ['memoryUtilization'],
1506
1810
  docAnchor: '#bottleneck-memory-utilization',
1507
1811
  thresholds: {
1508
1812
  idleCoreWarn: 0.50, // WastedCoresAlertsReducer
@@ -1510,14 +1814,7 @@ export const DETECTORS = [
1510
1814
  bandTooHigh: 0.70, // below this => over-provisioned (cost signal)
1511
1815
  wasteBufferMultiplier: 1.5, // UNVERIFIED
1512
1816
  },
1513
- detect(
1514
-
1515
-
1516
-
1517
-
1518
-
1519
- ctx ,
1520
- ) {
1817
+ detect(ctx, thresholds) {
1521
1818
  const { app, executorsAdded, executorsRemoved, runAggregates, stages } = ctx;
1522
1819
  const out = [];
1523
1820
  // Nullish (not falsy) check: a literal startTime:0 must not be treated as "missing".
@@ -1535,10 +1832,11 @@ export const DETECTORS = [
1535
1832
  const allocatedMB = app.resources?.executor?.memoryMB ?? null;
1536
1833
 
1537
1834
  // ── 1a idle-cores rate ────────────────────────────────────────────────
1538
- if (runAggregates && totalCores > 0) {
1835
+ const busyCoreMs = runAggregates?.busyCoreMs;
1836
+ if (busyCoreMs != null && totalCores > 0) {
1539
1837
  const capacityCoreMs = totalCores * appDurationMs;
1540
- const idleRate = capacityCoreMs > 0 ? 1 - (runAggregates.busyCoreMs / capacityCoreMs) : 0;
1541
- if (idleRate > this.thresholds.idleCoreWarn) {
1838
+ const idleRate = capacityCoreMs > 0 ? 1 - (busyCoreMs / capacityCoreMs) : 0;
1839
+ if (idleRate > thresholds.idleCoreWarn) {
1542
1840
  const value = Math.round(idleRate * 100);
1543
1841
  out.push({
1544
1842
  type: 'memoryUtilization', variant: 'idleCores', stageId: null,
@@ -1573,14 +1871,14 @@ export const DETECTORS = [
1573
1871
  const ratio = heap / allocatedBytes;
1574
1872
  // The two bands are opposite signals: an explicit `rule` discriminator lets consumers
1575
1873
  // tell OOM-risk from over-provisioning without re-deriving the ratio.
1576
- if (ratio > this.thresholds.bandTooSmall) {
1874
+ if (ratio > thresholds.bandTooSmall) {
1577
1875
  out.push({
1578
1876
  type: 'memoryUtilization', variant: 'memoryBand', rule: 'heapNearCapacity',
1579
1877
  stageId: null, executorId: execId,
1580
1878
  impactBand: 'warning', metric: 'heapUsedRatio', value: Math.round(ratio * 100),
1581
1879
  recommendation: `Executor ${execId} peaked at ${Math.round(ratio * 100)}% of allocated heap: memory may be too small; raise spark.executor.memory to avoid OOM/spill.`,
1582
1880
  });
1583
- } else if (ratio < this.thresholds.bandTooHigh) {
1881
+ } else if (ratio < thresholds.bandTooHigh) {
1584
1882
  out.push({
1585
1883
  type: 'memoryUtilization', variant: 'memoryBand', rule: 'heapOverProvisioned',
1586
1884
  stageId: null, executorId: execId,
@@ -1601,13 +1899,13 @@ export const DETECTORS = [
1601
1899
  for (const s of stages.values()) usedRunTimeMs += s.executorRunTime ?? 0;
1602
1900
  const usedMBSeconds = allocatedMB * (usedRunTimeMs / 1000);
1603
1901
  const wastedMBSeconds = allocatedMBSeconds - usedMBSeconds;
1604
- if (wastedMBSeconds > this.thresholds.wasteBufferMultiplier * usedMBSeconds) {
1902
+ if (wastedMBSeconds > thresholds.wasteBufferMultiplier * usedMBSeconds) {
1605
1903
  const value = Math.round(wastedMBSeconds);
1606
1904
  out.push({
1607
1905
  type: 'memoryUtilization', variant: 'wasteModel', stageId: null,
1608
1906
  impactBand: 'info', metric: 'wastedMBSeconds', value,
1609
- confidence: memoryWasteConfidence(wastedMBSeconds, usedMBSeconds, this.thresholds.wasteBufferMultiplier),
1610
- validationRequired: 'Memory-waste estimate uses allocated-vs-used memory-time and a 1.5x buffer: confirm against the Spark UI before acting.',
1907
+ confidence: memoryWasteConfidence(wastedMBSeconds, usedMBSeconds, thresholds.wasteBufferMultiplier),
1908
+ validationRequired: `Memory-waste estimate uses allocated-vs-used memory-time and a ${thresholds.wasteBufferMultiplier}x buffer: confirm against the Spark UI before acting.`,
1611
1909
  recommendation: `Allocated executor memory sat largely idle over the run (~${value.toLocaleString('en-US')} MB-seconds wasted): review spark.executor.memory and executor count.`,
1612
1910
  });
1613
1911
  }
@@ -1615,97 +1913,153 @@ export const DETECTORS = [
1615
1913
 
1616
1914
  return out;
1617
1915
  },
1618
- },
1619
- {
1916
+ estimate(finding) {
1917
+ // The wasteModel variant reports metric: 'wastedMBSeconds', value: <MB-seconds>.
1918
+ if (finding.variant === 'wasteModel' && typeof finding.value === 'number') {
1919
+ return costOnly('measured', { value: finding.value, unit: 'mbSeconds' });
1920
+ }
1921
+ if (finding.variant === 'idleCores') {
1922
+ // Idle core-time priced as memory held but unused: the same MB-seconds unit as wasteModel, so comparable.
1923
+ const idleRateFraction = finding.idleRateFraction ;
1924
+ const allocatedMB = finding.allocatedMB ;
1925
+ const peakExecutors = finding.peakExecutors ;
1926
+ const appDurationMs = finding.appDurationMs ;
1927
+ if (idleRateFraction != null && allocatedMB != null && peakExecutors != null && appDurationMs != null) {
1928
+ const wastedMBSeconds = idleRateFraction * allocatedMB * peakExecutors * (appDurationMs / 1000);
1929
+ return costOnly('modeled', { value: wastedMBSeconds, unit: 'mbSeconds' });
1930
+ }
1931
+ return costOnly('modeled');
1932
+ }
1933
+ // Only the over-provisioned band is a waste; the near-capacity band is an OOM-risk signal with
1934
+ // no magnitude, and the dataUnavailable shape has no inputs: both stay informational.
1935
+ if (finding.variant === 'memoryBand' && finding.rule === 'heapOverProvisioned') {
1936
+ const allocatedBytes = finding.allocatedBytes ;
1937
+ const heap = finding.heap ;
1938
+ const appDurationMs = finding.appDurationMs ;
1939
+ if (allocatedBytes != null && heap != null && appDurationMs != null) {
1940
+ const unusedMB = (allocatedBytes - heap) / (1024 * 1024);
1941
+ const wastedMBSeconds = unusedMB * (appDurationMs / 1000);
1942
+ return costOnly('modeled', { value: wastedMBSeconds, unit: 'mbSeconds' });
1943
+ }
1944
+ }
1945
+ return costOnly('modeled');
1946
+ },
1947
+ }),
1948
+ defineAppDetector({
1620
1949
  // Per-RDD cache-utilization proxies (this repo's own design: Spark event logs carry no
1621
1950
  // block-access events, so a literal cache hit rate isn't derivable). Two per-RDD tiered
1622
- // checks over rddInfo snapshots: partial caching and disk spillover. An RDD can produce both.
1623
- type: 'cacheUtilization', scope: 'app', order: 103, fixEffort: 'code', version: 1,
1951
+ // checks over rddInfo: partial caching and disk spillover. An RDD can produce both. rddInfo's
1952
+ // sizes come from SparkListenerBlockUpdated when the log has it, else from StageSubmitted's
1953
+ // RDD Info (0 since Spark 2.3; Spark 1.x fills it only on StageCompleted); with neither, on
1954
+ // Spark 2.3+ with logBlockUpdates off, a storageUnobserved caveat replaces them.
1955
+ type: 'cacheUtilization', order: 103, fixEffort: 'code', version: 2,
1956
+ emits: ['cacheUtilization'],
1624
1957
  docAnchor: '#bottleneck-cache-utilization',
1625
1958
  thresholds: {
1626
1959
  cachedRatioWarn: 0.50, cachedRatioInfo: 0.90,
1627
1960
  diskRatioWarn: 0.40, diskRatioInfo: 0.15,
1628
1961
  },
1629
- detect(
1630
-
1631
-
1632
-
1633
-
1634
-
1635
- ctx ,
1636
- ) {
1962
+ detect(ctx, thresholds) {
1637
1963
  const rddInfo = ctx.app?.rddInfo;
1638
1964
  if (!(rddInfo instanceof Map)) return null;
1639
1965
  const out = [];
1966
+ let persistedRddCount = 0;
1967
+ // With block-update logging on, zero rdd_* updates means nothing was ever cached, not a gap.
1968
+ const blockUpdatesLogged = String(ctx.app?.config?.['spark.eventLog.logBlockUpdates.enabled']).toLowerCase() === 'true';
1969
+ // Spark before 2.3 has no block-update logging and writes RDD Info's cache figures only on
1970
+ // StageCompleted, which isn't read: the caveat's advice doesn't apply there. A log with no
1971
+ // version is pre-1.3 (no SparkListenerLogStart); every 2.3+ log records one.
1972
+ const version = /^(\d+)\.(\d+)/.exec(ctx.app?.sparkVersion ?? '');
1973
+ const preBlockUpdates = version == null || Number(version[1]) < 2 || (Number(version[1]) === 2 && Number(version[2]) < 3);
1974
+ let anyStorageEvidence = blockUpdatesLogged || preBlockUpdates || (ctx.app?.rddBlockUpdates ?? 0) > 0;
1640
1975
  for (const rdd of rddInfo.values()) {
1641
1976
  const sl = rdd.storageLevel ?? {};
1642
1977
  if (!(sl.useMemory || sl.useDisk)) continue;
1978
+ persistedRddCount++;
1643
1979
  if (!((rdd.numCachedPartitions ?? 0) > 0)) continue;
1980
+ anyStorageEvidence = true;
1644
1981
 
1645
1982
  if ((rdd.numPartitions ?? 0) > 0) {
1646
1983
  const cachedRatio = rdd.numCachedPartitions / rdd.numPartitions;
1647
- if (cachedRatio < this.thresholds.cachedRatioWarn) out.push(partialCacheFinding(rdd, cachedRatio, 'warning'));
1648
- else if (cachedRatio < this.thresholds.cachedRatioInfo) out.push(partialCacheFinding(rdd, cachedRatio, 'info'));
1984
+ if (cachedRatio < thresholds.cachedRatioWarn) out.push(partialCacheFinding(rdd, cachedRatio, 'warning'));
1985
+ else if (cachedRatio < thresholds.cachedRatioInfo) out.push(partialCacheFinding(rdd, cachedRatio, 'info'));
1649
1986
  }
1650
1987
 
1651
1988
  if (sl.useMemory && sl.useDisk) {
1652
1989
  const total = (rdd.memorySize ?? 0) + (rdd.diskSize ?? 0);
1653
1990
  if (total > 0) {
1654
1991
  const diskRatio = (rdd.diskSize ?? 0) / total;
1655
- if (diskRatio > this.thresholds.diskRatioWarn) out.push(diskSpilloverFinding(rdd, diskRatio, 'warning'));
1656
- else if (diskRatio > this.thresholds.diskRatioInfo) out.push(diskSpilloverFinding(rdd, diskRatio, 'info'));
1992
+ if (diskRatio > thresholds.diskRatioWarn) out.push(diskSpilloverFinding(rdd, diskRatio, 'warning'));
1993
+ else if (diskRatio > thresholds.diskRatioInfo) out.push(diskSpilloverFinding(rdd, diskRatio, 'info'));
1657
1994
  }
1658
1995
  }
1659
1996
  }
1997
+ if (persistedRddCount > 0 && !anyStorageEvidence) out.push(storageUnobservedFinding(persistedRddCount));
1660
1998
  return out;
1661
1999
  },
1662
- },
1663
- {
2000
+ estimate(finding) {
2001
+ // storageUnobserved reports missing evidence: no sizes, so nothing to model.
2002
+ if (finding.dataUnavailable) return costOnly('none');
2003
+ const memorySize = (finding.memorySize ) ?? 0;
2004
+ const diskSize = (finding.diskSize ) ?? 0;
2005
+ const numCachedPartitions = (finding.numCachedPartitions ) ?? 0;
2006
+ const numPartitions = (finding.numPartitions ) ?? 0;
2007
+ const numUncachedPartitions = Math.max(0, numPartitions - numCachedPartitions);
2008
+ const cachedBytes = memorySize + diskSize;
2009
+ // Extrapolate never-cached partitions' size from the CACHED partitions' average (uncached/
2010
+ // cached, not uncached/total: numCachedPartitions produced cachedBytes). diskSize is added
2011
+ // once more: those bytes are cached but on disk, so re-reading them still costs I/O like an uncached partition.
2012
+ const uncachedBytes = numCachedPartitions > 0 ? (cachedBytes / numCachedPartitions) * numUncachedPartitions : 0;
2013
+ const uncachedOrSpilledBytes = uncachedBytes + diskSize;
2014
+ const wasteMs = (uncachedOrSpilledBytes / RE_READ_THROUGHPUT_BPS) * 1000;
2015
+ return costOnly('modeled', { value: wasteMs, unit: 'ms' });
2016
+ },
2017
+ }),
2018
+ defineAppDetector({
1664
2019
  // Non-local task ratio across stage.localityStats (RACK_LOCAL + ANY vs all tasks), the other
1665
2020
  // half of the "Wasted Cores Ratio" (idle-core half is memoryUtilization's idleCores).
1666
2021
  // NO_PREF stays in the denominator only: shuffle-read stages legitimately report it.
1667
- type: 'coreLocality', scope: 'app', order: 103, fixEffort: 'config', version: 1,
2022
+ type: 'coreLocality', order: 103, fixEffort: 'config', version: 1,
2023
+ emits: ['coreLocality'],
1668
2024
  docAnchor: '#bottleneck-core-locality',
1669
2025
  thresholds: { minTasks: 50, warnRatio: 0.15, critRatio: 0.35 },
1670
- detect(
1671
-
1672
- ctx ,
1673
- ) {
2026
+ detect(ctx, thresholds) {
1674
2027
  const { totalTasks, nonLocalTasks, ratio } = computeCoreLocalityRatio([...ctx.stages.values()]);
1675
- if (totalTasks == null || totalTasks < this.thresholds.minTasks) return null;
2028
+ if (totalTasks == null || totalTasks < thresholds.minTasks) return null;
1676
2029
  // computeCoreLocalityRatio only returns ratio:null together with totalTasks:null (shared
1677
2030
  // EMPTY sentinel); the totalTasks guard above rules that out, so ratio is non-null here.
1678
- if (ratio < this.thresholds.warnRatio) return null;
2031
+ if (ratio < thresholds.warnRatio) return null;
1679
2032
 
1680
2033
  const value = Math.round(ratio * 100);
1681
2034
  return {
1682
2035
  type: 'coreLocality', stageId: null,
1683
- impactBand: ratio >= this.thresholds.critRatio ? 'critical' : 'warning',
2036
+ impactBand: ratio >= thresholds.critRatio ? 'critical' : 'warning',
1684
2037
  metric: 'nonLocalRatio', value,
1685
2038
  // Raw count behind the ratio, for the impact estimator. Non-null whenever totalTasks is.
1686
2039
  nonLocalTaskCount: nonLocalTasks ,
1687
- confidence: coreLocalityConfidence(ratio , totalTasks, this.thresholds),
1688
- validationRequired: 'This finding is gated by 15%/35% non-local-ratio thresholds (and a 50-task minimum), our own noise floor for this metric.',
2040
+ confidence: coreLocalityConfidence(ratio , totalTasks, thresholds),
2041
+ validationRequired: `This finding is gated by ${shareLabel(thresholds.warnRatio)}/${shareLabel(thresholds.critRatio)} non-local-ratio thresholds (and a ${thresholds.minTasks}-task minimum), our own noise floor for this metric.`,
1689
2042
  recommendation: `${value}% of tasks (${nonLocalTasks }) ran without process- or node-local data placement: check spark.locality.wait settings and executor/data colocation.`,
1690
2043
  };
1691
2044
  },
1692
- },
1693
- {
2045
+ estimate(finding) {
2046
+ const nonLocal = (finding.nonLocalTaskCount ) ?? 0;
2047
+ const coreMs = nonLocal * NETWORK_FETCH_PENALTY_MS;
2048
+ return costOnly('modeled', { value: coreMs, unit: 'coreMs' });
2049
+ },
2050
+ }),
2051
+ defineAppDetector({
1694
2052
  // Short-lived executors: stood up and torn down before doing useful work (wasteful
1695
2053
  // re-provisioning, not normal scale-down). Reuses utilization's add/remove matching, but
1696
2054
  // measures lifetime against a threshold instead of aggregate active-time.
1697
- type: 'autoscalingChurn', scope: 'app', order: 103, fixEffort: 'config', version: 1,
2055
+ type: 'autoscalingChurn', order: 103, fixEffort: 'config', version: 1,
2056
+ emits: ['autoscalingChurn'],
1698
2057
  docAnchor: '#bottleneck-autoscaling-churn',
1699
2058
  thresholds: { shortLivedMs: 120_000, warningPct: 0.30, criticalPct: 0.60, minExecutors: 5 },
1700
- detect(
1701
-
1702
-
1703
-
1704
- ctx ,
1705
- ) {
2059
+ detect(ctx, thresholds) {
1706
2060
  const { app, executorsAdded, executorsRemoved } = ctx;
1707
2061
  if (!app || executorsAdded.length === 0 || app.endTime == null) return null;
1708
- if (executorsAdded.length < this.thresholds.minExecutors) return null;
2062
+ if (executorsAdded.length < thresholds.minExecutors) return null;
1709
2063
 
1710
2064
  const removedAt = new Map ();
1711
2065
  for (const ev of executorsRemoved) removedAt.set(ev.executorId, ev.timestamp);
@@ -1714,12 +2068,12 @@ export const DETECTORS = [
1714
2068
  for (const ev of executorsAdded) {
1715
2069
  const endedAt = removedAt.has(ev.executorId) ? removedAt.get(ev.executorId) : app.endTime;
1716
2070
  const lifetime = endedAt - ev.timestamp;
1717
- if (lifetime < this.thresholds.shortLivedMs) shortLivedCount++;
2071
+ if (lifetime < thresholds.shortLivedMs) shortLivedCount++;
1718
2072
  }
1719
2073
 
1720
2074
  const shortLivedPct = shortLivedCount / executorsAdded.length;
1721
- const impactBand = shortLivedPct > this.thresholds.criticalPct ? 'critical'
1722
- : shortLivedPct > this.thresholds.warningPct ? 'warning' : null;
2075
+ const impactBand = shortLivedPct > thresholds.criticalPct ? 'critical'
2076
+ : shortLivedPct > thresholds.warningPct ? 'warning' : null;
1723
2077
  if (!impactBand) return null;
1724
2078
 
1725
2079
  const pct = Math.round(shortLivedPct * 100);
@@ -1728,18 +2082,24 @@ export const DETECTORS = [
1728
2082
  metric: 'shortLivedExecutorPct', value: pct,
1729
2083
  // Raw count behind the percentage, for the impact estimator's startup-overhead figure.
1730
2084
  shortLivedExecutorCount: shortLivedCount,
1731
- confidence: autoscalingChurnConfidence(shortLivedPct, this.thresholds.warningPct, this.thresholds.criticalPct),
2085
+ confidence: autoscalingChurnConfidence(shortLivedPct, thresholds.warningPct, thresholds.criticalPct),
1732
2086
  recommendation: `${pct}% of executors ran for under 2 minutes before being removed. This looks like wasteful re-provisioning rather than normal scale-down; consider raising spark.dynamicAllocation.executorIdleTimeout or widening the minExecutors/maxExecutors bounds to reduce flapping.`,
1733
2087
  };
1734
2088
  },
1735
- },
1736
- {
2089
+ estimate(finding) {
2090
+ const shortLived = (finding.shortLivedExecutorCount ) ?? 0;
2091
+ const executorHours = (shortLived * EXECUTOR_STARTUP_OVERHEAD_MS) / 3.6e6;
2092
+ return costOnly('modeled', { value: executorHours, unit: 'coreHours' });
2093
+ },
2094
+ }),
2095
+ defineAppDetector({
1737
2096
  // Cross-execution relation reuse: flags an input relation scanned by two or more SQL
1738
2097
  // executions in one run, firing on real relation names (parquet:..., jdbc:...).
1739
- type: 'cachingOpportunity', scope: 'app', order: 105, fixEffort: 'code', version: 1,
2098
+ type: 'cachingOpportunity', order: 105, fixEffort: 'code', version: 1,
2099
+ emits: ['cachingOpportunity'],
1740
2100
  docAnchor: '#bottleneck-caching-opportunity',
1741
2101
  thresholds: { minExecutions: 2 },
1742
- detect( ctx ) {
2102
+ detect(ctx, thresholds) {
1743
2103
  const sql = ctx.sql;
1744
2104
  if (!(sql instanceof Map) || sql.size === 0) return null;
1745
2105
 
@@ -1827,7 +2187,7 @@ export const DETECTORS = [
1827
2187
  // Qualifying = enough distinct executions on its own. Nested-dedupe: a qualifying composite
1828
2188
  // with a qualifying ANCESTOR is subsumed, fully (equal sets) or partially (residual).
1829
2189
  const isQualifying = (fp ) =>
1830
- byComposite.has(fp) && byComposite.get(fp) .executionIds.size >= this.thresholds.minExecutions;
2190
+ byComposite.has(fp) && byComposite.get(fp) .executionIds.size >= thresholds.minExecutions;
1831
2191
  const compositeResolutions = new Map ();
1832
2192
  for (const [fingerprint, agg] of byComposite) {
1833
2193
  if (!isQualifying(fingerprint)) { compositeResolutions.set(fingerprint, { finalExecutionIds: agg.executionIds, suppressed: true }); continue; }
@@ -1836,7 +2196,7 @@ export const DETECTORS = [
1836
2196
  const coveredByAncestors = new Set(qualifyingAncestors.flatMap(outer => [...outer.executionIds]));
1837
2197
  const residual = new Set([...agg.executionIds].filter(id => !coveredByAncestors.has(id)));
1838
2198
  if (residual.size === 0) compositeResolutions.set(fingerprint, { finalExecutionIds: residual, suppressed: true });
1839
- else if (residual.size < this.thresholds.minExecutions) compositeResolutions.set(fingerprint, { finalExecutionIds: residual, suppressed: true });
2199
+ else if (residual.size < thresholds.minExecutions) compositeResolutions.set(fingerprint, { finalExecutionIds: residual, suppressed: true });
1840
2200
  else compositeResolutions.set(fingerprint, { finalExecutionIds: residual, suppressed: false });
1841
2201
  }
1842
2202
 
@@ -1869,7 +2229,7 @@ export const DETECTORS = [
1869
2229
  metric: 'executionReuse', value,
1870
2230
  format: 'derived', relations, operator: agg.operator, relation: relationDisplay,
1871
2231
  executionIds: finalExecutionIds, totalReadBytes,
1872
- confidence: cachingReuseConfidence(value, this.thresholds.minExecutions),
2232
+ confidence: cachingReuseConfidence(value, thresholds.minExecutions),
1873
2233
  validationRequired:
1874
2234
  'Composite reuse is inferred from a structural plan-shape match (operator + normalized ' +
1875
2235
  'join/filter condition + child shapes) across SQL executions; confirm these executions ' +
@@ -1888,7 +2248,7 @@ export const DETECTORS = [
1888
2248
  const residualExecutionIds = covered
1889
2249
  ? [...agg.executionIds].filter(id => !covered.has(id))
1890
2250
  : [...agg.executionIds];
1891
- if (residualExecutionIds.length < this.thresholds.minExecutions) continue;
2251
+ if (residualExecutionIds.length < thresholds.minExecutions) continue;
1892
2252
  const value = residualExecutionIds.length;
1893
2253
  const totalReadBytes = residualExecutionIds.reduce((sum, id) => sum + (agg.executionBytes.get(id) ?? 0), 0);
1894
2254
  const recommendation = totalReadBytes >= 128 * MB
@@ -1900,7 +2260,7 @@ export const DETECTORS = [
1900
2260
  relation: agg.relation, format: agg.format,
1901
2261
  executionIds: residualExecutionIds.sort((a, b) => a - b),
1902
2262
  totalReadBytes,
1903
- confidence: cachingReuseConfidence(value, this.thresholds.minExecutions),
2263
+ confidence: cachingReuseConfidence(value, thresholds.minExecutions),
1904
2264
  validationRequired:
1905
2265
  'Relation-reuse is inferred from the pre-AQE plan scan identity across SQL ' +
1906
2266
  'executions; confirm the reads are the same data and cacheable within one ' +
@@ -1910,15 +2270,18 @@ export const DETECTORS = [
1910
2270
  }
1911
2271
  return out;
1912
2272
  },
1913
- },
1914
- {
1915
- type: 'jobFailureRate', scope: 'app', order: 110, fixEffort: 'code', version: 1,
2273
+ estimate(finding) {
2274
+ const totalReadBytes = (finding.totalReadBytes ) ?? 0;
2275
+ const wasteMs = (totalReadBytes / RE_READ_THROUGHPUT_BPS) * 1000;
2276
+ return costOnly('modeled', { value: wasteMs, unit: 'ms' });
2277
+ },
2278
+ }),
2279
+ defineAppDetector({
2280
+ type: 'jobFailureRate', order: 110, fixEffort: 'code', version: 1,
2281
+ emits: ['jobFailureRate'],
1916
2282
  docAnchor: '#bottleneck-job-failure-rate',
1917
2283
  thresholds: { infoRate: 0.10, warnRate: 0.30, critRate: 0.50 },
1918
- detect(
1919
-
1920
- ctx ,
1921
- ) {
2284
+ detect(ctx, thresholds) {
1922
2285
  const { jobs, stages } = ctx;
1923
2286
  const all = jobs ? [...jobs.values()] : [];
1924
2287
  const completed = all.filter(j => j.result != null);
@@ -1926,7 +2289,7 @@ export const DETECTORS = [
1926
2289
  const failedJobList = completed.filter(j => j.succeeded === false);
1927
2290
  const failedJobs = failedJobList.length;
1928
2291
  const rate = failedJobs / completed.length;
1929
- if (rate < this.thresholds.infoRate) return null;
2292
+ if (rate < thresholds.infoRate) return null;
1930
2293
  let totalTasks = 0, failedTasks = 0;
1931
2294
  for (const s of stages.values()) { totalTasks += s.taskCount ?? 0; failedTasks += s.failedTasks ?? 0; }
1932
2295
  const taskFailureRate = totalTasks > 0 ? failedTasks / totalTasks : 0;
@@ -1939,113 +2302,121 @@ export const DETECTORS = [
1939
2302
  const totalJobs = completed.length;
1940
2303
  return {
1941
2304
  type: 'jobFailureRate', stageId: null,
1942
- impactBand: rate >= this.thresholds.critRate ? 'critical' : rate >= this.thresholds.warnRate ? 'warning' : 'info',
2305
+ impactBand: rate >= thresholds.critRate ? 'critical' : rate >= thresholds.warnRate ? 'warning' : 'info',
1943
2306
  metric: 'jobFailureRate', value: Math.round(rate * 1000) / 10,
1944
2307
  failedJobs, totalJobs, failedTasks, totalTasks, avgJobDurationMs,
1945
2308
  taskFailureRate: Math.round(taskFailureRate * 1000) / 10,
1946
2309
  recommendation: `${failedJobs} of ${totalJobs} jobs never recovered: inspect the driver log for the failed job(s) and the stage failures that triggered them.`,
1947
2310
  };
1948
2311
  },
1949
- },
2312
+ estimate(finding) {
2313
+ const failedJobs = (finding.failedJobs ) ?? 0;
2314
+ const avgJobDurationMs = (finding.avgJobDurationMs ) ?? 0;
2315
+ const coreHoursIsh = (failedJobs * avgJobDurationMs) / 3.6e6;
2316
+ return costOnly('modeled', { value: coreHoursIsh, unit: 'coreHours' });
2317
+ },
2318
+ }),
1950
2319
  // ── Config-sanity entries (scope:'config', inScorecard:false) ────────────────
1951
- {
1952
- type: 'configAudit', scope: 'config', order: 120, fixEffort: 'config', version: 1, inScorecard: false,
2320
+ defineConfigDetector({
2321
+ type: 'configAudit', order: 120, fixEffort: 'config', version: 1, inScorecard: false,
2322
+ emits: ['configAudit'],
1953
2323
  docAnchor: '#config-shuffle-service', thresholds: {}, property: 'spark.shuffle.service.enabled',
1954
- detect(ctx ) {
1955
- const res = ctx.app?.resources ?? null;
2324
+ detect(target) {
2325
+ const res = target.app?.resources ?? null;
1956
2326
  if (res?.dynamicAllocationEnabled === true && res?.shuffleServiceEnabled === false) {
1957
2327
  return {
1958
2328
  type: 'configAudit', property: 'spark.shuffle.service.enabled',
1959
- impactBand: 'warning', metric: 'config', value: 'false',
2329
+ impactBand: 'warning', metric: 'config', valueText: 'false',
1960
2330
  recommendation: 'Dynamic allocation is on but the external shuffle service is off: set spark.shuffle.service.enabled=true so shuffle data survives executor removal.',
1961
2331
  };
1962
2332
  }
1963
2333
  return null;
1964
2334
  },
1965
- },
1966
- {
1967
- type: 'configAudit', scope: 'config', order: 121, fixEffort: 'config', version: 1, inScorecard: false,
2335
+ estimate: noWasteModel,
2336
+ }),
2337
+ defineConfigDetector({
2338
+ type: 'configAudit', order: 121, fixEffort: 'config', version: 1, inScorecard: false,
2339
+ emits: ['configAudit'],
1968
2340
  docAnchor: '#config-autoscale-bounds', thresholds: {}, property: 'spark.dynamicAllocation.maxExecutors',
1969
- detect(ctx ) {
1970
- const app = ctx.app; const config = app?.config ?? {}; const res = app?.resources ?? null;
2341
+ detect(target) {
2342
+ const app = target.app; const config = app?.config ?? {}; const res = app?.resources ?? null;
1971
2343
  if (res?.dynamicAllocationEnabled !== true) return null;
1972
2344
  const minN = config['spark.dynamicAllocation.minExecutors'] != null ? parseInt(config['spark.dynamicAllocation.minExecutors'], 10) : null;
1973
2345
  const maxN = config['spark.dynamicAllocation.maxExecutors'] != null ? parseInt(config['spark.dynamicAllocation.maxExecutors'], 10) : null;
1974
2346
  if (minN != null && maxN != null && minN > maxN) {
1975
2347
  return {
1976
2348
  type: 'configAudit', property: 'spark.dynamicAllocation.minExecutors',
1977
- impactBand: 'critical', metric: 'config', value: `${minN} > ${maxN}`,
2349
+ impactBand: 'critical', metric: 'config', valueText: `${minN} > ${maxN}`,
1978
2350
  recommendation: `Autoscaling bounds are inverted: spark.dynamicAllocation.minExecutors (${minN}) exceeds maxExecutors (${maxN}). Set min ≤ max.`,
1979
2351
  };
1980
2352
  }
1981
2353
  if (maxN == null) {
1982
2354
  return {
1983
2355
  type: 'configAudit', property: 'spark.dynamicAllocation.maxExecutors',
1984
- impactBand: 'info', metric: 'config', value: '(unset)',
2356
+ impactBand: 'info', metric: 'config', valueText: '(unset)',
1985
2357
  recommendation: 'Dynamic allocation is on with no upper bound: set spark.dynamicAllocation.maxExecutors to cap cluster growth.',
1986
2358
  };
1987
2359
  }
1988
2360
  return null;
1989
2361
  },
1990
- },
1991
- {
1992
- type: 'configAudit', scope: 'config', order: 122, fixEffort: 'config', version: 1, inScorecard: false,
2362
+ estimate: noWasteModel,
2363
+ }),
2364
+ defineConfigDetector({
2365
+ type: 'configAudit', order: 122, fixEffort: 'config', version: 1, inScorecard: false,
2366
+ emits: ['configAudit'],
1993
2367
  docAnchor: '#config-serializer', thresholds: {}, property: 'spark.serializer',
1994
- detect(ctx ) {
1995
- const app = ctx.app; const config = app?.config ?? {}; const res = app?.resources ?? null;
2368
+ detect(target) {
2369
+ const app = target.app; const config = app?.config ?? {}; const res = app?.resources ?? null;
1996
2370
  if (Object.keys(config).length === 0) return null;
1997
2371
  const ser = res?.serializer ?? config['spark.serializer'] ?? null;
1998
2372
  const isKryo = typeof ser === 'string' && /kryo/i.test(ser);
1999
2373
  if (isKryo) return null;
2000
2374
  return {
2001
2375
  type: 'configAudit', property: 'spark.serializer',
2002
- impactBand: 'info', metric: 'config', value: ser ?? '(default JavaSerializer)',
2376
+ impactBand: 'info', metric: 'config', valueText: ser ?? '(default JavaSerializer)',
2003
2377
  recommendation: `Current serializer is ${ser ?? 'the default JavaSerializer'}: consider spark.serializer=org.apache.spark.serializer.KryoSerializer for faster, smaller buffers.`,
2004
2378
  };
2005
2379
  },
2006
- },
2007
- {
2008
- type: 'configAudit', scope: 'config', order: 123, fixEffort: 'config', version: 1, inScorecard: false,
2380
+ estimate: noWasteModel,
2381
+ }),
2382
+ defineConfigDetector({
2383
+ type: 'configAudit', order: 123, fixEffort: 'config', version: 1, inScorecard: false,
2384
+ emits: ['configAudit'],
2009
2385
  docAnchor: '#config-memory-overhead', thresholds: { floorMB: 384, floorPct: 0.1 }, property: 'spark.executor.memoryOverhead',
2010
- detect(
2011
-
2012
- ctx ,
2013
- ) {
2014
- const res = ctx.app?.resources ?? null;
2386
+ detect(target, thresholds) {
2387
+ const res = target.app?.resources ?? null;
2015
2388
  const memMB = res?.executor?.memoryMB ?? null;
2016
2389
  const ovMB = res?.executor?.memoryOverheadMB ?? null;
2017
2390
  if (memMB == null || ovMB == null) return null;
2018
- const floor = Math.max(this.thresholds.floorMB, Math.round(memMB * this.thresholds.floorPct));
2391
+ const floor = Math.max(thresholds.floorMB, Math.round(memMB * thresholds.floorPct));
2019
2392
  if (ovMB >= floor) return null;
2020
2393
  return {
2021
2394
  type: 'configAudit', property: 'spark.executor.memoryOverhead',
2022
- impactBand: 'info', metric: 'config', value: `${ovMB} MiB`,
2395
+ impactBand: 'info', metric: 'config', valueText: `${ovMB} MiB`,
2023
2396
  recommendation: `Executor memoryOverhead (${ovMB} MiB) is below Spark's default floor of ${floor} MiB (max of 384 MiB or 10% of executor memory): raise it to avoid off-heap OOM-kills.`,
2024
2397
  };
2025
2398
  },
2026
- },
2399
+ estimate: noWasteModel,
2400
+ }),
2027
2401
  // ── Plan-metric entries (scope:'sql') ────────────────────────────────────
2028
- {
2029
- type: 'duplicatePlanSubtree', scope: 'sql', order: 130, fixEffort: 'code', version: 2,
2402
+ defineSqlDetector({
2403
+ type: 'duplicatePlanSubtree', order: 130, fixEffort: 'code', version: 2,
2404
+ emits: ['duplicatePlanSubtree'],
2030
2405
  docAnchor: '#bottleneck-duplicate-plan-subtree',
2031
2406
  // stageFloorPct: the 0.5% runtime floor. The claim counts at most each linked stage's own
2032
2407
  // task-active time, so a repeat whose stages together lasted less than this share of the run
2033
2408
  // graded info: 340 of 545 findings on the 14 real logs. The repeat is still in the plan; the
2034
2409
  // floor is why they're dropped. A repeat with no linked stage time is kept.
2035
2410
  thresholds: { minSubtreeSize: 3, minOccurrences: 2, stageFloorPct: 0.005 },
2036
- detect(
2037
-
2038
- sqlExec ,
2039
- ctx ,
2040
- ) {
2411
+ detect(sqlExec, ctx, thresholds) {
2041
2412
  if (!sqlExec.planTree) return null;
2042
- const groups = findDuplicateSubtrees(sqlExec.planTree, this.thresholds);
2413
+ const groups = findDuplicateSubtrees(sqlExec.planTree, thresholds);
2043
2414
  if (groups.length === 0) return null;
2044
2415
  const fallbackStageIds = stageIdsForSqlExec(sqlExec.id, ctx.stages);
2045
2416
  const executionNodes = [];
2046
2417
  walkPlanTree(sqlExec.planTree, (node) => executionNodes.push(node));
2047
2418
  const operatorsByStage = operatorCountByStage(executionNodes);
2048
- const appDurationMs = computeAppDurationMs(ctx);
2419
+ const runMs = appDurationMs(ctx.app);
2049
2420
  const findings = groups.map((g) => {
2050
2421
  const nodes = [];
2051
2422
  for (const n of g.nodes) walkPlanTree(n, (node) => nodes.push(node));
@@ -2055,7 +2426,7 @@ export const DETECTORS = [
2055
2426
  const stage = ctx.stages.get(id);
2056
2427
  if (stage) stagesMs += Math.max(0, (stage.completedAt ?? 0) - (stage.submittedAt ?? 0));
2057
2428
  }
2058
- if (appDurationMs != null && stagesMs > 0 && stagesMs < appDurationMs * this.thresholds.stageFloorPct) return null;
2429
+ if (runMs != null && stagesMs > 0 && stagesMs < runMs * thresholds.stageFloorPct) return null;
2059
2430
  const occurrencesIdentical = occurrencesHaveIdenticalDetails(g.nodes);
2060
2431
  const stageShares = stageOperatorShares(nodes, operatorsByStage);
2061
2432
  // resolvePlanTree always sets id; safe downstream of it.
@@ -2072,7 +2443,7 @@ export const DETECTORS = [
2072
2443
  metric: 'subtreeOccurrences', value: g.occurrences,
2073
2444
  rootName: g.rootName, subtreeSize: g.subtreeSize, sampleRelation: g.sampleRelation,
2074
2445
  groupIndex: g.groupIndex,
2075
- confidence: occurrencesIdentical ? duplicateSubtreeConfidence(g.subtreeSize, g.occurrences, this.thresholds) : 'low',
2446
+ confidence: occurrencesIdentical ? duplicateSubtreeConfidence(g.subtreeSize, g.occurrences, thresholds) : 'low',
2076
2447
  validationRequired: 'Duplicate-subtree matching compares operator names and metric names only, not literal values or expr IDs: confirm the repeated work is real in the Spark SQL plan tab before acting.',
2077
2448
  recommendation: (g.isExchangeRoot
2078
2449
  ? `A ${g.subtreeSize}-node subtree rooted at ${pathBasename(g.rootName)} repeats ${g.occurrences}x in this plan${touching}: this looks like a possible missed exchange reuse; check whether the same shuffle could be computed once and reused.`
@@ -2081,18 +2452,50 @@ export const DETECTORS = [
2081
2452
  }).filter((f) => f !== null);
2082
2453
  return findings.length > 0 ? findings : null;
2083
2454
  },
2084
- },
2085
- {
2086
- type: 'smallFiles', scope: 'sql', order: 131, fixEffort: 'config', version: 2,
2455
+ estimate(finding, ctx) {
2456
+ const stageIds = finding.stageIds ;
2457
+ if (!stageIds || stageIds.length === 0) return null;
2458
+ // detect() reports subtreeOccurrences >= 2. Only repeats past the first are redundant:
2459
+ // computing the subtree once is real work, so waste is (occurrences-1)/occurrences of the stages' time.
2460
+ const occurrences = typeof finding.value === 'number' ? finding.value : 0;
2461
+ // Defensive only: minOccurrences guarantees occurrences >= 2 on real data; a malformed-value fallback.
2462
+ if (occurrences < 2) return costOnly('none');
2463
+ // Same-shaped repeats whose details differ compute different data: nothing is known to be
2464
+ // recomputed, so there is no time to claim.
2465
+ if (finding.occurrencesIdentical === false) return costOnly('none');
2466
+ // Each stage contributes the share of its operators inside the repeated subtree: a stage it
2467
+ // shares with other operators (the consuming join, the join's other side) isn't all its
2468
+ // time, and claiming whole stages let sibling groups claim the same stage twice. Findings
2469
+ // built without the field (hand-made fixtures) count every linked stage whole.
2470
+ const shares = (finding.stageShares ?? null) ;
2471
+ const redundantFraction = (occurrences - 1) / occurrences;
2472
+ const wasteMsByStage = new Map ();
2473
+ for (const id of stageIds) {
2474
+ const s = ctx.stages.get(id);
2475
+ const share = shares ? (shares[id] ?? 0) : 1;
2476
+ if (s && share > 0) {
2477
+ // Time with tasks running, not submit-to-complete: a stage left waiting for cores
2478
+ // (2491s open, 60s of tasks on a real log) isn't recomputing anything while it waits.
2479
+ const activeMs = s.taskActiveMs ?? Math.max(0, (s.completedAt ?? 0) - (s.submittedAt ?? 0));
2480
+ wasteMsByStage.set(id, activeMs * share * redundantFraction);
2481
+ }
2482
+ }
2483
+ // No operator of the subtree ran in a known stage: no time to attribute.
2484
+ if (wasteMsByStage.size === 0) return costOnly('none');
2485
+ const totalWasteMs = [...wasteMsByStage.values()].reduce((sum, ms) => sum + ms, 0);
2486
+ const rawWaste = { value: totalWasteMs, unit: 'ms' };
2487
+ return multiStageImpact([...wasteMsByStage.keys()], wasteMsByStage, ctx, 'measured', rawWaste)
2488
+ ?? costOnly('measured', rawWaste);
2489
+ },
2490
+ }),
2491
+ defineSqlDetector({
2492
+ type: 'smallFiles', order: 131, fixEffort: 'config', version: 2,
2493
+ emits: ['smallFiles'],
2087
2494
  docAnchor: '#bottleneck-small-files',
2088
2495
  thresholds: { minFiles: 100, maxAvgFileSizeMB: 3 },
2089
- detect(
2090
-
2091
- sqlExec ,
2092
- ctx ,
2093
- ) {
2496
+ detect(sqlExec, ctx, thresholds) {
2094
2497
  if (!sqlExec.planTree) return null;
2095
- const { minFiles, maxAvgFileSizeMB } = this.thresholds;
2498
+ const { minFiles, maxAvgFileSizeMB } = thresholds;
2096
2499
 
2097
2500
  const hits = [];
2098
2501
  walkPlanTree(sqlExec.planTree, (node) => {
@@ -2127,27 +2530,36 @@ export const DETECTORS = [
2127
2530
  };
2128
2531
  });
2129
2532
  },
2130
- },
2131
- {
2533
+ estimate(finding, ctx) {
2534
+ const fileMs = ((finding.fileCount ) ?? 0) * FILE_OPEN_OVERHEAD_MS;
2535
+ // A read's files are opened by the scan's tasks, in parallel: spread the per-file cost over
2536
+ // the most tasks its stages ran at once (91344 files x 10ms is 913s, claimed against a 117s
2537
+ // stage that ran 314 tasks at once). A write keeps the serial sum: the job commit moves each
2538
+ // output file on the driver, one after another.
2539
+ const stageIds = finding.stageIds ;
2540
+ let slots = 1;
2541
+ if (finding.direction === 'read') {
2542
+ for (const id of stageIds ?? []) slots = Math.max(slots, ctx.stages.get(id)?.peakConcurrentTasks ?? 1);
2543
+ }
2544
+ return stageMappableWasteOrCostOnly(fileMs / slots, stageIds, ctx);
2545
+ },
2546
+ }),
2547
+ defineSqlDetector({
2132
2548
  // Entry-level type is an identifier only; it never appears on an emitted finding. Findings
2133
2549
  // carry 'underBroadcast'/'overBroadcast' since one shared plan-walk covers both
2134
2550
  // opposite-direction rules (JoinToBroadcastAlert / BroadcastTooLargeAlert).
2135
- type: 'broadcastSizing', scope: 'sql', order: 132, fixEffort: 'config', version: 2,
2551
+ type: 'broadcastSizing', order: 132, fixEffort: 'config', version: 2,
2552
+ // Listed over-first: the two share order 132, and this list order is their display tie-break.
2553
+ emits: ['overBroadcast', 'underBroadcast'],
2136
2554
  docAnchor: '#bottleneck-broadcast-sizing',
2137
2555
  thresholds: {
2138
2556
  broadcastTiers: [10 * MB, 100 * MB, GB, 5 * GB],
2139
2557
  comparisonTiers: [10 * GB, 300 * GB, TB],
2140
2558
  overBroadcastBytes: GB,
2141
2559
  },
2142
- detect(
2143
-
2144
-
2145
-
2146
- sqlExec ,
2147
- ctx ,
2148
- ) {
2560
+ detect(sqlExec, ctx, thresholds) {
2149
2561
  if (!sqlExec.planTree) return null;
2150
- const { broadcastTiers, comparisonTiers, overBroadcastBytes } = this.thresholds;
2562
+ const { broadcastTiers, comparisonTiers, overBroadcastBytes } = thresholds;
2151
2563
  const fallbackStageIds = stageIdsForSqlExec(sqlExec.id, ctx.stages);
2152
2564
  const out = [];
2153
2565
  walkPlanTree(sqlExec.planTree, (node) => {
@@ -2188,12 +2600,60 @@ export const DETECTORS = [
2188
2600
  // resolvePlanTree always sets id; safe downstream of it.
2189
2601
  planNodeIds: [node.id ].filter(Boolean),
2190
2602
  impactBand: 'warning', metric: 'broadcastBytes', value: m.value,
2191
- recommendation: `This broadcast (${formatBytes(m.value)}) exceeds the 1 GB threshold: check for a misapplied broadcast hint or a misconfigured spark.sql.autoBroadcastJoinThreshold.`,
2603
+ recommendation: `This broadcast (${formatBytes(m.value)}) exceeds the ${binaryThresholdLabel(overBroadcastBytes)} threshold: check for a misapplied broadcast hint or a misconfigured spark.sql.autoBroadcastJoinThreshold.`,
2192
2604
  });
2193
2605
  }
2194
2606
  }
2195
2607
  });
2196
2608
  return out.length ? out : null;
2197
2609
  },
2198
- },
2199
- ];
2610
+ estimate(finding, ctx) {
2611
+ // Both finding types carry bytes as `value`: overBroadcast's broadcastBytes, underBroadcast's
2612
+ // smallerSideBytes (the smaller join side), each priced as one broadcast transfer.
2613
+ const wasteMs = (((finding.value ) ?? 0) / BROADCAST_BANDWIDTH_BPS) * 1000;
2614
+ return stageMappableWasteOrCostOnly(wasteMs, finding.stageIds , ctx);
2615
+ },
2616
+ }),
2617
+ ] ;
2618
+
2619
+ /** The entry that emits each finding type: the one whose estimate prices it and whose thresholds
2620
+ * and order describe it. Several entries can emit one type (the four configAudit audits), and the
2621
+ * first declared wins. */
2622
+ export const ENTRY_BY_TYPE = (() => {
2623
+ const byType = new Map ();
2624
+ for (const entry of DETECTORS ) {
2625
+ for (const type of entry.emits) if (!byType.has(type)) byType.set(type, entry);
2626
+ }
2627
+ return byType;
2628
+ })();
2629
+
2630
+ /** A `DETECTORS` entry's own `type`: every emitted finding type, plus broadcastSizing. */
2631
+
2632
+
2633
+ /** Every finding `type` a detector can emit, from the entries' `emits` lists. */
2634
+
2635
+
2636
+
2637
+
2638
+
2639
+
2640
+ /** The `thresholds` of the entry (or entries, for configAudit) that emit finding type `T`. */
2641
+
2642
+
2643
+ // `emits` already only names Finding members (Detector.emits); this makes the reverse hold too, so a
2644
+ // Finding member no detector emits, or a detector whose type has no Finding member, fails to compile.
2645
+
2646
+
2647
+
2648
+
2649
+ // A `suppressedBy` that names no entry would never suppress anything; this makes it a compile error.
2650
+ // An entry without one infers the bare `string` constraint, which contributes nothing here.
2651
+
2652
+
2653
+
2654
+
2655
+ /** Per-detector threshold overrides, keyed by entry `type`: each value a partial of that entry's
2656
+ * own thresholds. Built by threshold-overrides.ts from a user's config file. */
2657
+
2658
+
2659
+