sparkforensics-mcp 0.2.0 → 0.2.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (44) hide show
  1. package/package.json +5 -4
  2. package/vendor-core/cli/collect-run.js +3 -2
  3. package/vendor-core/cli/native-zstd.js +351 -0
  4. package/vendor-core/detectors.js +390 -75
  5. package/vendor-core/docs-config.js +34 -8
  6. package/vendor-core/docs-content/chapters/03-memory-model.md +39 -0
  7. package/vendor-core/docs-content/chapters/11-cluster-config.md +40 -0
  8. package/vendor-core/docs-content/detection/cache.md +4 -3
  9. package/vendor-core/docs-content/detection/chrn.md +4 -2
  10. package/vendor-core/docs-content/detection/gc.md +2 -0
  11. package/vendor-core/docs-content/detection/host.md +2 -1
  12. package/vendor-core/docs-content/detection/local.md +2 -3
  13. package/vendor-core/docs-content/detection/mem.md +3 -3
  14. package/vendor-core/docs-content/detection/plan.md +3 -1
  15. package/vendor-core/docs-content/detection/shape.md +2 -1
  16. package/vendor-core/docs-content/detection/shfl.md +2 -1
  17. package/vendor-core/docs-content/detection/spec.md +4 -3
  18. package/vendor-core/docs-content/detection/spill.md +1 -1
  19. package/vendor-core/docs-content/detection/strag.md +2 -1
  20. package/vendor-core/docs-content/detection/tiny.md +2 -1
  21. package/vendor-core/docs-content/tuning/failures.md +1 -1
  22. package/vendor-core/docs-content/tuning/gc.md +11 -4
  23. package/vendor-core/docs-content/tuning/shuffle.md +25 -5
  24. package/vendor-core/docs-content/tuning/skew.md +14 -6
  25. package/vendor-core/docs-content/tuning/small-files.md +12 -7
  26. package/vendor-core/docs-content/tuning/straggler.md +34 -0
  27. package/vendor-core/docs-content/tuning/tiny-tasks.md +1 -1
  28. package/vendor-core/docs-content/tuning/utilization.md +57 -7
  29. package/vendor-core/docs-content/upstream.json +4 -0
  30. package/vendor-core/event-handlers.js +321 -69
  31. package/vendor-core/event-schemas.js +8 -6
  32. package/vendor-core/evidence-report.js +4 -2
  33. package/vendor-core/impact-estimator.js +150 -34
  34. package/vendor-core/mcp-tools.js +20 -7
  35. package/vendor-core/occupancy.js +70 -2
  36. package/vendor-core/parser-worker.js +30 -14
  37. package/vendor-core/plan-summary.js +4 -0
  38. package/vendor-core/run-comparison.js +22 -17
  39. package/vendor-core/shs-fetch.js +18 -7
  40. package/vendor-core/shs-load.js +2 -1
  41. package/vendor-core/stage-quantiles.js +59 -3
  42. package/vendor-core/string-hash.js +15 -0
  43. package/vendor-core/types.js +9 -1
  44. package/vendor-core/vendor/fzstd.js +94 -18
@@ -3,8 +3,9 @@ import { scanRelationId } from './plan-summary.js';
3
3
  import { computePeakConcurrentCores, computePeakConcurrentExecutorCount } from './core-count.js';
4
4
  import { walkPlanTree } from './plan-tree-walk.js';
5
5
  import { computeCoreLocalityRatio } from './core-locality-ratio.js';
6
- import { estimateSingleStage, } from './occupancy.js';
6
+ import { estimateSingleStage, tailRecoveryMs, tailRemovedWorkMs, stragglerFixLongestTaskMs, } from './occupancy.js';
7
7
  import { isExchangeNode, isBroadcastExchangeNode } from './plan-node-detail.js';
8
+ import { cyrb53 } from './string-hash.js';
8
9
 
9
10
 
10
11
  const MB = 1024 * 1024;
@@ -74,6 +75,9 @@ const TB = 1024 * GB;
74
75
 
75
76
 
76
77
 
78
+
79
+
80
+
77
81
 
78
82
 
79
83
 
@@ -154,6 +158,7 @@ const TB = 1024 * GB;
154
158
 
155
159
 
156
160
 
161
+
157
162
 
158
163
 
159
164
  // SpillPressureDetector (5a) + SpillSkewDetector (5b).
@@ -206,8 +211,10 @@ export function unionStageIds(nodes , fallback ) {
206
211
 
207
212
  // Bottom-up shape computation for duplicate-subtree detection and the
208
213
  // cachingOpportunity composite detector. `size` is the subtree node count; the default
209
- // fingerprint encodes operator name + sorted metric NAMES (never values, per spec) + child
210
- // fingerprints, so two subtrees with the same shape but different values still collide.
214
+ // fingerprint encodes operator name + sorted metric NAMES (never values, per spec) + a digest
215
+ // of each child's fingerprint, so two subtrees with the same shape but different values still
216
+ // collide. Digesting children keeps each fingerprint O(own size): embedding the full child
217
+ // strings made every node carry its whole subtree, O(n x depth) text on deep real plans.
211
218
  //
212
219
  // opts.includeDetail (default false) folds root's own detail (via opts.normalizeDetail) into
213
220
  // ONLY root's fingerprint, never a child's: lets a caller match "this node's own detail + shape"
@@ -239,7 +246,7 @@ export function computePlanShapes(
239
246
  const childShapes = realChildren.map((c) => visit(c, false));
240
247
  const size = 1 + childShapes.reduce((sum, c) => sum + c.size, 0);
241
248
  const metricNames = (realMetrics ?? []).map((m) => m.name).sort().join(',');
242
- const childFingerprints = childShapes.map((c) => c.fingerprint).join(',');
249
+ const childFingerprints = childShapes.map((c) => cyrb53(c.fingerprint)).join(',');
243
250
  const fingerprint = isRoot && includeDetail
244
251
  ? `${node.name}[${metricNames}]<${normalize(node.detail ?? '')}>{${childFingerprints}}`
245
252
  : `${node.name}[${metricNames}]{${childFingerprints}}`;
@@ -261,7 +268,11 @@ export function normalizeDetail(detail ) {
261
268
  .replace(/,?\s*plan_id=\d+/g, '')
262
269
  .replace(/\[codegen id\s*:\s*\d+\]/gi, '[codegen id]')
263
270
  .replace(/\bBuild(Left|Right)\b/g, 'BuildSide');
264
- s = s.replace(/([A-Za-z_][\w.]*)\s*=\s*([A-Za-z_][\w.]*)/g, (_m, l, r) => {
271
+ // The lookbehind only skips starts inside an identifier, which can't match unless the
272
+ // identifier's own start already did: same output, without re-scanning each identifier from
273
+ // every one of its letters (O(length^2) on the 6.8 KB average join detail of the largest real
274
+ // log; 64ms -> 40ms per analyze()).
275
+ s = s.replace(/(?<![A-Za-z_])([A-Za-z_][\w.]*)\s*=\s*([A-Za-z_][\w.]*)/g, (_m, l, r) => {
265
276
  const [a, b] = [l, r].sort();
266
277
  return `${a} = ${b}`;
267
278
  });
@@ -270,6 +281,21 @@ export function normalizeDetail(detail ) {
270
281
 
271
282
  const JOIN_NAME_RE = /Join/i;
272
283
 
284
+ // scanRelationId per plan node, memoized: cachingOpportunity's relation walk and
285
+ // findCompositeCandidates both classify every node of every execution, and a JDBC scan's detail
286
+ // carries its whole inner SQL, so the repeated regex passes were a measurable share of
287
+ // analyze(). Keyed by node identity: a resolved plan tree is never mutated after the parser
288
+ // posts it.
289
+ const relationIdByNode = new WeakMap ();
290
+ function relationIdOf(node ) {
291
+ let rid = relationIdByNode.get(node);
292
+ if (rid === undefined) {
293
+ rid = scanRelationId(node.name ?? '', node.detail ?? '');
294
+ relationIdByNode.set(node, rid);
295
+ }
296
+ return rid;
297
+ }
298
+
273
299
  // Structural operator kind for cachingOpportunity's composite detection: 'join' covers every
274
300
  // Spark join physical operator; 'union' is Spark's exact `Union` node. CartesianProduct is
275
301
  // deliberately excluded (out of scope, as in plan-summary.ts).
@@ -302,11 +328,12 @@ export function findCompositeCandidates(root ) {
302
328
  path.pop();
303
329
 
304
330
  const metricNames = (node.metrics ?? []).map((m) => m.name).sort().join(',');
305
- const childFingerprints = childResults.map((r) => r.fingerprint).join(',');
331
+ // Child digests, as in computePlanShapes: deterministic, so still comparable across executions.
332
+ const childFingerprints = childResults.map((r) => cyrb53(r.fingerprint)).join(',');
306
333
  const fingerprint = `${node.name}[${metricNames}]{${childFingerprints}}`;
307
334
 
308
335
  const leafRelationBytes = new Map ();
309
- const rid = scanRelationId(node.name ?? '', node.detail ?? '');
336
+ const rid = relationIdOf(node);
310
337
  if (rid) {
311
338
  const bytesMetric = (node.metrics ?? []).find((m) => m.name === FILES_READ_BYTES);
312
339
  leafRelationBytes.set(rid, (leafRelationBytes.get(rid) ?? 0) + (bytesMetric ? bytesMetric.value : 0));
@@ -343,7 +370,7 @@ function firstLeafRelationId(node ) {
343
370
  let found = null;
344
371
  walkPlanTree(node, (n) => {
345
372
  if (found) return;
346
- found = scanRelationId(n.name ?? '', n.detail ?? '');
373
+ found = relationIdOf(n);
347
374
  });
348
375
  return found;
349
376
  }
@@ -411,6 +438,74 @@ export function findDuplicateSubtrees(
411
438
  return results;
412
439
  }
413
440
 
441
+ // True when every occurrence also agrees node-for-node on normalized detail (filters, columns,
442
+ // scanned table, literals). The fingerprint ignores detail, so occurrences can be same-shaped
443
+ // branches over different data; only identical ones are repeated work that computing once would
444
+ // save. AQE query-stage numbers (`ShuffleQueryStage 718` vs `720`) name the same computation's
445
+ // runtime stage and are ignored too. On the 14 real logs, 270 of 546 groups differed beyond that.
446
+ function occurrencesHaveIdenticalDetails(occurrences ) {
447
+ const detailsOf = (root ) => {
448
+ const out = [];
449
+ walkPlanTree(root, (n) => out.push(n.detail ?? ''));
450
+ return out;
451
+ };
452
+ const normalize = (d ) => normalizeDetail(d).replace(/\b(\w+QueryStage)\s+\d+/g, '$1');
453
+ const first = detailsOf(occurrences[0]);
454
+ const firstNormalized = [];
455
+ for (const other of occurrences.slice(1)) {
456
+ const details = detailsOf(other);
457
+ if (details.length !== first.length) return false;
458
+ for (let i = 0; i < first.length; i++) {
459
+ if (details[i] === first[i]) continue;
460
+ firstNormalized[i] ??= normalize(first[i]);
461
+ if (normalize(details[i]) !== firstNormalized[i]) return false;
462
+ }
463
+ }
464
+ return true;
465
+ }
466
+
467
+ // Stages that run nothing but the duplicated occurrences' operators: every operator attributed to
468
+ // the stage is inside `nodes`. A stage shared with operators outside them (the join consuming the
469
+ // subtree, the other join side) does other work too, so its time isn't the subtree's to claim;
470
+ // counting it also let sibling groups claim the same stage twice. A WholeStageCodegen wrapper and
471
+ // an Exchange's write half aren't other work (the fused pipeline around an operator, the shuffle
472
+ // write of its output), so they're left out of both counts; on the 14 real logs they were the
473
+ // only outside node on 426 stages.
474
+ function countsTowardStage(node ) {
475
+ return node.exchangeRole !== 'write' && !node.name.startsWith('WholeStageCodegen');
476
+ }
477
+
478
+ function operatorCountByStage(nodes ) {
479
+ const counts = new Map ();
480
+ for (const node of nodes) {
481
+ if (!countsTowardStage(node)) continue;
482
+ for (const sid of node.stageIds ?? []) counts.set(sid, (counts.get(sid) ?? 0) + 1);
483
+ }
484
+ return counts;
485
+ }
486
+
487
+ function stageOperatorShares(nodes , executionCounts ) {
488
+ const shares = {};
489
+ for (const [sid, count] of operatorCountByStage(nodes)) shares[sid] = count / (executionCounts.get(sid) ?? count);
490
+ return shares;
491
+ }
492
+
493
+ // Fingerprint matching compares operator + metric names only, not literal values or expr IDs
494
+ // (see the finding's validationRequired text), so a small pattern repeated the bare minimum
495
+ // number of times is the case most likely to be coincidental rather than real duplicated work.
496
+ // A bigger matched subtree, or more repeats, are each on their own strong corroborating evidence
497
+ // that the match is real: the odds of two semantically-different query branches producing an
498
+ // identical operator-name sequence shrink fast as the sequence grows or repeats.
499
+ function duplicateSubtreeConfidence(
500
+ subtreeSize ,
501
+ occurrences ,
502
+ thresholds ,
503
+ ) {
504
+ if (subtreeSize <= thresholds.minSubtreeSize && occurrences <= thresholds.minOccurrences) return 'low';
505
+ if (subtreeSize >= thresholds.minSubtreeSize * 2 || occurrences >= thresholds.minOccurrences + 2) return 'high';
506
+ return 'medium';
507
+ }
508
+
414
509
  // Exact metric names Spark emits, verified against a real SQLExecutionStart's sparkPlanInfo.
415
510
  // The write-side byte metric is "written output", not "size of written files".
416
511
  const FILES_READ_COUNT = 'number of files read';
@@ -505,12 +600,27 @@ function meetsRuntimeFloor(wasteMs , appDurationMs , floorP
505
600
  return appDurationMs == null || wasteMs >= appDurationMs * floorPct;
506
601
  }
507
602
 
603
+ // A stage that ran for less than floorPct of the run (a known duration): an estimate clipped to
604
+ // the stage can't reach floorPct, so every finding there grades info. A zero-length stage (no
605
+ // submission time on older Spark) is not skipped: it gets no estimate and keeps its own band.
606
+ function stageBelowRuntimeFloor(stage , ctx , floorPct ) {
607
+ const stageDurationMs = (stage.completedAt ?? 0) - (stage.submittedAt ?? 0);
608
+ const appDurationMs = computeAppDurationMs(ctx);
609
+ return appDurationMs != null && stageDurationMs > 0 && stageDurationMs < appDurationMs * floorPct;
610
+ }
611
+
508
612
  // Runs a raw waste delta through the same occupancy clip impact-estimator.ts applies before
509
613
  // display, so the runtime floor checks recoverable wall-clock, not a delta a physical floor
510
614
  // leaves unrecoverable. Falls back to the raw delta when occupancy data is unavailable.
511
- function clippedWasteMs(wasteMs , stageId , ctx ) {
615
+ // skew/straggler claims shorten the stage's longest task, hence shortensLongestTask (see occupancy.ts).
616
+ function clippedWasteMs(
617
+ wasteMs , stageId , ctx , removedCoreWorkMs , longestTaskAfterFixMs = 0,
618
+ ) {
512
619
  if (!ctx) return wasteMs;
513
- const est = estimateSingleStage(wasteMs, stageId, ctx.stages , ctx.occupancy);
620
+ const est = estimateSingleStage(
621
+ wasteMs, stageId, ctx.stages , ctx.occupancy,
622
+ { shortensLongestTask: true, removedCoreWorkMs, longestTaskAfterFixMs },
623
+ );
514
624
  return est ? est.wallClock.high : wasteMs;
515
625
  }
516
626
 
@@ -519,6 +629,19 @@ function clippedWasteMs(wasteMs , stageId , ctx )
519
629
  const CACHE_UTILIZATION_VALIDATION =
520
630
  "This ratio is a point-in-time storage snapshot from stage-submission events, not a runtime read-count. Confirm against the Spark UI's Storage tab before acting.";
521
631
 
632
+ // Both cachedRatio and diskRatio are percentages of an RDD's partition/byte count: with few
633
+ // partitions, one partition flipping cached/evicted (or disk-resident/memory-resident) swings the
634
+ // reported percentage by a large amount, so the point estimate is noisy. More partitions average
635
+ // that noise out into a stable ratio. numPartitions is the only sample-size signal
636
+ // DetectorRddInfo carries, so it drives confidence for both variants rather than a flat guess.
637
+ // 10/50 mirror this file's other "trust the sample" cutoffs (skew's minTasksForP95: 20,
638
+ // coreLocality's minTasks: 50).
639
+ function cacheSampleConfidence(numPartitions ) {
640
+ if (numPartitions < 10) return 'low';
641
+ if (numPartitions >= 50) return 'high';
642
+ return 'medium';
643
+ }
644
+
522
645
  function partialCacheFinding(rdd , cachedRatio , impactBand ) {
523
646
  const rddName = rdd.name || `RDD ${rdd.id}`;
524
647
  const cachedPct = Math.round(cachedRatio * 100);
@@ -527,7 +650,7 @@ function partialCacheFinding(rdd , cachedRatio , impactBa
527
650
  type: 'cacheUtilization', variant: 'partialCache', stageId: null,
528
651
  rddId: rdd.id, rddName,
529
652
  impactBand, metric: 'cachedRatio', value: cachedPct,
530
- confidence: 'medium',
653
+ confidence: cacheSampleConfidence(rdd.numPartitions),
531
654
  validationRequired: CACHE_UTILIZATION_VALIDATION,
532
655
  memorySize: rdd.memorySize, diskSize: rdd.diskSize,
533
656
  numCachedPartitions: rdd.numCachedPartitions, numPartitions: rdd.numPartitions,
@@ -542,7 +665,7 @@ function diskSpilloverFinding(rdd , diskRatio , impactBan
542
665
  type: 'cacheUtilization', variant: 'diskSpillover', stageId: null,
543
666
  rddId: rdd.id, rddName,
544
667
  impactBand, metric: 'diskRatio', value: diskPct,
545
- confidence: 'medium',
668
+ confidence: cacheSampleConfidence(rdd.numPartitions),
546
669
  validationRequired: CACHE_UTILIZATION_VALIDATION,
547
670
  memorySize: rdd.memorySize, diskSize: rdd.diskSize,
548
671
  numCachedPartitions: rdd.numCachedPartitions, numPartitions: rdd.numPartitions,
@@ -584,6 +707,103 @@ function diskSpilloverFinding(rdd , diskRatio , impactBan
584
707
  export const STRAGGLER_FLOOR_PCT_WARN = 0.005;
585
708
  export const STRAGGLER_FLOOR_PCT_CRIT = 0.02;
586
709
 
710
+ // A ratio just past ratioWarn is the case most likely to be ordinary task-duration variance
711
+ // rather than real skew; a ratio many multiples past it (a 50x P95/median vs. a 3.1x one) is
712
+ // unambiguous. 1.5x/5x mirror this file's other "just past the floor vs. clearly past it" splits
713
+ // (duplicateSubtreeConfidence's 2x, cacheSampleConfidence's 10/50-partition cutoffs).
714
+ function skewConfidence(ratio , ratioWarn ) {
715
+ if (ratio <= ratioWarn * 1.5) return 'low';
716
+ if (ratio >= ratioWarn * 5) return 'high';
717
+ return 'medium';
718
+ }
719
+
720
+ // warnPct100/lowInfoPct100 are this detector's own two thresholds; scale confidence as a multiple
721
+ // of whichever one gates the branch that fired, the same way skewConfidence scales off ratioWarn.
722
+ // High-GC: a pct just past warnPct100 (10%) is likely normal variance, 3x past it is unambiguous.
723
+ // Low-GC ("cost", over-provisioned): a pct just under lowInfoPct100 (5%) is borderline, a pct near
724
+ // zero is unambiguous idle GC.
725
+ function gcConfidence(
726
+ pct ,
727
+ thresholds ,
728
+ direction ,
729
+ ) {
730
+ if (direction === 'high') {
731
+ if (pct <= thresholds.warnPct100 * 1.5) return 'low';
732
+ if (pct >= thresholds.warnPct100 * 3) return 'high';
733
+ return 'medium';
734
+ }
735
+ if (pct >= thresholds.lowInfoPct100 * 0.66) return 'low';
736
+ if (pct <= thresholds.lowInfoPct100 * 0.2) return 'high';
737
+ return 'medium';
738
+ }
739
+
740
+ // warnFloor gates the finding, so a value just past it is the weakest evidence this detector can
741
+ // produce. highFloor is critPct: straggler's own second (speculative-share) tier, 2x warnPct
742
+ // (0.10 -> 0.20); reused as the high-confidence bar for the stragglerShare path too since that
743
+ // metric has no dedicated critical tier of its own (see the "no dedicated critical tier" comment
744
+ // on the straggler detector) but is the same 0-1 task-share magnitude.
745
+ function stragglerConfidence(shareValue , warnFloor , highFloor ) {
746
+ if (shareValue < warnFloor * 1.5) return 'low';
747
+ if (shareValue >= highFloor) return 'high';
748
+ return 'medium';
749
+ }
750
+
751
+ // minWasteMs is speculationWaste's own floor; a run that barely clears it (under 1.5x) is the
752
+ // weakest case, one that clears it several times over (4x, i.e. 4 minutes against a 1-minute
753
+ // floor) is unambiguous.
754
+ function speculationWasteConfidence(wastedMs , minWasteMs ) {
755
+ if (wastedMs <= minWasteMs * 1.5) return 'low';
756
+ if (wastedMs >= minWasteMs * 4) return 'high';
757
+ return 'medium';
758
+ }
759
+
760
+ // The finding already gates on wastedMBSeconds > wasteBufferMultiplier * usedMBSeconds, i.e. a
761
+ // ratio of 1 at the floor; scale confidence off that same ratio the way skewConfidence scales off
762
+ // ratioWarn, instead of introducing a second, unrelated multiplier.
763
+ function memoryWasteConfidence(wastedMBSeconds , usedMBSeconds , wasteBufferMultiplier ) {
764
+ const ratio = usedMBSeconds > 0 ? wastedMBSeconds / (wasteBufferMultiplier * usedMBSeconds) : Infinity;
765
+ if (ratio <= 1.5) return 'low';
766
+ if (ratio >= 3) return 'high';
767
+ return 'medium';
768
+ }
769
+
770
+ // Two independent weak spots can each undercut this finding: a ratio just past warnRatio (could
771
+ // be one bad stage), or too few sampled tasks (mirrors cacheSampleConfidence's use of
772
+ // numPartitions as a sample-size signal). Report whichever signal is weaker rather than
773
+ // averaging them away. critRatio is this detector's own existing second tier, reused directly as
774
+ // the ratio high-bar; minTasks*2/*4 mirror the same "2x floor is still weak, 4x is strong" spread
775
+ // used elsewhere in this file.
776
+ function coreLocalityConfidence(
777
+ ratio ,
778
+ totalTasks ,
779
+ thresholds ,
780
+ ) {
781
+ const rank = { low: 0, medium: 1, high: 2 } ;
782
+ const ratioTier = ratio < thresholds.warnRatio * 1.5 ? 'low' : ratio >= thresholds.critRatio ? 'high' : 'medium';
783
+ const sampleTier = totalTasks < thresholds.minTasks * 2 ? 'low' : totalTasks >= thresholds.minTasks * 4 ? 'high' : 'medium';
784
+ return rank[ratioTier] <= rank[sampleTier] ? ratioTier : sampleTier;
785
+ }
786
+
787
+ // warningPct/criticalPct are this detector's own two tiers (0.30/0.60); a churn rate just past
788
+ // warningPct is the borderline call the impact-band split already treats as the weaker tier, so
789
+ // reuse criticalPct directly as the high-confidence bar instead of inventing a third figure.
790
+ function autoscalingChurnConfidence(shortLivedPct , warningPct , criticalPct ) {
791
+ if (shortLivedPct <= warningPct * 1.5) return 'low';
792
+ if (shortLivedPct >= criticalPct) return 'high';
793
+ return 'medium';
794
+ }
795
+
796
+ // minExecutions is the bare minimum occurrence count this detector will even emit a finding for;
797
+ // a match at exactly that count is the weakest reuse signal (as likely to be coincidental overlap
798
+ // as real shared work), while 3x the floor is several independent executions all hitting the same
799
+ // relation/composite shape, unambiguous. Mirrors duplicateSubtreeConfidence's occurrences handling
800
+ // for the same reason: repetition count is the strength signal for a structural-match detector.
801
+ function cachingReuseConfidence(occurrences , minExecutions ) {
802
+ if (occurrences <= minExecutions) return 'low';
803
+ if (occurrences >= minExecutions * 3) return 'high';
804
+ return 'medium';
805
+ }
806
+
587
807
  export const DETECTORS = [
588
808
  {
589
809
  type: 'skew', scope: 'stage', order: 30, fixEffort: 'code', version: 1,
@@ -602,17 +822,18 @@ export const DETECTORS = [
602
822
  if (ratio <= this.thresholds.ratioWarn) return null;
603
823
  // Same absolute delta impact-estimator.ts's 'skew' case reports as savings; clipped the
604
824
  // same way before the floor check so the gate agrees with what's displayed.
605
- const wasteMs = Math.max(0, metric === 'P95/median' ? stage.taskDurationP95 - stage.taskDurationP50 : stage.taskDurationMax - stage.taskDurationP50);
825
+ const singleDelta = Math.max(0, metric === 'P95/median' ? stage.taskDurationP95 - stage.taskDurationP50 : stage.taskDurationMax - stage.taskDurationP50);
826
+ const wasteMs = tailRecoveryMs(stage, singleDelta);
606
827
  const appDurationMs = computeAppDurationMs(ctx);
607
- const floorWasteMs = clippedWasteMs(wasteMs, stage.id, ctx);
828
+ const floorWasteMs = clippedWasteMs(wasteMs, stage.id, ctx, tailRemovedWorkMs(stage, singleDelta), stragglerFixLongestTaskMs(stage));
608
829
  if (!meetsRuntimeFloor(floorWasteMs, appDurationMs, this.thresholds.floorPctWarn)) return null;
609
830
  const value = Math.round(ratio * 10) / 10;
610
831
  return {
611
832
  type: 'skew', stageId: stage.id,
612
833
  impactBand: 'warning',
613
834
  metric, value,
614
- confidence: 'low',
615
- validationRequired: 'The 0.5% runtime-floor percentage that gates this finding is our own noise floor, unvalidated: no external tool publishes an equivalent metric to calibrate against. Confirm against known-good/known-bad real logs before trusting the impact-band split.',
835
+ confidence: skewConfidence(ratio, this.thresholds.ratioWarn),
836
+ validationRequired: 'This finding is gated by a 0.5% runtime-floor threshold, our own noise floor for this metric.',
616
837
  recommendation: `Task duration ratio (${metric}) is ${value}×: for join-driven skew, enable AQE skew-join handling (spark.sql.adaptive.skewJoin.enabled); otherwise salt the key or repartition on a better key to reduce task skew.`,
617
838
  };
618
839
  },
@@ -620,9 +841,13 @@ export const DETECTORS = [
620
841
  {
621
842
  type: 'stageShape', scope: 'stage', order: 35, fixEffort: 'code', version: 1,
622
843
  docAnchor: '#bottleneck-stage-shape',
623
- thresholds: { pRatioMax: 0.5, oiRatioMax: 10, skewWarn: 3 },
844
+ // lowParallelismFloorPct: the same 0.5% runtime floor the tiered detectors use. Parallelizing a
845
+ // stage can't save more than the stage's own duration, so a shorter stage can't clear it; on
846
+ // the 14 real logs that was 2839 of 3005 lowParallelism findings (2168 on sub-second stages).
847
+ // App-wide idle capacity stays covered by utilization.
848
+ thresholds: { pRatioMax: 0.5, oiRatioMax: 10, skewWarn: 3, lowParallelismFloorPct: 0.005 },
624
849
  detect(
625
-
850
+
626
851
  stage ,
627
852
  ctx ,
628
853
  ) {
@@ -630,8 +855,9 @@ export const DETECTORS = [
630
855
  const execCount = (stage.executorStats ?? []).length;
631
856
  const cores = ctx?.app?.resources?.executor?.cores ?? 1;
632
857
  const totalCores = execCount * cores;
858
+ const stageDurationMs = (stage.completedAt ?? 0) - (stage.submittedAt ?? 0);
633
859
  // PRatio: under-parallelization.
634
- if (totalCores > 0) {
860
+ if (totalCores > 0 && !stageBelowRuntimeFloor(stage, ctx, this.thresholds.lowParallelismFloorPct)) {
635
861
  const pRatio = stage.taskCount / totalCores;
636
862
  if (pRatio < this.thresholds.pRatioMax) {
637
863
  out.push({
@@ -657,7 +883,6 @@ export const DETECTORS = [
657
883
  // TaskStageSkew: straggler cost vs stage wall-clock. Skip near-zero duration. Always info
658
884
  // like its siblings: this trigger forces the occupancy-clipped estimate to exactly zero on
659
885
  // every firing, so there's no wall-clock-backed tier left to gate on.
660
- const stageDurationMs = (stage.completedAt ?? 0) - (stage.submittedAt ?? 0);
661
886
  if (stageDurationMs > 0) {
662
887
  const ratio = stage.taskDurationMax / stageDurationMs;
663
888
  if (ratio > this.thresholds.skewWarn) {
@@ -676,13 +901,18 @@ export const DETECTORS = [
676
901
  {
677
902
  type: 'shuffle', scope: 'stage', order: 20, fixEffort: 'config', version: 1,
678
903
  docAnchor: '#bottleneck-shuffle',
679
- thresholds: { minBytes: 50 * MB },
904
+ // stageFloorPct: the 0.5% runtime floor. The shuffle claim is clipped to the stage, so on a
905
+ // shorter stage it graded info: 182 of 284 shuffle findings on the 14 real logs. The shuffle
906
+ // is still there on those stages; the floor is why they're dropped.
907
+ thresholds: { minBytes: 50 * MB, stageFloorPct: 0.005 },
680
908
  detect(
681
-
909
+
682
910
  stage ,
911
+ ctx ,
683
912
  ) {
684
913
  const bytes = stage.shuffleReadBytes;
685
914
  if (bytes <= this.thresholds.minBytes) return null;
915
+ if (stageBelowRuntimeFloor(stage, ctx, this.thresholds.stageFloorPct)) return null;
686
916
  return {
687
917
  type: 'shuffle', stageId: stage.id,
688
918
  impactBand: 'info',
@@ -693,7 +923,7 @@ export const DETECTORS = [
693
923
  },
694
924
  {
695
925
  type: 'partitionSizing', scope: 'stage', order: 22, fixEffort: 'config', version: 1,
696
- docAnchor: '#bottleneck-shuffle',
926
+ docAnchor: '#bottleneck-partition-sizing',
697
927
  thresholds: { skewRatio: 5, skewFloorBytes: 256 * MB, lowParTotalBytes: GB, lowParMaxTasks: 7, maxPartBytes: 5 * GB },
698
928
  detect(
699
929
 
@@ -742,9 +972,12 @@ export const DETECTORS = [
742
972
  {
743
973
  type: 'spill', scope: 'stage', order: 10, fixEffort: 'code', version: 1,
744
974
  docAnchor: '#bottleneck-spill',
745
- thresholds: { singleTaskDiskGiB: 1, singleTaskMemGiB: 4, highDiskGiB: 1, highTaskDiskMB: 512, highMemGiB: 4, medDiskMB: 256, medMemGiB: 1, skewRatio: 5, skewDiskFloorMB: 128, skewMemFloorMB: 256, skewMinTasks: 10 },
746
- detect( stage ) {
975
+ // stageFloorPct: the 0.5% runtime floor, as for shuffle (12 of 38 spill findings on the 14 real
976
+ // logs, all info). The spill is still there on those stages; the floor is why they're dropped.
977
+ thresholds: { singleTaskDiskGiB: 1, singleTaskMemGiB: 4, highDiskGiB: 1, highTaskDiskMB: 512, highMemGiB: 4, medDiskMB: 256, medMemGiB: 1, skewRatio: 5, skewDiskFloorMB: 128, skewMemFloorMB: 256, skewMinTasks: 10, stageFloorPct: 0.005 },
978
+ detect( stage , ctx ) {
747
979
  if (stage.memoryBytesSpilled === 0) return null;
980
+ if (stageBelowRuntimeFloor(stage, ctx, this.thresholds.stageFloorPct)) return null;
748
981
  const cls = stage.spillClassification;
749
982
  const classified = cls === 'skew' || cls === 'volume';
750
983
  const mag = computeSpillMagnitude(stage, this.thresholds);
@@ -766,21 +999,26 @@ export const DETECTORS = [
766
999
  {
767
1000
  type: 'gc', scope: 'stage', order: 50, fixEffort: 'config', version: 1,
768
1001
  docAnchor: '#bottleneck-gc',
769
- confidence: 'low',
770
- validationRequired: 'The 10-second minimum-runtime floor that gates this finding is our own noise floor, unvalidated: no external tool publishes an equivalent metric to calibrate against. Confirm against known-good/known-bad real logs before trusting the impact-band split.',
1002
+ validationRequired: 'This finding is gated by a 10-second minimum-runtime floor, our own noise floor for this metric.',
771
1003
  thresholds: {
772
1004
  warnPct100: 10,
773
1005
  // Descending tier: ExecutorGcHeuristic, ported as-is.
774
1006
  lowInfoPct100: 5,
775
1007
  // NOT SOURCED: our own noise floor so a stage that barely ran doesn't flag either direction.
776
1008
  minRunTimeMs: 10000,
1009
+ // lowInfoFloorPct: the 0.5% runtime floor stageShape's lowParallelism uses. The low-GC note is
1010
+ // an app-level memory-sizing signal; on a stage shorter than this share of the run it adds
1011
+ // nothing to that call. On the 14 real logs that was 464 of 685 low-GC notes (of 701 gc
1012
+ // findings); the low-GC pattern is still true on those stages, the floor is why they're dropped.
1013
+ lowInfoFloorPct: 0.005,
777
1014
  },
778
1015
  detect(
779
1016
 
780
-
781
-
1017
+
1018
+
782
1019
 
783
1020
  stage ,
1021
+ ctx ,
784
1022
  ) {
785
1023
  const pct = stage.gcPct;
786
1024
  if ((stage.executorRunTime ?? 0) >= this.thresholds.minRunTimeMs
@@ -790,19 +1028,20 @@ export const DETECTORS = [
790
1028
  type: 'gc', stageId: stage.id,
791
1029
  impactBand: 'warning',
792
1030
  metric: 'gcPct', value,
793
- confidence: this.confidence, validationRequired: this.validationRequired,
1031
+ confidence: gcConfidence(pct, this.thresholds, 'high'), validationRequired: this.validationRequired,
794
1032
  recommendation: `GC consumed ${value}% of executor run time: reduce object creation, use primitive types, avoid UDFs, increase executor memory.`,
795
1033
  };
796
1034
  }
797
1035
  // Low-GC (cost) branch: only for stages that ran long enough to be meaningful.
798
1036
  if ((stage.executorRunTime ?? 0) >= this.thresholds.minRunTimeMs
799
- && pct < this.thresholds.lowInfoPct100) {
1037
+ && pct < this.thresholds.lowInfoPct100
1038
+ && !stageBelowRuntimeFloor(stage, ctx, this.thresholds.lowInfoFloorPct)) {
800
1039
  const value = Math.round(pct * 10) / 10;
801
1040
  return {
802
1041
  type: 'gc', stageId: stage.id, direction: 'low',
803
1042
  impactBand: 'info',
804
1043
  metric: 'gcPct', value,
805
- confidence: this.confidence, validationRequired: this.validationRequired,
1044
+ confidence: gcConfidence(pct, this.thresholds, 'low'), validationRequired: this.validationRequired,
806
1045
  recommendation: `GC consumed only ${value}% of executor run time: memory may be over-provisioned; consider reducing spark.executor.memory for cost savings.`,
807
1046
  };
808
1047
  }
@@ -818,19 +1057,29 @@ export const DETECTORS = [
818
1057
  // stages, sub-second/sub-64MB host differences produce huge noise ratios. 1s is above
819
1058
  // per-task jitter but below genuine slow-host stages; 64MB mirrors the spill disk-skew floor.
820
1059
  floorMs: 1000, floorBytes: 64 * MB,
1060
+ // stageFloorPct: the tiered detectors' 0.5% runtime floor. On a stage shorter than this share
1061
+ // of the run a slow host can't cost that much: duration findings are clipped to the stage
1062
+ // and graded info, and a byte-dimension imbalance (no time estimate, so its ratio tier was
1063
+ // its band) was graded info here since 6298149. So the stage is skipped: on the 14 real logs
1064
+ // 324 of 452 slowHost findings, all info. The imbalance is still there on those stages; the
1065
+ // floor is why they're dropped.
1066
+ stageFloorPct: 0.005,
821
1067
  },
822
1068
  detect(
823
1069
 
824
1070
 
825
1071
 
826
1072
 
1073
+
827
1074
 
828
1075
 
829
1076
  stage ,
1077
+ ctx ,
830
1078
  ) {
831
1079
  const hosts = stage.hostStats ?? [];
832
1080
  const execs0 = stage.executorStats ?? [];
833
1081
  if ((hosts.length < this.thresholds.minHosts && execs0.length < this.thresholds.minHosts) || stage.taskCount < this.thresholds.minTasks) return null;
1082
+ if (stageBelowRuntimeFloor(stage, ctx, this.thresholds.stageFloorPct)) return null;
834
1083
  const out = [];
835
1084
  if (hosts.length >= this.thresholds.minHosts) {
836
1085
  const means = hosts.map(h => ({ host: h.host, taskCount: h.taskCount, mean: h.totalDuration / h.taskCount }));
@@ -985,31 +1234,51 @@ export const DETECTORS = [
985
1234
  docAnchor: '#bottleneck-straggler',
986
1235
  // floorPctWarn/floorPctCrit are re-exported as STRAGGLER_FLOOR_PCT_WARN/CRIT and reused as
987
1236
  // impact-band.ts's global noise floor: keep the two in sync.
988
- thresholds: { minTasks: 10, shareWarn: 0.05, warnPct: 0.10, critPct: 0.20, floorPctWarn: STRAGGLER_FLOOR_PCT_WARN, floorPctCrit: STRAGGLER_FLOOR_PCT_CRIT },
1237
+ // shareWarnAtFloor: in a large stage, the few stragglers that gate it for tens of seconds can be
1238
+ // only 2.5-5% of its tasks. Scored against a task-level replay of every stage on 14 real logs
1239
+ // (recoverable = replay with each task over 4x P50 capped at P50), admitting 2.5-5% shares
1240
+ // whose clipped tail already clears floorPctWarn found 3 such stages (10-48s) for 1 borderline
1241
+ // miss; admitting every 2.5% share instead added 86 findings below the floor.
1242
+ // A stage shorter than floorPctWarn of the run is skipped outright: its tail can't cost more
1243
+ // than the stage's own duration, so every finding there graded info. On the 14 real logs that
1244
+ // was 671 of 753 straggler findings, none above info; the slow tail is still real on those
1245
+ // stages, the floor is why they're dropped.
1246
+ thresholds: { minTasks: 10, shareWarn: 0.05, shareWarnAtFloor: 0.025, warnPct: 0.10, critPct: 0.20, floorPctWarn: STRAGGLER_FLOOR_PCT_WARN, floorPctCrit: STRAGGLER_FLOOR_PCT_CRIT },
989
1247
  detect(
990
1248
 
991
-
1249
+
1250
+
1251
+
1252
+
992
1253
 
993
1254
  stage ,
994
1255
  ctx ,
995
1256
  ) {
996
1257
  if (stage.taskCount < this.thresholds.minTasks) return null;
1258
+ const appDurationMs = computeAppDurationMs(ctx);
1259
+ if (stageBelowRuntimeFloor(stage, ctx, this.thresholds.floorPctWarn)) return null;
997
1260
  const stragglerShare = (stage.stragglerCount ?? 0) / stage.taskCount;
998
- if ((stage.speculativeTasks ?? 0) === 0 && stragglerShare <= this.thresholds.shareWarn) return null;
999
1261
  const useSpeculative = (stage.speculativeTasks ?? 0) > 0;
1262
+ if (!useSpeculative && stragglerShare <= this.thresholds.shareWarnAtFloor) return null;
1000
1263
  const speculativeShare = useSpeculative ? stage.speculativeTasks / stage.taskCount : 0;
1001
1264
  // Same absolute delta impact-estimator.ts's straggler/stageShape case reports as savings: a
1002
1265
  // high straggler/speculative share on a stage whose tasks barely vary models near-zero
1003
1266
  // savings, so it must not outrank 'info'. Clipped the same way before the floor check.
1004
- const wasteMs = Math.max(0, stage.taskDurationMax - stage.taskDurationP50);
1005
- const appDurationMs = computeAppDurationMs(ctx);
1006
- const floorWasteMs = clippedWasteMs(wasteMs, stage.id, ctx);
1267
+ const longestTaskAfterFixMs = stragglerFixLongestTaskMs(stage);
1268
+ const singleDelta = Math.max(0, stage.taskDurationMax - longestTaskAfterFixMs);
1269
+ const wasteMs = tailRecoveryMs(stage, singleDelta);
1270
+ const floorWasteMs = clippedWasteMs(wasteMs, stage.id, ctx, tailRemovedWorkMs(stage, singleDelta), longestTaskAfterFixMs);
1007
1271
  const meetsWarnFloor = meetsRuntimeFloor(floorWasteMs, appDurationMs, this.thresholds.floorPctWarn);
1008
1272
  const meetsCritFloor = meetsRuntimeFloor(floorWasteMs, appDurationMs, this.thresholds.floorPctCrit);
1273
+ // The lower gate needs positive evidence the tail matters: meetsRuntimeFloor passes by
1274
+ // default when the app's duration is unknown (an incomplete run), which isn't that.
1275
+ const stragglerShareFires = stragglerShare > this.thresholds.shareWarn
1276
+ || (stragglerShare > this.thresholds.shareWarnAtFloor && appDurationMs != null && meetsWarnFloor);
1277
+ if (!useSpeculative && !stragglerShareFires) return null;
1009
1278
  const speculativeTier = speculativeShare >= this.thresholds.critPct && meetsCritFloor ? 'critical'
1010
1279
  : speculativeShare >= this.thresholds.warnPct && meetsWarnFloor ? 'warning' : 'info';
1011
1280
  // Straggler share has no dedicated critical tier per detector-contract.md; only warning.
1012
- const stragglerTier = stragglerShare > this.thresholds.shareWarn && meetsWarnFloor ? 'warning' : 'info';
1281
+ const stragglerTier = stragglerShareFires && meetsWarnFloor ? 'warning' : 'info';
1013
1282
  // Fixed fallback: overwritten by deriveImpactBand when this finding gets a real wallClock
1014
1283
  // estimate (the common case). Only surfaces on the rare occupancy-sweep miss.
1015
1284
  const impactBand = 'info';
@@ -1028,20 +1297,21 @@ export const DETECTORS = [
1028
1297
  unit: useSpeculativeMetric ? 'count' : 'pct',
1029
1298
  speculativeTasks: stage.speculativeTasks ?? 0,
1030
1299
  stragglerCount: stage.stragglerCount ?? 0,
1031
- confidence: 'low',
1032
- validationRequired: 'The 0.5%/2% runtime-floor percentages that gate this finding are our own noise floor, unvalidated: no external tool publishes an equivalent metric to calibrate against. Confirm against known-good/known-bad real logs before trusting the impact-band split.',
1300
+ confidence: useSpeculativeMetric
1301
+ ? stragglerConfidence(speculativeShare, this.thresholds.warnPct, this.thresholds.critPct)
1302
+ : stragglerConfidence(stragglerShare, this.thresholds.shareWarn, this.thresholds.critPct),
1303
+ validationRequired: 'This finding is gated by 0.5%/2% runtime-floor thresholds, our own noise floor for this metric.',
1033
1304
  recommendation: `${detail}: rule out a GC pause or a slow shuffle fetch before assuming a hardware issue; if a skewed key is the real cause, that's a candidate for AQE's skew-join handling.`,
1034
1305
  };
1035
1306
  },
1036
1307
  },
1037
1308
  {
1038
1309
  type: 'speculationWaste', scope: 'stage', order: 71, fixEffort: 'config', version: 1,
1039
- docAnchor: '#bottleneck-straggler', confidence: 'low',
1310
+ docAnchor: '#bottleneck-speculation-waste',
1040
1311
  thresholds: { minWasted: 5, minWasteMs: 60000 },
1041
1312
  detect(
1042
1313
 
1043
1314
 
1044
-
1045
1315
 
1046
1316
  stage ,
1047
1317
  ) {
@@ -1052,7 +1322,7 @@ export const DETECTORS = [
1052
1322
  type: 'speculationWaste', stageId: stage.id,
1053
1323
  impactBand: 'warning',
1054
1324
  metric: 'speculationWasteMs', value: wastedMs,
1055
- confidence: this.confidence,
1325
+ confidence: speculationWasteConfidence(wastedMs, this.thresholds.minWasteMs),
1056
1326
  recommendation: `Speculative execution discarded ${Math.round(wastedMs / 1000)}s of executor time in this stage; if task durations are naturally variable rather than genuine stragglers, consider tuning spark.speculation.multiplier/quantile.`,
1057
1327
  };
1058
1328
  },
@@ -1083,12 +1353,17 @@ export const DETECTORS = [
1083
1353
  {
1084
1354
  type: 'tinyTask', scope: 'stage', order: 80, fixEffort: 'code', version: 1,
1085
1355
  docAnchor: '#bottleneck-tiny-tasks',
1086
- thresholds: { minTasks: 100, maxP50: 500, maxP95: 1000 },
1356
+ // stageFloorPct: the tiered detectors' 0.5% runtime floor. Coalescing can't save more than the
1357
+ // stage's own duration, so on a shorter stage every finding graded info: on the 14 real logs
1358
+ // 132 of 151 tinyTask findings. The tasks are still tiny there; the floor is why they're dropped.
1359
+ thresholds: { minTasks: 100, maxP50: 500, maxP95: 1000, stageFloorPct: 0.005 },
1087
1360
  detect(
1088
-
1361
+
1089
1362
  stage ,
1363
+ ctx ,
1090
1364
  ) {
1091
1365
  if (stage.taskCount < this.thresholds.minTasks) return null;
1366
+ if (stageBelowRuntimeFloor(stage, ctx, this.thresholds.stageFloorPct)) return null;
1092
1367
  if (stage.taskDurationP50 > this.thresholds.maxP50 || stage.taskDurationP95 > this.thresholds.maxP95) return null;
1093
1368
  const coalesceTo = Math.max(1, Math.round(stage.taskCount / 10));
1094
1369
  const fix = stage.shuffleReadBytes > 0
@@ -1124,23 +1399,42 @@ export const DETECTORS = [
1124
1399
 
1125
1400
  ctx ,
1126
1401
  ) {
1127
- const { app, stages } = ctx;
1402
+ const { app, stages, executorsAdded, executorsRemoved } = ctx;
1128
1403
  // Nullish (not falsy) check: a literal startTime:0 must not be treated as "missing".
1129
1404
  if (!app || app.startTime == null || stages.size === 0) return null;
1130
- let firstTaskLaunch = Infinity;
1405
+ let firstStageSubmitted = Infinity;
1131
1406
  for (const stage of stages.values()) {
1132
- if (stage.submittedAt > 0 && stage.submittedAt < firstTaskLaunch) firstTaskLaunch = stage.submittedAt;
1407
+ if (stage.submittedAt > 0 && stage.submittedAt < firstStageSubmitted) firstStageSubmitted = stage.submittedAt;
1133
1408
  }
1134
1409
  // No stage ever recorded a submission timestamp: no basis to measure a startup gap against.
1135
- // Exposed now that a literal app.startTime:0 no longer short-circuits this detector entirely.
1136
- if (!Number.isFinite(firstTaskLaunch)) return null;
1137
- const gapSeconds = (firstTaskLaunch - app.startTime) / 1000;
1410
+ if (!Number.isFinite(firstStageSubmitted)) return null;
1411
+ // The wait is from the first runnable stage to the first executor, not from app start: the
1412
+ // driver's own startup before its first job (36-47s on every real log, whatever the
1413
+ // executors did) isn't something executors could shorten. On the 9 real logs that fired,
1414
+ // the first executor arrived 146-192s after the first stage on two whose old gap read ~40s,
1415
+ // and 2s after it on one the old gap flagged critical at 42s.
1416
+ // An executor added before the first stage only counts if it was still alive at submission:
1417
+ // with dynamic allocation scaling to zero, early executors can idle out before any job runs.
1418
+ const removedAt = new Map ();
1419
+ for (const ev of executorsRemoved) removedAt.set(ev.executorId, ev.timestamp);
1420
+ let firstExecutorAdded = Infinity;
1421
+ for (const e of executorsAdded) {
1422
+ if (!(e.timestamp > 0)) continue;
1423
+ if (e.timestamp <= firstStageSubmitted) {
1424
+ const removed = removedAt.get(e.executorId);
1425
+ if (removed == null || removed > firstStageSubmitted) return null;
1426
+ } else if (e.timestamp < firstExecutorAdded) {
1427
+ firstExecutorAdded = e.timestamp;
1428
+ }
1429
+ }
1430
+ if (!Number.isFinite(firstExecutorAdded)) return null;
1431
+ const gapSeconds = (firstExecutorAdded - firstStageSubmitted) / 1000;
1138
1432
  if (gapSeconds <= this.thresholds.gapSeconds) return null;
1139
1433
  const value = Math.round(gapSeconds);
1140
1434
  return {
1141
1435
  type: 'coldStart', stageId: null, impactBand: 'warning',
1142
1436
  metric: 'startupGapSeconds', value,
1143
- recommendation: `The first task waited ${value}s for executors to become available: keep a warm pool of idle executors, or if using dynamic allocation, raise the minimum/initial executor count so it doesn't scale up from zero.`,
1437
+ recommendation: `The first stage waited ${value}s for an executor to become available: keep a warm pool of idle executors, or if using dynamic allocation, raise the minimum/initial executor count so it doesn't scale up from zero.`,
1144
1438
  };
1145
1439
  },
1146
1440
  },
@@ -1298,8 +1592,8 @@ export const DETECTORS = [
1298
1592
  out.push({
1299
1593
  type: 'memoryUtilization', variant: 'wasteModel', stageId: null,
1300
1594
  impactBand: 'info', metric: 'wastedMBSeconds', value,
1301
- confidence: 'low',
1302
- validationRequired: 'Memory-waste estimate uses allocated-vs-used memory-time and an unverified 1.5x buffer: confirm against the Spark UI before acting.',
1595
+ confidence: memoryWasteConfidence(wastedMBSeconds, usedMBSeconds, this.thresholds.wasteBufferMultiplier),
1596
+ validationRequired: 'Memory-waste estimate uses allocated-vs-used memory-time and a 1.5x buffer: confirm against the Spark UI before acting.',
1303
1597
  recommendation: `Allocated executor memory sat largely idle over the run (~${value.toLocaleString('en-US')} MB-seconds wasted): review spark.executor.memory and executor count.`,
1304
1598
  });
1305
1599
  }
@@ -1313,7 +1607,7 @@ export const DETECTORS = [
1313
1607
  // block-access events, so a literal cache hit rate isn't derivable). Two per-RDD tiered
1314
1608
  // checks over rddInfo snapshots: partial caching and disk spillover. An RDD can produce both.
1315
1609
  type: 'cacheUtilization', scope: 'app', order: 103, fixEffort: 'code', version: 1,
1316
- docAnchor: '#memory-model',
1610
+ docAnchor: '#bottleneck-cache-utilization',
1317
1611
  thresholds: {
1318
1612
  cachedRatioWarn: 0.50, cachedRatioInfo: 0.90,
1319
1613
  diskRatioWarn: 0.40, diskRatioInfo: 0.15,
@@ -1357,7 +1651,7 @@ export const DETECTORS = [
1357
1651
  // half of the "Wasted Cores Ratio" (idle-core half is memoryUtilization's idleCores).
1358
1652
  // NO_PREF stays in the denominator only: shuffle-read stages legitimately report it.
1359
1653
  type: 'coreLocality', scope: 'app', order: 103, fixEffort: 'config', version: 1,
1360
- docAnchor: '#bottleneck-utilization',
1654
+ docAnchor: '#bottleneck-core-locality',
1361
1655
  thresholds: { minTasks: 50, warnRatio: 0.15, critRatio: 0.35 },
1362
1656
  detect(
1363
1657
 
@@ -1376,8 +1670,8 @@ export const DETECTORS = [
1376
1670
  metric: 'nonLocalRatio', value,
1377
1671
  // Raw count behind the ratio, for the impact estimator. Non-null whenever totalTasks is.
1378
1672
  nonLocalTaskCount: nonLocalTasks ,
1379
- confidence: 'low',
1380
- validationRequired: 'The 15%/35% non-local-ratio thresholds (and the 50-task minimum) are unvalidated design-spike values: no external tool publishes an equivalent metric to calibrate against. Confirm against known-good/known-bad real logs before trusting the impact-band split.',
1673
+ confidence: coreLocalityConfidence(ratio , totalTasks, this.thresholds),
1674
+ validationRequired: 'This finding is gated by 15%/35% non-local-ratio thresholds (and a 50-task minimum), our own noise floor for this metric.',
1381
1675
  recommendation: `${value}% of tasks (${nonLocalTasks }) ran without process- or node-local data placement: check spark.locality.wait settings and executor/data colocation.`,
1382
1676
  };
1383
1677
  },
@@ -1387,12 +1681,11 @@ export const DETECTORS = [
1387
1681
  // re-provisioning, not normal scale-down). Reuses utilization's add/remove matching, but
1388
1682
  // measures lifetime against a threshold instead of aggregate active-time.
1389
1683
  type: 'autoscalingChurn', scope: 'app', order: 103, fixEffort: 'config', version: 1,
1390
- confidence: 'low',
1684
+ docAnchor: '#bottleneck-autoscaling-churn',
1391
1685
  thresholds: { shortLivedMs: 120_000, warningPct: 0.30, criticalPct: 0.60, minExecutors: 5 },
1392
1686
  detect(
1393
1687
 
1394
1688
 
1395
-
1396
1689
 
1397
1690
  ctx ,
1398
1691
  ) {
@@ -1421,7 +1714,7 @@ export const DETECTORS = [
1421
1714
  metric: 'shortLivedExecutorPct', value: pct,
1422
1715
  // Raw count behind the percentage, for the impact estimator's startup-overhead figure.
1423
1716
  shortLivedExecutorCount: shortLivedCount,
1424
- confidence: this.confidence,
1717
+ confidence: autoscalingChurnConfidence(shortLivedPct, this.thresholds.warningPct, this.thresholds.criticalPct),
1425
1718
  recommendation: `${pct}% of executors ran for under 2 minutes before being removed. This looks like wasteful re-provisioning rather than normal scale-down; consider raising spark.dynamicAllocation.executorIdleTimeout or widening the minExecutors/maxExecutors bounds to reduce flapping.`,
1426
1719
  };
1427
1720
  },
@@ -1430,7 +1723,7 @@ export const DETECTORS = [
1430
1723
  // Cross-execution relation reuse: flags an input relation scanned by two or more SQL
1431
1724
  // executions in one run, firing on real relation names (parquet:..., jdbc:...).
1432
1725
  type: 'cachingOpportunity', scope: 'app', order: 105, fixEffort: 'code', version: 1,
1433
- docAnchor: '#bottleneck-utilization',
1726
+ docAnchor: '#bottleneck-caching-opportunity',
1434
1727
  thresholds: { minExecutions: 2 },
1435
1728
  detect( ctx ) {
1436
1729
  const sql = ctx.sql;
@@ -1457,7 +1750,7 @@ export const DETECTORS = [
1457
1750
  // Dedupe relations within one execution (self-joins count once), summing read bytes per relation.
1458
1751
  const perExec = new Map ();
1459
1752
  walkPlanTree(exec.planTree, (node) => {
1460
- const rid = scanRelationId(node.name ?? '', node.detail ?? '');
1753
+ const rid = relationIdOf(node);
1461
1754
  if (!rid) return;
1462
1755
  const bytesMetric = (node.metrics ?? []).find(m => m.name === FILES_READ_BYTES);
1463
1756
  perExec.set(rid, (perExec.get(rid) ?? 0) + (bytesMetric ? bytesMetric.value : 0));
@@ -1562,7 +1855,7 @@ export const DETECTORS = [
1562
1855
  metric: 'executionReuse', value,
1563
1856
  format: 'derived', relations, operator: agg.operator, relation: relationDisplay,
1564
1857
  executionIds: finalExecutionIds, totalReadBytes,
1565
- confidence: 'low',
1858
+ confidence: cachingReuseConfidence(value, this.thresholds.minExecutions),
1566
1859
  validationRequired:
1567
1860
  'Composite reuse is inferred from a structural plan-shape match (operator + normalized ' +
1568
1861
  'join/filter condition + child shapes) across SQL executions; confirm these executions ' +
@@ -1593,7 +1886,7 @@ export const DETECTORS = [
1593
1886
  relation: agg.relation, format: agg.format,
1594
1887
  executionIds: residualExecutionIds.sort((a, b) => a - b),
1595
1888
  totalReadBytes,
1596
- confidence: 'low',
1889
+ confidence: cachingReuseConfidence(value, this.thresholds.minExecutions),
1597
1890
  validationRequired:
1598
1891
  'Relation-reuse is inferred from the pre-AQE plan scan identity across SQL ' +
1599
1892
  'executions; confirm the reads are the same data and cacheable within one ' +
@@ -1721,9 +2014,13 @@ export const DETECTORS = [
1721
2014
  {
1722
2015
  type: 'duplicatePlanSubtree', scope: 'sql', order: 130, fixEffort: 'code', version: 2,
1723
2016
  docAnchor: '#bottleneck-duplicate-plan-subtree',
1724
- thresholds: { minSubtreeSize: 3, minOccurrences: 2 },
2017
+ // stageFloorPct: the 0.5% runtime floor. The claim counts at most each linked stage's own
2018
+ // task-active time, so a repeat whose stages together lasted less than this share of the run
2019
+ // graded info: 340 of 545 findings on the 14 real logs. The repeat is still in the plan; the
2020
+ // floor is why they're dropped. A repeat with no linked stage time is kept.
2021
+ thresholds: { minSubtreeSize: 3, minOccurrences: 2, stageFloorPct: 0.005 },
1725
2022
  detect(
1726
-
2023
+
1727
2024
  sqlExec ,
1728
2025
  ctx ,
1729
2026
  ) {
@@ -1731,26 +2028,44 @@ export const DETECTORS = [
1731
2028
  const groups = findDuplicateSubtrees(sqlExec.planTree, this.thresholds);
1732
2029
  if (groups.length === 0) return null;
1733
2030
  const fallbackStageIds = stageIdsForSqlExec(sqlExec.id, ctx.stages);
1734
- return groups.map((g) => {
2031
+ const executionNodes = [];
2032
+ walkPlanTree(sqlExec.planTree, (node) => executionNodes.push(node));
2033
+ const operatorsByStage = operatorCountByStage(executionNodes);
2034
+ const appDurationMs = computeAppDurationMs(ctx);
2035
+ const findings = groups.map((g) => {
1735
2036
  const nodes = [];
1736
2037
  for (const n of g.nodes) walkPlanTree(n, (node) => nodes.push(node));
1737
2038
  const stageIds = unionStageIds(nodes, fallbackStageIds);
2039
+ let stagesMs = 0;
2040
+ for (const id of stageIds) {
2041
+ const stage = ctx.stages.get(id);
2042
+ if (stage) stagesMs += Math.max(0, (stage.completedAt ?? 0) - (stage.submittedAt ?? 0));
2043
+ }
2044
+ if (appDurationMs != null && stagesMs > 0 && stagesMs < appDurationMs * this.thresholds.stageFloorPct) return null;
2045
+ const occurrencesIdentical = occurrencesHaveIdenticalDetails(g.nodes);
2046
+ const stageShares = stageOperatorShares(nodes, operatorsByStage);
1738
2047
  // resolvePlanTree always sets id; safe downstream of it.
1739
2048
  const planNodeIds = nodes.map((n) => n.id ).filter(Boolean);
1740
2049
  const touching = g.sampleRelation ? ` (touching ${g.sampleRelation})` : '';
2050
+ const differing = occurrencesIdentical ? '' : ' Their filters, columns or scanned tables differ, so the repeats may compute different data.';
1741
2051
  return {
1742
2052
  type: 'duplicatePlanSubtree', executionId: sqlExec.id, stageIds, planNodeIds,
2053
+ stageShares, occurrencesIdentical,
1743
2054
  // Fixed fallback: overwritten by deriveImpactBand when this finding gets a real
1744
- // wallClock estimate (the common case). Only surfaces on the rare occupancy-sweep miss.
1745
- impactBand: 'warning', metric: 'subtreeOccurrences', value: g.occurrences,
2055
+ // wallClock estimate. Repeats with differing details get no estimate (nothing is
2056
+ // known to be recomputed), and neither does a subtree with no stage of its own.
2057
+ impactBand: occurrencesIdentical && Object.keys(stageShares).length > 0 ? 'warning' : 'info',
2058
+ metric: 'subtreeOccurrences', value: g.occurrences,
1746
2059
  rootName: g.rootName, subtreeSize: g.subtreeSize, sampleRelation: g.sampleRelation,
1747
- groupIndex: g.groupIndex, confidence: 'medium',
2060
+ groupIndex: g.groupIndex,
2061
+ confidence: occurrencesIdentical ? duplicateSubtreeConfidence(g.subtreeSize, g.occurrences, this.thresholds) : 'low',
1748
2062
  validationRequired: 'Duplicate-subtree matching compares operator names and metric names only, not literal values or expr IDs: confirm the repeated work is real in the Spark SQL plan tab before acting.',
1749
- recommendation: g.isExchangeRoot
2063
+ recommendation: (g.isExchangeRoot
1750
2064
  ? `A ${g.subtreeSize}-node subtree rooted at ${pathBasename(g.rootName)} repeats ${g.occurrences}x in this plan${touching}: this looks like a possible missed exchange reuse; check whether the same shuffle could be computed once and reused.`
1751
- : `A ${g.subtreeSize}-node subtree rooted at ${pathBasename(g.rootName)} repeats ${g.occurrences}x in this plan${touching}: consider caching/persisting the shared computation or check for a duplicated query branch.`,
2065
+ : `A ${g.subtreeSize}-node subtree rooted at ${pathBasename(g.rootName)} repeats ${g.occurrences}x in this plan${touching}: consider caching/persisting the shared computation or check for a duplicated query branch.`) + differing,
1752
2066
  };
1753
- });
2067
+ }).filter((f) => f !== null);
2068
+ return findings.length > 0 ? findings : null;
1754
2069
  },
1755
2070
  },
1756
2071
  {