sparkforensics-cli 0.2.0 → 0.2.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (107) hide show
  1. package/export-template/docs/404.html +1 -1
  2. package/export-template/docs/assets/{app.CndaAS6v.js → app.DQTZyGL1.js} +1 -1
  3. package/export-template/docs/assets/chunks/@localSearchIndexroot.DppnXnDE.js +1 -0
  4. package/export-template/docs/assets/chunks/{VPLocalSearchBox.yJbZbsEo.js → VPLocalSearchBox.BkBIPFs6.js} +1 -1
  5. package/export-template/docs/assets/chunks/theme.DP0u1AUq.js +2 -0
  6. package/export-template/docs/assets/contributor-guide_architecture_board-widgets.md.CWpj01WU.js +1 -0
  7. package/export-template/docs/assets/contributor-guide_architecture_board-widgets.md.CWpj01WU.lean.js +1 -0
  8. package/export-template/docs/assets/contributor-guide_architecture_detector-contract.md.CgzUsQ6W.js +1 -0
  9. package/export-template/docs/assets/{contributor-guide_architecture_impact-estimation.md.DYCDPgkh.js → contributor-guide_architecture_impact-estimation.md.CooslVJt.js} +1 -1
  10. package/export-template/docs/assets/contributor-guide_architecture_state-and-history.md.C-xxn0q7.js +1 -0
  11. package/export-template/docs/assets/contributor-guide_architecture_state-and-history.md.C-xxn0q7.lean.js +1 -0
  12. package/export-template/docs/assets/contributor-guide_architecture_widget-rendering.md.R27gQrgY.js +1 -0
  13. package/export-template/docs/assets/contributor-guide_architecture_worker-protocol.md.IbnfNrV3.js +6 -0
  14. package/export-template/docs/assets/{style.DXOMCXxn.css → style.DSixAiZE.css} +1 -1
  15. package/export-template/docs/assets/{user-guide_alternative-log-retrieval.md.sU3KGarf.js → user-guide_alternative-log-retrieval.md.B4tPGIal.js} +1 -1
  16. package/export-template/docs/assets/{user-guide_alternative-log-retrieval.md.sU3KGarf.lean.js → user-guide_alternative-log-retrieval.md.B4tPGIal.lean.js} +1 -1
  17. package/export-template/docs/assets/user-guide_getting-started.md.BJvwLEIM.js +3 -0
  18. package/export-template/docs/assets/user-guide_getting-started.md.BJvwLEIM.lean.js +1 -0
  19. package/export-template/docs/assets/{user-guide_mcp-tools.md.C8MiIu7F.js → user-guide_mcp-tools.md.Vi3RoflJ.js} +3 -3
  20. package/export-template/docs/assets/{user-guide_mcp-tools.md.C8MiIu7F.lean.js → user-guide_mcp-tools.md.Vi3RoflJ.lean.js} +1 -1
  21. package/export-template/docs/assets/user-guide_run-comparison.md.CQc1aoU8.js +1 -0
  22. package/export-template/docs/assets/user-guide_run-comparison.md.CQc1aoU8.lean.js +1 -0
  23. package/export-template/docs/assets/user-guide_understanding-findings.md.DL1UDhvR.js +1 -0
  24. package/export-template/docs/assets/user-guide_understanding-findings.md.DL1UDhvR.lean.js +1 -0
  25. package/export-template/docs/contributor-guide/architecture/board-widgets.html +2 -2
  26. package/export-template/docs/contributor-guide/architecture/detector-contract.html +2 -2
  27. package/export-template/docs/contributor-guide/architecture/drill-down.html +1 -1
  28. package/export-template/docs/contributor-guide/architecture/impact-estimation.html +2 -2
  29. package/export-template/docs/contributor-guide/architecture/index.html +1 -1
  30. package/export-template/docs/contributor-guide/architecture/overview.html +1 -1
  31. package/export-template/docs/contributor-guide/architecture/state-and-history.html +2 -2
  32. package/export-template/docs/contributor-guide/architecture/widget-rendering.html +2 -2
  33. package/export-template/docs/contributor-guide/architecture/worker-protocol.html +2 -2
  34. package/export-template/docs/contributor-guide/contributing.html +1 -1
  35. package/export-template/docs/contributor-guide/development-setup.html +1 -1
  36. package/export-template/docs/contributor-guide/testing.html +1 -1
  37. package/export-template/docs/index.html +1 -1
  38. package/export-template/docs/tuning-reference/anti-patterns.html +1 -1
  39. package/export-template/docs/tuning-reference/aqe.html +1 -1
  40. package/export-template/docs/tuning-reference/bottleneck-broadcast-sizing.html +1 -1
  41. package/export-template/docs/tuning-reference/bottleneck-cold-start.html +1 -1
  42. package/export-template/docs/tuning-reference/bottleneck-duplicate-plan-subtree.html +1 -1
  43. package/export-template/docs/tuning-reference/bottleneck-failures.html +1 -1
  44. package/export-template/docs/tuning-reference/bottleneck-gc.html +1 -1
  45. package/export-template/docs/tuning-reference/bottleneck-job-failure-rate.html +1 -1
  46. package/export-template/docs/tuning-reference/bottleneck-memory-utilization.html +1 -1
  47. package/export-template/docs/tuning-reference/bottleneck-retry-waste.html +1 -1
  48. package/export-template/docs/tuning-reference/bottleneck-shuffle.html +1 -1
  49. package/export-template/docs/tuning-reference/bottleneck-skew.html +1 -1
  50. package/export-template/docs/tuning-reference/bottleneck-slow-host.html +1 -1
  51. package/export-template/docs/tuning-reference/bottleneck-small-files.html +1 -1
  52. package/export-template/docs/tuning-reference/bottleneck-spill.html +1 -1
  53. package/export-template/docs/tuning-reference/bottleneck-straggler.html +1 -1
  54. package/export-template/docs/tuning-reference/bottleneck-tiny-tasks.html +1 -1
  55. package/export-template/docs/tuning-reference/bottleneck-utilization.html +1 -1
  56. package/export-template/docs/tuning-reference/caching.html +1 -1
  57. package/export-template/docs/tuning-reference/cluster-config.html +1 -1
  58. package/export-template/docs/tuning-reference/config.html +1 -1
  59. package/export-template/docs/tuning-reference/data-formats.html +1 -1
  60. package/export-template/docs/tuning-reference/index.html +1 -1
  61. package/export-template/docs/tuning-reference/intro.html +1 -1
  62. package/export-template/docs/tuning-reference/joins.html +1 -1
  63. package/export-template/docs/tuning-reference/memory-model.html +1 -1
  64. package/export-template/docs/tuning-reference/metrics.html +1 -1
  65. package/export-template/docs/tuning-reference/partitioning.html +1 -1
  66. package/export-template/docs/tuning-reference/pyspark.html +1 -1
  67. package/export-template/docs/tuning-reference/shuffle.html +1 -1
  68. package/export-template/docs/tuning-reference/spark-architecture.html +1 -1
  69. package/export-template/docs/tuning-reference/table-formats.html +1 -1
  70. package/export-template/docs/user-guide/alternative-log-retrieval.html +2 -2
  71. package/export-template/docs/user-guide/getting-started.html +3 -3
  72. package/export-template/docs/user-guide/mcp-tools.html +4 -4
  73. package/export-template/docs/user-guide/run-comparison.html +2 -2
  74. package/export-template/docs/user-guide/understanding-findings.html +2 -2
  75. package/export-template/index.html +75 -79
  76. package/export-template/parser-worker-DyjiQvfP.js +112 -0
  77. package/export-template/sample-runs/sample-run.ndjson.gz +0 -0
  78. package/package.json +5 -4
  79. package/vendor-core/detectors.js +149 -24
  80. package/vendor-core/docs-content/detection/cache.md +4 -3
  81. package/vendor-core/docs-content/detection/chrn.md +4 -2
  82. package/vendor-core/docs-content/detection/local.md +2 -3
  83. package/vendor-core/docs-content/detection/mem.md +3 -3
  84. package/vendor-core/docs-content/detection/spec.md +4 -3
  85. package/vendor-core/evidence-report.js +1 -1
  86. package/vendor-core/parser-worker.js +2 -2
  87. package/vendor-core/run-comparison.js +21 -2
  88. package/export-template/docs/assets/chunks/@localSearchIndexroot.DNY8bVcl.js +0 -1
  89. package/export-template/docs/assets/chunks/theme.Df2VAG9w.js +0 -2
  90. package/export-template/docs/assets/contributor-guide_architecture_board-widgets.md.B-OsL91z.js +0 -1
  91. package/export-template/docs/assets/contributor-guide_architecture_board-widgets.md.B-OsL91z.lean.js +0 -1
  92. package/export-template/docs/assets/contributor-guide_architecture_detector-contract.md.BOeH4d1J.js +0 -1
  93. package/export-template/docs/assets/contributor-guide_architecture_state-and-history.md.m3S3UdMk.js +0 -1
  94. package/export-template/docs/assets/contributor-guide_architecture_state-and-history.md.m3S3UdMk.lean.js +0 -1
  95. package/export-template/docs/assets/contributor-guide_architecture_widget-rendering.md.DbqPf2OT.js +0 -1
  96. package/export-template/docs/assets/contributor-guide_architecture_worker-protocol.md.B93qJ_tT.js +0 -6
  97. package/export-template/docs/assets/user-guide_getting-started.md.DtEM37MK.js +0 -3
  98. package/export-template/docs/assets/user-guide_getting-started.md.DtEM37MK.lean.js +0 -1
  99. package/export-template/docs/assets/user-guide_run-comparison.md.S0TWWmLY.js +0 -1
  100. package/export-template/docs/assets/user-guide_run-comparison.md.S0TWWmLY.lean.js +0 -1
  101. package/export-template/docs/assets/user-guide_understanding-findings.md.D0R_Y-R2.js +0 -1
  102. package/export-template/docs/assets/user-guide_understanding-findings.md.D0R_Y-R2.lean.js +0 -1
  103. package/export-template/parser-worker-QqyEE4m9.js +0 -64
  104. /package/export-template/docs/assets/{contributor-guide_architecture_detector-contract.md.BOeH4d1J.lean.js → contributor-guide_architecture_detector-contract.md.CgzUsQ6W.lean.js} +0 -0
  105. /package/export-template/docs/assets/{contributor-guide_architecture_impact-estimation.md.DYCDPgkh.lean.js → contributor-guide_architecture_impact-estimation.md.CooslVJt.lean.js} +0 -0
  106. /package/export-template/docs/assets/{contributor-guide_architecture_widget-rendering.md.DbqPf2OT.lean.js → contributor-guide_architecture_widget-rendering.md.R27gQrgY.lean.js} +0 -0
  107. /package/export-template/docs/assets/{contributor-guide_architecture_worker-protocol.md.B93qJ_tT.lean.js → contributor-guide_architecture_worker-protocol.md.IbnfNrV3.lean.js} +0 -0
@@ -411,6 +411,22 @@ export function findDuplicateSubtrees(
411
411
  return results;
412
412
  }
413
413
 
414
+ // Fingerprint matching compares operator + metric names only, not literal values or expr IDs
415
+ // (see the finding's validationRequired text), so a small pattern repeated the bare minimum
416
+ // number of times is the case most likely to be coincidental rather than real duplicated work.
417
+ // A bigger matched subtree, or more repeats, are each on their own strong corroborating evidence
418
+ // that the match is real: the odds of two semantically-different query branches producing an
419
+ // identical operator-name sequence shrink fast as the sequence grows or repeats.
420
+ function duplicateSubtreeConfidence(
421
+ subtreeSize ,
422
+ occurrences ,
423
+ thresholds ,
424
+ ) {
425
+ if (subtreeSize <= thresholds.minSubtreeSize && occurrences <= thresholds.minOccurrences) return 'low';
426
+ if (subtreeSize >= thresholds.minSubtreeSize * 2 || occurrences >= thresholds.minOccurrences + 2) return 'high';
427
+ return 'medium';
428
+ }
429
+
414
430
  // Exact metric names Spark emits, verified against a real SQLExecutionStart's sparkPlanInfo.
415
431
  // The write-side byte metric is "written output", not "size of written files".
416
432
  const FILES_READ_COUNT = 'number of files read';
@@ -519,6 +535,19 @@ function clippedWasteMs(wasteMs , stageId , ctx )
519
535
  const CACHE_UTILIZATION_VALIDATION =
520
536
  "This ratio is a point-in-time storage snapshot from stage-submission events, not a runtime read-count. Confirm against the Spark UI's Storage tab before acting.";
521
537
 
538
+ // Both cachedRatio and diskRatio are percentages of an RDD's partition/byte count: with few
539
+ // partitions, one partition flipping cached/evicted (or disk-resident/memory-resident) swings the
540
+ // reported percentage by a large amount, so the point estimate is noisy. More partitions average
541
+ // that noise out into a stable ratio. numPartitions is the only sample-size signal
542
+ // DetectorRddInfo carries, so it drives confidence for both variants rather than a flat guess.
543
+ // 10/50 mirror this file's other "trust the sample" cutoffs (skew's minTasksForP95: 20,
544
+ // coreLocality's minTasks: 50).
545
+ function cacheSampleConfidence(numPartitions ) {
546
+ if (numPartitions < 10) return 'low';
547
+ if (numPartitions >= 50) return 'high';
548
+ return 'medium';
549
+ }
550
+
522
551
  function partialCacheFinding(rdd , cachedRatio , impactBand ) {
523
552
  const rddName = rdd.name || `RDD ${rdd.id}`;
524
553
  const cachedPct = Math.round(cachedRatio * 100);
@@ -527,7 +556,7 @@ function partialCacheFinding(rdd , cachedRatio , impactBa
527
556
  type: 'cacheUtilization', variant: 'partialCache', stageId: null,
528
557
  rddId: rdd.id, rddName,
529
558
  impactBand, metric: 'cachedRatio', value: cachedPct,
530
- confidence: 'medium',
559
+ confidence: cacheSampleConfidence(rdd.numPartitions),
531
560
  validationRequired: CACHE_UTILIZATION_VALIDATION,
532
561
  memorySize: rdd.memorySize, diskSize: rdd.diskSize,
533
562
  numCachedPartitions: rdd.numCachedPartitions, numPartitions: rdd.numPartitions,
@@ -542,7 +571,7 @@ function diskSpilloverFinding(rdd , diskRatio , impactBan
542
571
  type: 'cacheUtilization', variant: 'diskSpillover', stageId: null,
543
572
  rddId: rdd.id, rddName,
544
573
  impactBand, metric: 'diskRatio', value: diskPct,
545
- confidence: 'medium',
574
+ confidence: cacheSampleConfidence(rdd.numPartitions),
546
575
  validationRequired: CACHE_UTILIZATION_VALIDATION,
547
576
  memorySize: rdd.memorySize, diskSize: rdd.diskSize,
548
577
  numCachedPartitions: rdd.numCachedPartitions, numPartitions: rdd.numPartitions,
@@ -584,6 +613,103 @@ function diskSpilloverFinding(rdd , diskRatio , impactBan
584
613
  export const STRAGGLER_FLOOR_PCT_WARN = 0.005;
585
614
  export const STRAGGLER_FLOOR_PCT_CRIT = 0.02;
586
615
 
616
+ // A ratio just past ratioWarn is the case most likely to be ordinary task-duration variance
617
+ // rather than real skew; a ratio many multiples past it (a 50x P95/median vs. a 3.1x one) is
618
+ // unambiguous. 1.5x/5x mirror this file's other "just past the floor vs. clearly past it" splits
619
+ // (duplicateSubtreeConfidence's 2x, cacheSampleConfidence's 10/50-partition cutoffs).
620
+ function skewConfidence(ratio , ratioWarn ) {
621
+ if (ratio <= ratioWarn * 1.5) return 'low';
622
+ if (ratio >= ratioWarn * 5) return 'high';
623
+ return 'medium';
624
+ }
625
+
626
+ // warnPct100/lowInfoPct100 are this detector's own two thresholds; scale confidence as a multiple
627
+ // of whichever one gates the branch that fired, the same way skewConfidence scales off ratioWarn.
628
+ // High-GC: a pct just past warnPct100 (10%) is likely normal variance, 3x past it is unambiguous.
629
+ // Low-GC ("cost", over-provisioned): a pct just under lowInfoPct100 (5%) is borderline, a pct near
630
+ // zero is unambiguous idle GC.
631
+ function gcConfidence(
632
+ pct ,
633
+ thresholds ,
634
+ direction ,
635
+ ) {
636
+ if (direction === 'high') {
637
+ if (pct <= thresholds.warnPct100 * 1.5) return 'low';
638
+ if (pct >= thresholds.warnPct100 * 3) return 'high';
639
+ return 'medium';
640
+ }
641
+ if (pct >= thresholds.lowInfoPct100 * 0.66) return 'low';
642
+ if (pct <= thresholds.lowInfoPct100 * 0.2) return 'high';
643
+ return 'medium';
644
+ }
645
+
646
+ // warnFloor gates the finding, so a value just past it is the weakest evidence this detector can
647
+ // produce. highFloor is critPct: straggler's own second (speculative-share) tier, 2x warnPct
648
+ // (0.10 -> 0.20); reused as the high-confidence bar for the stragglerShare path too since that
649
+ // metric has no dedicated critical tier of its own (see the "no dedicated critical tier" comment
650
+ // on the straggler detector) but is the same 0-1 task-share magnitude.
651
+ function stragglerConfidence(shareValue , warnFloor , highFloor ) {
652
+ if (shareValue < warnFloor * 1.5) return 'low';
653
+ if (shareValue >= highFloor) return 'high';
654
+ return 'medium';
655
+ }
656
+
657
+ // minWasteMs is speculationWaste's own floor; a run that barely clears it (under 1.5x) is the
658
+ // weakest case, one that clears it several times over (4x, i.e. 4 minutes against a 1-minute
659
+ // floor) is unambiguous.
660
+ function speculationWasteConfidence(wastedMs , minWasteMs ) {
661
+ if (wastedMs <= minWasteMs * 1.5) return 'low';
662
+ if (wastedMs >= minWasteMs * 4) return 'high';
663
+ return 'medium';
664
+ }
665
+
666
+ // The finding already gates on wastedMBSeconds > wasteBufferMultiplier * usedMBSeconds, i.e. a
667
+ // ratio of 1 at the floor; scale confidence off that same ratio the way skewConfidence scales off
668
+ // ratioWarn, instead of introducing a second, unrelated multiplier.
669
+ function memoryWasteConfidence(wastedMBSeconds , usedMBSeconds , wasteBufferMultiplier ) {
670
+ const ratio = usedMBSeconds > 0 ? wastedMBSeconds / (wasteBufferMultiplier * usedMBSeconds) : Infinity;
671
+ if (ratio <= 1.5) return 'low';
672
+ if (ratio >= 3) return 'high';
673
+ return 'medium';
674
+ }
675
+
676
+ // Two independent weak spots can each undercut this finding: a ratio just past warnRatio (could
677
+ // be one bad stage), or too few sampled tasks (mirrors cacheSampleConfidence's use of
678
+ // numPartitions as a sample-size signal). Report whichever signal is weaker rather than
679
+ // averaging them away. critRatio is this detector's own existing second tier, reused directly as
680
+ // the ratio high-bar; minTasks*2/*4 mirror the same "2x floor is still weak, 4x is strong" spread
681
+ // used elsewhere in this file.
682
+ function coreLocalityConfidence(
683
+ ratio ,
684
+ totalTasks ,
685
+ thresholds ,
686
+ ) {
687
+ const rank = { low: 0, medium: 1, high: 2 } ;
688
+ const ratioTier = ratio < thresholds.warnRatio * 1.5 ? 'low' : ratio >= thresholds.critRatio ? 'high' : 'medium';
689
+ const sampleTier = totalTasks < thresholds.minTasks * 2 ? 'low' : totalTasks >= thresholds.minTasks * 4 ? 'high' : 'medium';
690
+ return rank[ratioTier] <= rank[sampleTier] ? ratioTier : sampleTier;
691
+ }
692
+
693
+ // warningPct/criticalPct are this detector's own two tiers (0.30/0.60); a churn rate just past
694
+ // warningPct is the borderline call the impact-band split already treats as the weaker tier, so
695
+ // reuse criticalPct directly as the high-confidence bar instead of inventing a third figure.
696
+ function autoscalingChurnConfidence(shortLivedPct , warningPct , criticalPct ) {
697
+ if (shortLivedPct <= warningPct * 1.5) return 'low';
698
+ if (shortLivedPct >= criticalPct) return 'high';
699
+ return 'medium';
700
+ }
701
+
702
+ // minExecutions is the bare minimum occurrence count this detector will even emit a finding for;
703
+ // a match at exactly that count is the weakest reuse signal (as likely to be coincidental overlap
704
+ // as real shared work), while 3x the floor is several independent executions all hitting the same
705
+ // relation/composite shape, unambiguous. Mirrors duplicateSubtreeConfidence's occurrences handling
706
+ // for the same reason: repetition count is the strength signal for a structural-match detector.
707
+ function cachingReuseConfidence(occurrences , minExecutions ) {
708
+ if (occurrences <= minExecutions) return 'low';
709
+ if (occurrences >= minExecutions * 3) return 'high';
710
+ return 'medium';
711
+ }
712
+
587
713
  export const DETECTORS = [
588
714
  {
589
715
  type: 'skew', scope: 'stage', order: 30, fixEffort: 'code', version: 1,
@@ -611,8 +737,8 @@ export const DETECTORS = [
611
737
  type: 'skew', stageId: stage.id,
612
738
  impactBand: 'warning',
613
739
  metric, value,
614
- confidence: 'low',
615
- validationRequired: 'The 0.5% runtime-floor percentage that gates this finding is our own noise floor, unvalidated: no external tool publishes an equivalent metric to calibrate against. Confirm against known-good/known-bad real logs before trusting the impact-band split.',
740
+ confidence: skewConfidence(ratio, this.thresholds.ratioWarn),
741
+ validationRequired: 'This finding is gated by a 0.5% runtime-floor threshold, our own noise floor for this metric.',
616
742
  recommendation: `Task duration ratio (${metric}) is ${value}×: for join-driven skew, enable AQE skew-join handling (spark.sql.adaptive.skewJoin.enabled); otherwise salt the key or repartition on a better key to reduce task skew.`,
617
743
  };
618
744
  },
@@ -766,8 +892,7 @@ export const DETECTORS = [
766
892
  {
767
893
  type: 'gc', scope: 'stage', order: 50, fixEffort: 'config', version: 1,
768
894
  docAnchor: '#bottleneck-gc',
769
- confidence: 'low',
770
- validationRequired: 'The 10-second minimum-runtime floor that gates this finding is our own noise floor, unvalidated: no external tool publishes an equivalent metric to calibrate against. Confirm against known-good/known-bad real logs before trusting the impact-band split.',
895
+ validationRequired: 'This finding is gated by a 10-second minimum-runtime floor, our own noise floor for this metric.',
771
896
  thresholds: {
772
897
  warnPct100: 10,
773
898
  // Descending tier: ExecutorGcHeuristic, ported as-is.
@@ -778,7 +903,7 @@ export const DETECTORS = [
778
903
  detect(
779
904
 
780
905
 
781
-
906
+
782
907
 
783
908
  stage ,
784
909
  ) {
@@ -790,7 +915,7 @@ export const DETECTORS = [
790
915
  type: 'gc', stageId: stage.id,
791
916
  impactBand: 'warning',
792
917
  metric: 'gcPct', value,
793
- confidence: this.confidence, validationRequired: this.validationRequired,
918
+ confidence: gcConfidence(pct, this.thresholds, 'high'), validationRequired: this.validationRequired,
794
919
  recommendation: `GC consumed ${value}% of executor run time: reduce object creation, use primitive types, avoid UDFs, increase executor memory.`,
795
920
  };
796
921
  }
@@ -802,7 +927,7 @@ export const DETECTORS = [
802
927
  type: 'gc', stageId: stage.id, direction: 'low',
803
928
  impactBand: 'info',
804
929
  metric: 'gcPct', value,
805
- confidence: this.confidence, validationRequired: this.validationRequired,
930
+ confidence: gcConfidence(pct, this.thresholds, 'low'), validationRequired: this.validationRequired,
806
931
  recommendation: `GC consumed only ${value}% of executor run time: memory may be over-provisioned; consider reducing spark.executor.memory for cost savings.`,
807
932
  };
808
933
  }
@@ -1028,20 +1153,21 @@ export const DETECTORS = [
1028
1153
  unit: useSpeculativeMetric ? 'count' : 'pct',
1029
1154
  speculativeTasks: stage.speculativeTasks ?? 0,
1030
1155
  stragglerCount: stage.stragglerCount ?? 0,
1031
- confidence: 'low',
1032
- validationRequired: 'The 0.5%/2% runtime-floor percentages that gate this finding are our own noise floor, unvalidated: no external tool publishes an equivalent metric to calibrate against. Confirm against known-good/known-bad real logs before trusting the impact-band split.',
1156
+ confidence: useSpeculativeMetric
1157
+ ? stragglerConfidence(speculativeShare, this.thresholds.warnPct, this.thresholds.critPct)
1158
+ : stragglerConfidence(stragglerShare, this.thresholds.shareWarn, this.thresholds.critPct),
1159
+ validationRequired: 'This finding is gated by 0.5%/2% runtime-floor thresholds, our own noise floor for this metric.',
1033
1160
  recommendation: `${detail}: rule out a GC pause or a slow shuffle fetch before assuming a hardware issue; if a skewed key is the real cause, that's a candidate for AQE's skew-join handling.`,
1034
1161
  };
1035
1162
  },
1036
1163
  },
1037
1164
  {
1038
1165
  type: 'speculationWaste', scope: 'stage', order: 71, fixEffort: 'config', version: 1,
1039
- docAnchor: '#bottleneck-straggler', confidence: 'low',
1166
+ docAnchor: '#bottleneck-straggler',
1040
1167
  thresholds: { minWasted: 5, minWasteMs: 60000 },
1041
1168
  detect(
1042
1169
 
1043
1170
 
1044
-
1045
1171
 
1046
1172
  stage ,
1047
1173
  ) {
@@ -1052,7 +1178,7 @@ export const DETECTORS = [
1052
1178
  type: 'speculationWaste', stageId: stage.id,
1053
1179
  impactBand: 'warning',
1054
1180
  metric: 'speculationWasteMs', value: wastedMs,
1055
- confidence: this.confidence,
1181
+ confidence: speculationWasteConfidence(wastedMs, this.thresholds.minWasteMs),
1056
1182
  recommendation: `Speculative execution discarded ${Math.round(wastedMs / 1000)}s of executor time in this stage; if task durations are naturally variable rather than genuine stragglers, consider tuning spark.speculation.multiplier/quantile.`,
1057
1183
  };
1058
1184
  },
@@ -1298,8 +1424,8 @@ export const DETECTORS = [
1298
1424
  out.push({
1299
1425
  type: 'memoryUtilization', variant: 'wasteModel', stageId: null,
1300
1426
  impactBand: 'info', metric: 'wastedMBSeconds', value,
1301
- confidence: 'low',
1302
- validationRequired: 'Memory-waste estimate uses allocated-vs-used memory-time and an unverified 1.5x buffer: confirm against the Spark UI before acting.',
1427
+ confidence: memoryWasteConfidence(wastedMBSeconds, usedMBSeconds, this.thresholds.wasteBufferMultiplier),
1428
+ validationRequired: 'Memory-waste estimate uses allocated-vs-used memory-time and a 1.5x buffer: confirm against the Spark UI before acting.',
1303
1429
  recommendation: `Allocated executor memory sat largely idle over the run (~${value.toLocaleString('en-US')} MB-seconds wasted): review spark.executor.memory and executor count.`,
1304
1430
  });
1305
1431
  }
@@ -1376,8 +1502,8 @@ export const DETECTORS = [
1376
1502
  metric: 'nonLocalRatio', value,
1377
1503
  // Raw count behind the ratio, for the impact estimator. Non-null whenever totalTasks is.
1378
1504
  nonLocalTaskCount: nonLocalTasks ,
1379
- confidence: 'low',
1380
- validationRequired: 'The 15%/35% non-local-ratio thresholds (and the 50-task minimum) are unvalidated design-spike values: no external tool publishes an equivalent metric to calibrate against. Confirm against known-good/known-bad real logs before trusting the impact-band split.',
1505
+ confidence: coreLocalityConfidence(ratio , totalTasks, this.thresholds),
1506
+ validationRequired: 'This finding is gated by 15%/35% non-local-ratio thresholds (and a 50-task minimum), our own noise floor for this metric.',
1381
1507
  recommendation: `${value}% of tasks (${nonLocalTasks }) ran without process- or node-local data placement: check spark.locality.wait settings and executor/data colocation.`,
1382
1508
  };
1383
1509
  },
@@ -1387,12 +1513,10 @@ export const DETECTORS = [
1387
1513
  // re-provisioning, not normal scale-down). Reuses utilization's add/remove matching, but
1388
1514
  // measures lifetime against a threshold instead of aggregate active-time.
1389
1515
  type: 'autoscalingChurn', scope: 'app', order: 103, fixEffort: 'config', version: 1,
1390
- confidence: 'low',
1391
1516
  thresholds: { shortLivedMs: 120_000, warningPct: 0.30, criticalPct: 0.60, minExecutors: 5 },
1392
1517
  detect(
1393
1518
 
1394
1519
 
1395
-
1396
1520
 
1397
1521
  ctx ,
1398
1522
  ) {
@@ -1421,7 +1545,7 @@ export const DETECTORS = [
1421
1545
  metric: 'shortLivedExecutorPct', value: pct,
1422
1546
  // Raw count behind the percentage, for the impact estimator's startup-overhead figure.
1423
1547
  shortLivedExecutorCount: shortLivedCount,
1424
- confidence: this.confidence,
1548
+ confidence: autoscalingChurnConfidence(shortLivedPct, this.thresholds.warningPct, this.thresholds.criticalPct),
1425
1549
  recommendation: `${pct}% of executors ran for under 2 minutes before being removed. This looks like wasteful re-provisioning rather than normal scale-down; consider raising spark.dynamicAllocation.executorIdleTimeout or widening the minExecutors/maxExecutors bounds to reduce flapping.`,
1426
1550
  };
1427
1551
  },
@@ -1562,7 +1686,7 @@ export const DETECTORS = [
1562
1686
  metric: 'executionReuse', value,
1563
1687
  format: 'derived', relations, operator: agg.operator, relation: relationDisplay,
1564
1688
  executionIds: finalExecutionIds, totalReadBytes,
1565
- confidence: 'low',
1689
+ confidence: cachingReuseConfidence(value, this.thresholds.minExecutions),
1566
1690
  validationRequired:
1567
1691
  'Composite reuse is inferred from a structural plan-shape match (operator + normalized ' +
1568
1692
  'join/filter condition + child shapes) across SQL executions; confirm these executions ' +
@@ -1593,7 +1717,7 @@ export const DETECTORS = [
1593
1717
  relation: agg.relation, format: agg.format,
1594
1718
  executionIds: residualExecutionIds.sort((a, b) => a - b),
1595
1719
  totalReadBytes,
1596
- confidence: 'low',
1720
+ confidence: cachingReuseConfidence(value, this.thresholds.minExecutions),
1597
1721
  validationRequired:
1598
1722
  'Relation-reuse is inferred from the pre-AQE plan scan identity across SQL ' +
1599
1723
  'executions; confirm the reads are the same data and cacheable within one ' +
@@ -1744,7 +1868,8 @@ export const DETECTORS = [
1744
1868
  // wallClock estimate (the common case). Only surfaces on the rare occupancy-sweep miss.
1745
1869
  impactBand: 'warning', metric: 'subtreeOccurrences', value: g.occurrences,
1746
1870
  rootName: g.rootName, subtreeSize: g.subtreeSize, sampleRelation: g.sampleRelation,
1747
- groupIndex: g.groupIndex, confidence: 'medium',
1871
+ groupIndex: g.groupIndex,
1872
+ confidence: duplicateSubtreeConfidence(g.subtreeSize, g.occurrences, this.thresholds),
1748
1873
  validationRequired: 'Duplicate-subtree matching compares operator names and metric names only, not literal values or expr IDs: confirm the repeated work is real in the Spark SQL plan tab before acting.',
1749
1874
  recommendation: g.isExchangeRoot
1750
1875
  ? `A ${g.subtreeSize}-node subtree rooted at ${pathBasename(g.rootName)} repeats ${g.occurrences}x in this plan${touching}: this looks like a possible missed exchange reuse; check whether the same shuffle could be computed once and reused.`
@@ -1,6 +1,7 @@
1
1
  ### `CACHE`: Caching opportunity {#cache}
2
2
 
3
3
  A reusable dataset (re-read via the same SQL relation more than once) may be
4
- worth persisting between stages. Self-flagged low-confidence: reuse is only
5
- inferred, from plan-scan identity across SQL executions, so confirm the reads
6
- really do hit the same data before you cache anything.
4
+ worth persisting between stages. Self-flags a confidence that scales with
5
+ how many executions reuse the same relation: reuse is only inferred, from
6
+ plan-scan identity across SQL executions, so confirm the reads really do
7
+ hit the same data before you cache anything.
@@ -3,5 +3,7 @@
3
3
  Executors are stood up and torn down again before they can do useful work:
4
4
  re-provisioning churn rather than normal scale-down. Raise
5
5
  `spark.dynamicAllocation.executorIdleTimeout`, or widen the
6
- `minExecutors`/`maxExecutors` bounds to reduce flapping. Self-flagged
7
- low-confidence: a design spike, not yet validated against real-world runs.
6
+ `minExecutors`/`maxExecutors` bounds to reduce flapping. Self-flags a
7
+ confidence that scales with how far the short-lived-executor share sits
8
+ past the threshold: these thresholds are still a design spike, not yet
9
+ validated against real-world runs.
@@ -2,6 +2,5 @@
2
2
 
3
3
  Tasks run without process- or node-local data placement more often than
4
4
  expected. Check `spark.locality.wait` settings and executor/data colocation.
5
- Self-flagged low-confidence: the non-local-ratio thresholds are unvalidated
6
- design-spike values, and no external tool publishes an equivalent metric to
7
- calibrate them against.
5
+ Self-flags a confidence that scales with the non-local ratio and sample
6
+ size: the thresholds are our own noise floor for this metric.
@@ -4,7 +4,7 @@ Executor memory or core capacity may be over- or under-provisioned. Some
4
4
  detail here needs `spark.eventLog.logStageExecutorMetrics=true` on the run
5
5
  being analyzed; without it, per-executor memory usage can't be broken down.
6
6
  Review `spark.executor.memory` and executor count if allocated memory sat
7
- largely idle over the run. That idle-memory variant is self-flagged
8
- low-confidence: it estimates waste from allocated-versus-used memory-time
9
- against an unverified 1.5x buffer. Check it against
7
+ largely idle over the run. That idle-memory variant self-flags a confidence
8
+ that scales with how far the estimated waste sits past a 1.5x buffer: it
9
+ estimates waste from allocated-versus-used memory-time. Check it against
10
10
  the Spark UI before resizing anything.
@@ -1,7 +1,8 @@
1
1
  ### `SPEC`: Speculation waste {#spec}
2
2
 
3
3
  Speculative task attempts used a lot of executor time without confirming a
4
- genuine straggler. Self-flagged low-confidence: a design spike, not yet
5
- validated against real-world runs. If task durations are just naturally
6
- variable rather than genuine stragglers, tune
4
+ genuine straggler. Self-flags a confidence that scales with how far the
5
+ wasted time sits past the threshold: these thresholds are still a design
6
+ spike, not yet validated against real-world runs. If task durations are
7
+ just naturally variable rather than genuine stragglers, tune
7
8
  `spark.speculation.multiplier`/`spark.speculation.quantile`.
@@ -269,7 +269,7 @@ function renderEvidenceValue(key , value ) {
269
269
 
270
270
  function formatWallClockRange(low , high ) {
271
271
  const fmtMs = (ms ) => (ms === 0 ? '0s' : formatDuration(ms));
272
- return low === high ? `Est. ${fmtMs(high)}` : `Est. ${fmtMs(low)}-${fmtMs(high)}`;
272
+ return low === high ? `Estimated ${fmtMs(high)}` : `Estimated ${fmtMs(low)}-${fmtMs(high)}`;
273
273
  }
274
274
 
275
275
  function formatRawWaste(rawWaste ) {
@@ -147,7 +147,7 @@ export async function runParse(
147
147
  }
148
148
 
149
149
  if (!state.app) {
150
- emit({ type: 'error', message: 'Not a Spark event log: SparkListenerApplicationStart not found.' });
150
+ emit({ type: 'error', message: 'Not a Spark event log: no application-start event found. Choose a Spark event log file, or check the docs for supported formats.' });
151
151
  return;
152
152
  }
153
153
 
@@ -203,7 +203,7 @@ export async function runParseFiles(
203
203
  }
204
204
 
205
205
  if (!state.app) {
206
- emit({ type: 'error', message: 'Not a Spark event log: SparkListenerApplicationStart not found.' });
206
+ emit({ type: 'error', message: 'Not a Spark event log: no application-start event found. Choose a Spark event log file, or check the docs for supported formats.' });
207
207
  return;
208
208
  }
209
209
 
@@ -395,6 +395,14 @@ export function buildComparison(
395
395
  );
396
396
  }
397
397
 
398
+ // matchStages' coverage is a Dice coefficient: (2 * pairs.length) / (baseCount
399
+ // + candCount). Below 0.5, more than half of each run's stages went unpaired,
400
+ // so the stage-level rows (stageSkew, baseStages/candStages) mostly show
401
+ // unrelated work side by side rather than the same stage before/after -- the
402
+ // comparison is dominated by guesswork, not genuine pairing. 0.5 is thus the
403
+ // natural midpoint for "more matched than not," not an arbitrary tuning knob.
404
+ const LOW_COVERAGE_THRESHOLD = 0.5;
405
+
398
406
  export function compareRuns(
399
407
  baseline ,
400
408
  candidate ,
@@ -405,10 +413,21 @@ export function compareRuns(
405
413
  // is the normal way to label an A/B experiment, so a mismatch must not block
406
414
  // the (matching-free, name-independent) deltas. Surface it as `low` instead.
407
415
  const namesDiffer = namesConflict(baseSnap.app, candSnap.app);
416
+ // Coverage is the other half of the signal: identical names on two runs that
417
+ // barely share any stages are just as misleading as differing names on two
418
+ // runs that match well, so either condition alone drops confidence to `low`.
419
+ const lowCoverage = match.coverage < LOW_COVERAGE_THRESHOLD;
420
+ const reason = namesDiffer && lowCoverage
421
+ ? `Run names differ and only ${(match.coverage * 100).toFixed(0)}% of stages matched, so deltas may compare different work.`
422
+ : namesDiffer
423
+ ? 'Run names differ, so deltas may compare different work.'
424
+ : lowCoverage
425
+ ? `Only ${(match.coverage * 100).toFixed(0)}% of stages matched between runs, so per-stage rows mostly compare unrelated work.`
426
+ : null;
408
427
  return {
409
428
  baselineLabel: baseline.label, candidateLabel: candidate.label,
410
- confidence: namesDiffer ? 'low' : 'ok',
411
- reason: namesDiffer ? 'Run names differ, so deltas may compare different work.' : null,
429
+ confidence: namesDiffer || lowCoverage ? 'low' : 'ok',
430
+ reason,
412
431
  matchedCoverage: match.coverage,
413
432
  metrics: metricDeltas(baseSnap, candSnap),
414
433
  findings: findingsDelta(baseSnap, candSnap),