sparkforensics-mcp 0.2.2 → 0.2.4

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (42) hide show
  1. package/bin/sparkforensics-mcp.mjs +11 -5
  2. package/package.json +3 -3
  3. package/vendor-core/cli/collect-run.js +3 -2
  4. package/vendor-core/cli/native-zstd.js +351 -0
  5. package/vendor-core/detectors.js +243 -53
  6. package/vendor-core/docs-config.js +34 -8
  7. package/vendor-core/docs-content/chapters/03-memory-model.md +39 -0
  8. package/vendor-core/docs-content/chapters/11-cluster-config.md +40 -0
  9. package/vendor-core/docs-content/detection/gc.md +2 -0
  10. package/vendor-core/docs-content/detection/host.md +2 -1
  11. package/vendor-core/docs-content/detection/plan.md +3 -1
  12. package/vendor-core/docs-content/detection/shape.md +2 -1
  13. package/vendor-core/docs-content/detection/shfl.md +2 -1
  14. package/vendor-core/docs-content/detection/spill.md +1 -1
  15. package/vendor-core/docs-content/detection/strag.md +2 -1
  16. package/vendor-core/docs-content/detection/tiny.md +2 -1
  17. package/vendor-core/docs-content/tuning/failures.md +1 -1
  18. package/vendor-core/docs-content/tuning/gc.md +11 -4
  19. package/vendor-core/docs-content/tuning/shuffle.md +25 -5
  20. package/vendor-core/docs-content/tuning/skew.md +14 -6
  21. package/vendor-core/docs-content/tuning/small-files.md +12 -7
  22. package/vendor-core/docs-content/tuning/straggler.md +34 -0
  23. package/vendor-core/docs-content/tuning/tiny-tasks.md +1 -1
  24. package/vendor-core/docs-content/tuning/utilization.md +57 -7
  25. package/vendor-core/docs-content/upstream.json +4 -0
  26. package/vendor-core/event-handlers.js +321 -69
  27. package/vendor-core/event-schemas.js +8 -6
  28. package/vendor-core/evidence-report.js +3 -1
  29. package/vendor-core/impact-estimator.js +170 -34
  30. package/vendor-core/mcp-tools.js +20 -7
  31. package/vendor-core/occupancy.js +71 -2
  32. package/vendor-core/parser-worker.js +56 -24
  33. package/vendor-core/plan-summary.js +5 -1
  34. package/vendor-core/run-comparison.js +1 -15
  35. package/vendor-core/shs-fetch.js +18 -7
  36. package/vendor-core/shs-load.js +2 -1
  37. package/vendor-core/stage-quantiles.js +111 -3
  38. package/vendor-core/string-hash.js +15 -0
  39. package/vendor-core/types.js +14 -1
  40. package/vendor-core/vendor/fzstd.js +94 -18
  41. package/vendor-core/zstd-worker-client.js +180 -0
  42. package/vendor-core/zstd-worker.js +103 -0
@@ -3,8 +3,9 @@ import { scanRelationId } from './plan-summary.js';
3
3
  import { computePeakConcurrentCores, computePeakConcurrentExecutorCount } from './core-count.js';
4
4
  import { walkPlanTree } from './plan-tree-walk.js';
5
5
  import { computeCoreLocalityRatio } from './core-locality-ratio.js';
6
- import { estimateSingleStage, } from './occupancy.js';
6
+ import { estimateSingleStage, tailRecoveryMs, tailRemovedWorkMs, stragglerFixLongestTaskMs, } from './occupancy.js';
7
7
  import { isExchangeNode, isBroadcastExchangeNode } from './plan-node-detail.js';
8
+ import { cyrb53 } from './string-hash.js';
8
9
 
9
10
 
10
11
  const MB = 1024 * 1024;
@@ -74,6 +75,9 @@ const TB = 1024 * GB;
74
75
 
75
76
 
76
77
 
78
+
79
+
80
+
77
81
 
78
82
 
79
83
 
@@ -154,6 +158,7 @@ const TB = 1024 * GB;
154
158
 
155
159
 
156
160
 
161
+
157
162
 
158
163
 
159
164
  // SpillPressureDetector (5a) + SpillSkewDetector (5b).
@@ -206,8 +211,10 @@ export function unionStageIds(nodes , fallback ) {
206
211
 
207
212
  // Bottom-up shape computation for duplicate-subtree detection and the
208
213
  // cachingOpportunity composite detector. `size` is the subtree node count; the default
209
- // fingerprint encodes operator name + sorted metric NAMES (never values, per spec) + child
210
- // fingerprints, so two subtrees with the same shape but different values still collide.
214
+ // fingerprint encodes operator name + sorted metric NAMES (never values, per spec) + a digest
215
+ // of each child's fingerprint, so two subtrees with the same shape but different values still
216
+ // collide. Digesting children keeps each fingerprint O(own size): embedding the full child
217
+ // strings made every node carry its whole subtree, O(n x depth) text on deep real plans.
211
218
  //
212
219
  // opts.includeDetail (default false) folds root's own detail (via opts.normalizeDetail) into
213
220
  // ONLY root's fingerprint, never a child's: lets a caller match "this node's own detail + shape"
@@ -239,7 +246,7 @@ export function computePlanShapes(
239
246
  const childShapes = realChildren.map((c) => visit(c, false));
240
247
  const size = 1 + childShapes.reduce((sum, c) => sum + c.size, 0);
241
248
  const metricNames = (realMetrics ?? []).map((m) => m.name).sort().join(',');
242
- const childFingerprints = childShapes.map((c) => c.fingerprint).join(',');
249
+ const childFingerprints = childShapes.map((c) => cyrb53(c.fingerprint)).join(',');
243
250
  const fingerprint = isRoot && includeDetail
244
251
  ? `${node.name}[${metricNames}]<${normalize(node.detail ?? '')}>{${childFingerprints}}`
245
252
  : `${node.name}[${metricNames}]{${childFingerprints}}`;
@@ -261,7 +268,11 @@ export function normalizeDetail(detail ) {
261
268
  .replace(/,?\s*plan_id=\d+/g, '')
262
269
  .replace(/\[codegen id\s*:\s*\d+\]/gi, '[codegen id]')
263
270
  .replace(/\bBuild(Left|Right)\b/g, 'BuildSide');
264
- s = s.replace(/([A-Za-z_][\w.]*)\s*=\s*([A-Za-z_][\w.]*)/g, (_m, l, r) => {
271
+ // The lookbehind only skips starts inside an identifier, which can't match unless the
272
+ // identifier's own start already did: same output, without re-scanning each identifier from
273
+ // every one of its letters (O(length^2) on the 6.8 KB average join detail of the largest real
274
+ // log; 64ms -> 40ms per analyze()).
275
+ s = s.replace(/(?<![A-Za-z_])([A-Za-z_][\w.]*)\s*=\s*([A-Za-z_][\w.]*)/g, (_m, l, r) => {
265
276
  const [a, b] = [l, r].sort();
266
277
  return `${a} = ${b}`;
267
278
  });
@@ -270,6 +281,21 @@ export function normalizeDetail(detail ) {
270
281
 
271
282
  const JOIN_NAME_RE = /Join/i;
272
283
 
284
+ // scanRelationId per plan node, memoized: cachingOpportunity's relation walk and
285
+ // findCompositeCandidates both classify every node of every execution, and a JDBC scan's detail
286
+ // carries its whole inner SQL, so the repeated regex passes were a measurable share of
287
+ // analyze(). Keyed by node identity: a resolved plan tree is never mutated after the parser
288
+ // posts it.
289
+ const relationIdByNode = new WeakMap ();
290
+ function relationIdOf(node ) {
291
+ let rid = relationIdByNode.get(node);
292
+ if (rid === undefined) {
293
+ rid = scanRelationId(node.name ?? '', node.detail ?? '');
294
+ relationIdByNode.set(node, rid);
295
+ }
296
+ return rid;
297
+ }
298
+
273
299
  // Structural operator kind for cachingOpportunity's composite detection: 'join' covers every
274
300
  // Spark join physical operator; 'union' is Spark's exact `Union` node. CartesianProduct is
275
301
  // deliberately excluded (out of scope, as in plan-summary.ts).
@@ -302,11 +328,12 @@ export function findCompositeCandidates(root ) {
302
328
  path.pop();
303
329
 
304
330
  const metricNames = (node.metrics ?? []).map((m) => m.name).sort().join(',');
305
- const childFingerprints = childResults.map((r) => r.fingerprint).join(',');
331
+ // Child digests, as in computePlanShapes: deterministic, so still comparable across executions.
332
+ const childFingerprints = childResults.map((r) => cyrb53(r.fingerprint)).join(',');
306
333
  const fingerprint = `${node.name}[${metricNames}]{${childFingerprints}}`;
307
334
 
308
335
  const leafRelationBytes = new Map ();
309
- const rid = scanRelationId(node.name ?? '', node.detail ?? '');
336
+ const rid = relationIdOf(node);
310
337
  if (rid) {
311
338
  const bytesMetric = (node.metrics ?? []).find((m) => m.name === FILES_READ_BYTES);
312
339
  leafRelationBytes.set(rid, (leafRelationBytes.get(rid) ?? 0) + (bytesMetric ? bytesMetric.value : 0));
@@ -343,7 +370,7 @@ function firstLeafRelationId(node ) {
343
370
  let found = null;
344
371
  walkPlanTree(node, (n) => {
345
372
  if (found) return;
346
- found = scanRelationId(n.name ?? '', n.detail ?? '');
373
+ found = relationIdOf(n);
347
374
  });
348
375
  return found;
349
376
  }
@@ -411,6 +438,58 @@ export function findDuplicateSubtrees(
411
438
  return results;
412
439
  }
413
440
 
441
+ // True when every occurrence also agrees node-for-node on normalized detail (filters, columns,
442
+ // scanned table, literals). The fingerprint ignores detail, so occurrences can be same-shaped
443
+ // branches over different data; only identical ones are repeated work that computing once would
444
+ // save. AQE query-stage numbers (`ShuffleQueryStage 718` vs `720`) name the same computation's
445
+ // runtime stage and are ignored too. On the 14 real logs, 270 of 546 groups differed beyond that.
446
+ function occurrencesHaveIdenticalDetails(occurrences ) {
447
+ const detailsOf = (root ) => {
448
+ const out = [];
449
+ walkPlanTree(root, (n) => out.push(n.detail ?? ''));
450
+ return out;
451
+ };
452
+ const normalize = (d ) => normalizeDetail(d).replace(/\b(\w+QueryStage)\s+\d+/g, '$1');
453
+ const first = detailsOf(occurrences[0]);
454
+ const firstNormalized = [];
455
+ for (const other of occurrences.slice(1)) {
456
+ const details = detailsOf(other);
457
+ if (details.length !== first.length) return false;
458
+ for (let i = 0; i < first.length; i++) {
459
+ if (details[i] === first[i]) continue;
460
+ firstNormalized[i] ??= normalize(first[i]);
461
+ if (normalize(details[i]) !== firstNormalized[i]) return false;
462
+ }
463
+ }
464
+ return true;
465
+ }
466
+
467
+ // Stages that run nothing but the duplicated occurrences' operators: every operator attributed to
468
+ // the stage is inside `nodes`. A stage shared with operators outside them (the join consuming the
469
+ // subtree, the other join side) does other work too, so its time isn't the subtree's to claim;
470
+ // counting it also let sibling groups claim the same stage twice. A WholeStageCodegen wrapper and
471
+ // an Exchange's write half aren't other work (the fused pipeline around an operator, the shuffle
472
+ // write of its output), so they're left out of both counts; on the 14 real logs they were the
473
+ // only outside node on 426 stages.
474
+ function countsTowardStage(node ) {
475
+ return node.exchangeRole !== 'write' && !node.name.startsWith('WholeStageCodegen');
476
+ }
477
+
478
+ function operatorCountByStage(nodes ) {
479
+ const counts = new Map ();
480
+ for (const node of nodes) {
481
+ if (!countsTowardStage(node)) continue;
482
+ for (const sid of node.stageIds ?? []) counts.set(sid, (counts.get(sid) ?? 0) + 1);
483
+ }
484
+ return counts;
485
+ }
486
+
487
+ function stageOperatorShares(nodes , executionCounts ) {
488
+ const shares = {};
489
+ for (const [sid, count] of operatorCountByStage(nodes)) shares[sid] = count / (executionCounts.get(sid) ?? count);
490
+ return shares;
491
+ }
492
+
414
493
  // Fingerprint matching compares operator + metric names only, not literal values or expr IDs
415
494
  // (see the finding's validationRequired text), so a small pattern repeated the bare minimum
416
495
  // number of times is the case most likely to be coincidental rather than real duplicated work.
@@ -521,12 +600,27 @@ function meetsRuntimeFloor(wasteMs , appDurationMs , floorP
521
600
  return appDurationMs == null || wasteMs >= appDurationMs * floorPct;
522
601
  }
523
602
 
603
+ // A stage that ran for less than floorPct of the run (a known duration): an estimate clipped to
604
+ // the stage can't reach floorPct, so every finding there grades info. A zero-length stage (no
605
+ // submission time on older Spark) is not skipped: it gets no estimate and keeps its own band.
606
+ function stageBelowRuntimeFloor(stage , ctx , floorPct ) {
607
+ const stageDurationMs = (stage.completedAt ?? 0) - (stage.submittedAt ?? 0);
608
+ const appDurationMs = computeAppDurationMs(ctx);
609
+ return appDurationMs != null && stageDurationMs > 0 && stageDurationMs < appDurationMs * floorPct;
610
+ }
611
+
524
612
  // Runs a raw waste delta through the same occupancy clip impact-estimator.ts applies before
525
613
  // display, so the runtime floor checks recoverable wall-clock, not a delta a physical floor
526
614
  // leaves unrecoverable. Falls back to the raw delta when occupancy data is unavailable.
527
- function clippedWasteMs(wasteMs , stageId , ctx ) {
615
+ // skew/straggler claims shorten the stage's longest task, hence shortensLongestTask (see occupancy.ts).
616
+ function clippedWasteMs(
617
+ wasteMs , stageId , ctx , removedCoreWorkMs , longestTaskAfterFixMs = 0,
618
+ ) {
528
619
  if (!ctx) return wasteMs;
529
- const est = estimateSingleStage(wasteMs, stageId, ctx.stages , ctx.occupancy);
620
+ const est = estimateSingleStage(
621
+ wasteMs, stageId, ctx.stages , ctx.occupancy,
622
+ { shortensLongestTask: true, removedCoreWorkMs, longestTaskAfterFixMs },
623
+ );
530
624
  return est ? est.wallClock.high : wasteMs;
531
625
  }
532
626
 
@@ -728,9 +822,10 @@ export const DETECTORS = [
728
822
  if (ratio <= this.thresholds.ratioWarn) return null;
729
823
  // Same absolute delta impact-estimator.ts's 'skew' case reports as savings; clipped the
730
824
  // same way before the floor check so the gate agrees with what's displayed.
731
- const wasteMs = Math.max(0, metric === 'P95/median' ? stage.taskDurationP95 - stage.taskDurationP50 : stage.taskDurationMax - stage.taskDurationP50);
825
+ const singleDelta = Math.max(0, metric === 'P95/median' ? stage.taskDurationP95 - stage.taskDurationP50 : stage.taskDurationMax - stage.taskDurationP50);
826
+ const wasteMs = tailRecoveryMs(stage, singleDelta);
732
827
  const appDurationMs = computeAppDurationMs(ctx);
733
- const floorWasteMs = clippedWasteMs(wasteMs, stage.id, ctx);
828
+ const floorWasteMs = clippedWasteMs(wasteMs, stage.id, ctx, tailRemovedWorkMs(stage, singleDelta), stragglerFixLongestTaskMs(stage));
734
829
  if (!meetsRuntimeFloor(floorWasteMs, appDurationMs, this.thresholds.floorPctWarn)) return null;
735
830
  const value = Math.round(ratio * 10) / 10;
736
831
  return {
@@ -746,9 +841,13 @@ export const DETECTORS = [
746
841
  {
747
842
  type: 'stageShape', scope: 'stage', order: 35, fixEffort: 'code', version: 1,
748
843
  docAnchor: '#bottleneck-stage-shape',
749
- thresholds: { pRatioMax: 0.5, oiRatioMax: 10, skewWarn: 3 },
844
+ // lowParallelismFloorPct: the same 0.5% runtime floor the tiered detectors use. Parallelizing a
845
+ // stage can't save more than the stage's own duration, so a shorter stage can't clear it; on
846
+ // the 14 real logs that was 2839 of 3005 lowParallelism findings (2168 on sub-second stages).
847
+ // App-wide idle capacity stays covered by utilization.
848
+ thresholds: { pRatioMax: 0.5, oiRatioMax: 10, skewWarn: 3, lowParallelismFloorPct: 0.005 },
750
849
  detect(
751
-
850
+
752
851
  stage ,
753
852
  ctx ,
754
853
  ) {
@@ -756,8 +855,9 @@ export const DETECTORS = [
756
855
  const execCount = (stage.executorStats ?? []).length;
757
856
  const cores = ctx?.app?.resources?.executor?.cores ?? 1;
758
857
  const totalCores = execCount * cores;
858
+ const stageDurationMs = (stage.completedAt ?? 0) - (stage.submittedAt ?? 0);
759
859
  // PRatio: under-parallelization.
760
- if (totalCores > 0) {
860
+ if (totalCores > 0 && !stageBelowRuntimeFloor(stage, ctx, this.thresholds.lowParallelismFloorPct)) {
761
861
  const pRatio = stage.taskCount / totalCores;
762
862
  if (pRatio < this.thresholds.pRatioMax) {
763
863
  out.push({
@@ -783,7 +883,6 @@ export const DETECTORS = [
783
883
  // TaskStageSkew: straggler cost vs stage wall-clock. Skip near-zero duration. Always info
784
884
  // like its siblings: this trigger forces the occupancy-clipped estimate to exactly zero on
785
885
  // every firing, so there's no wall-clock-backed tier left to gate on.
786
- const stageDurationMs = (stage.completedAt ?? 0) - (stage.submittedAt ?? 0);
787
886
  if (stageDurationMs > 0) {
788
887
  const ratio = stage.taskDurationMax / stageDurationMs;
789
888
  if (ratio > this.thresholds.skewWarn) {
@@ -802,13 +901,18 @@ export const DETECTORS = [
802
901
  {
803
902
  type: 'shuffle', scope: 'stage', order: 20, fixEffort: 'config', version: 1,
804
903
  docAnchor: '#bottleneck-shuffle',
805
- thresholds: { minBytes: 50 * MB },
904
+ // stageFloorPct: the 0.5% runtime floor. The shuffle claim is clipped to the stage, so on a
905
+ // shorter stage it graded info: 182 of 284 shuffle findings on the 14 real logs. The shuffle
906
+ // is still there on those stages; the floor is why they're dropped.
907
+ thresholds: { minBytes: 50 * MB, stageFloorPct: 0.005 },
806
908
  detect(
807
-
909
+
808
910
  stage ,
911
+ ctx ,
809
912
  ) {
810
913
  const bytes = stage.shuffleReadBytes;
811
914
  if (bytes <= this.thresholds.minBytes) return null;
915
+ if (stageBelowRuntimeFloor(stage, ctx, this.thresholds.stageFloorPct)) return null;
812
916
  return {
813
917
  type: 'shuffle', stageId: stage.id,
814
918
  impactBand: 'info',
@@ -819,7 +923,7 @@ export const DETECTORS = [
819
923
  },
820
924
  {
821
925
  type: 'partitionSizing', scope: 'stage', order: 22, fixEffort: 'config', version: 1,
822
- docAnchor: '#bottleneck-shuffle',
926
+ docAnchor: '#bottleneck-partition-sizing',
823
927
  thresholds: { skewRatio: 5, skewFloorBytes: 256 * MB, lowParTotalBytes: GB, lowParMaxTasks: 7, maxPartBytes: 5 * GB },
824
928
  detect(
825
929
 
@@ -868,9 +972,12 @@ export const DETECTORS = [
868
972
  {
869
973
  type: 'spill', scope: 'stage', order: 10, fixEffort: 'code', version: 1,
870
974
  docAnchor: '#bottleneck-spill',
871
- thresholds: { singleTaskDiskGiB: 1, singleTaskMemGiB: 4, highDiskGiB: 1, highTaskDiskMB: 512, highMemGiB: 4, medDiskMB: 256, medMemGiB: 1, skewRatio: 5, skewDiskFloorMB: 128, skewMemFloorMB: 256, skewMinTasks: 10 },
872
- detect( stage ) {
975
+ // stageFloorPct: the 0.5% runtime floor, as for shuffle (12 of 38 spill findings on the 14 real
976
+ // logs, all info). The spill is still there on those stages; the floor is why they're dropped.
977
+ thresholds: { singleTaskDiskGiB: 1, singleTaskMemGiB: 4, highDiskGiB: 1, highTaskDiskMB: 512, highMemGiB: 4, medDiskMB: 256, medMemGiB: 1, skewRatio: 5, skewDiskFloorMB: 128, skewMemFloorMB: 256, skewMinTasks: 10, stageFloorPct: 0.005 },
978
+ detect( stage , ctx ) {
873
979
  if (stage.memoryBytesSpilled === 0) return null;
980
+ if (stageBelowRuntimeFloor(stage, ctx, this.thresholds.stageFloorPct)) return null;
874
981
  const cls = stage.spillClassification;
875
982
  const classified = cls === 'skew' || cls === 'volume';
876
983
  const mag = computeSpillMagnitude(stage, this.thresholds);
@@ -899,13 +1006,19 @@ export const DETECTORS = [
899
1006
  lowInfoPct100: 5,
900
1007
  // NOT SOURCED: our own noise floor so a stage that barely ran doesn't flag either direction.
901
1008
  minRunTimeMs: 10000,
1009
+ // lowInfoFloorPct: the 0.5% runtime floor stageShape's lowParallelism uses. The low-GC note is
1010
+ // an app-level memory-sizing signal; on a stage shorter than this share of the run it adds
1011
+ // nothing to that call. On the 14 real logs that was 464 of 685 low-GC notes (of 701 gc
1012
+ // findings); the low-GC pattern is still true on those stages, the floor is why they're dropped.
1013
+ lowInfoFloorPct: 0.005,
902
1014
  },
903
1015
  detect(
904
1016
 
905
-
1017
+
906
1018
 
907
1019
 
908
1020
  stage ,
1021
+ ctx ,
909
1022
  ) {
910
1023
  const pct = stage.gcPct;
911
1024
  if ((stage.executorRunTime ?? 0) >= this.thresholds.minRunTimeMs
@@ -921,7 +1034,8 @@ export const DETECTORS = [
921
1034
  }
922
1035
  // Low-GC (cost) branch: only for stages that ran long enough to be meaningful.
923
1036
  if ((stage.executorRunTime ?? 0) >= this.thresholds.minRunTimeMs
924
- && pct < this.thresholds.lowInfoPct100) {
1037
+ && pct < this.thresholds.lowInfoPct100
1038
+ && !stageBelowRuntimeFloor(stage, ctx, this.thresholds.lowInfoFloorPct)) {
925
1039
  const value = Math.round(pct * 10) / 10;
926
1040
  return {
927
1041
  type: 'gc', stageId: stage.id, direction: 'low',
@@ -943,19 +1057,29 @@ export const DETECTORS = [
943
1057
  // stages, sub-second/sub-64MB host differences produce huge noise ratios. 1s is above
944
1058
  // per-task jitter but below genuine slow-host stages; 64MB mirrors the spill disk-skew floor.
945
1059
  floorMs: 1000, floorBytes: 64 * MB,
1060
+ // stageFloorPct: the tiered detectors' 0.5% runtime floor. On a stage shorter than this share
1061
+ // of the run a slow host can't cost that much: duration findings are clipped to the stage
1062
+ // and graded info, and a byte-dimension imbalance (no time estimate, so its ratio tier was
1063
+ // its band) was graded info here since 6298149. So the stage is skipped: on the 14 real logs
1064
+ // 324 of 452 slowHost findings, all info. The imbalance is still there on those stages; the
1065
+ // floor is why they're dropped.
1066
+ stageFloorPct: 0.005,
946
1067
  },
947
1068
  detect(
948
1069
 
949
1070
 
950
1071
 
951
1072
 
1073
+
952
1074
 
953
1075
 
954
1076
  stage ,
1077
+ ctx ,
955
1078
  ) {
956
1079
  const hosts = stage.hostStats ?? [];
957
1080
  const execs0 = stage.executorStats ?? [];
958
1081
  if ((hosts.length < this.thresholds.minHosts && execs0.length < this.thresholds.minHosts) || stage.taskCount < this.thresholds.minTasks) return null;
1082
+ if (stageBelowRuntimeFloor(stage, ctx, this.thresholds.stageFloorPct)) return null;
959
1083
  const out = [];
960
1084
  if (hosts.length >= this.thresholds.minHosts) {
961
1085
  const means = hosts.map(h => ({ host: h.host, taskCount: h.taskCount, mean: h.totalDuration / h.taskCount }));
@@ -1110,31 +1234,51 @@ export const DETECTORS = [
1110
1234
  docAnchor: '#bottleneck-straggler',
1111
1235
  // floorPctWarn/floorPctCrit are re-exported as STRAGGLER_FLOOR_PCT_WARN/CRIT and reused as
1112
1236
  // impact-band.ts's global noise floor: keep the two in sync.
1113
- thresholds: { minTasks: 10, shareWarn: 0.05, warnPct: 0.10, critPct: 0.20, floorPctWarn: STRAGGLER_FLOOR_PCT_WARN, floorPctCrit: STRAGGLER_FLOOR_PCT_CRIT },
1237
+ // shareWarnAtFloor: in a large stage, the few stragglers that gate it for tens of seconds can be
1238
+ // only 2.5-5% of its tasks. Scored against a task-level replay of every stage on 14 real logs
1239
+ // (recoverable = replay with each task over 4x P50 capped at P50), admitting 2.5-5% shares
1240
+ // whose clipped tail already clears floorPctWarn found 3 such stages (10-48s) for 1 borderline
1241
+ // miss; admitting every 2.5% share instead added 86 findings below the floor.
1242
+ // A stage shorter than floorPctWarn of the run is skipped outright: its tail can't cost more
1243
+ // than the stage's own duration, so every finding there graded info. On the 14 real logs that
1244
+ // was 671 of 753 straggler findings, none above info; the slow tail is still real on those
1245
+ // stages, the floor is why they're dropped.
1246
+ thresholds: { minTasks: 10, shareWarn: 0.05, shareWarnAtFloor: 0.025, warnPct: 0.10, critPct: 0.20, floorPctWarn: STRAGGLER_FLOOR_PCT_WARN, floorPctCrit: STRAGGLER_FLOOR_PCT_CRIT },
1114
1247
  detect(
1115
1248
 
1116
-
1249
+
1250
+
1251
+
1252
+
1117
1253
 
1118
1254
  stage ,
1119
1255
  ctx ,
1120
1256
  ) {
1121
1257
  if (stage.taskCount < this.thresholds.minTasks) return null;
1258
+ const appDurationMs = computeAppDurationMs(ctx);
1259
+ if (stageBelowRuntimeFloor(stage, ctx, this.thresholds.floorPctWarn)) return null;
1122
1260
  const stragglerShare = (stage.stragglerCount ?? 0) / stage.taskCount;
1123
- if ((stage.speculativeTasks ?? 0) === 0 && stragglerShare <= this.thresholds.shareWarn) return null;
1124
1261
  const useSpeculative = (stage.speculativeTasks ?? 0) > 0;
1262
+ if (!useSpeculative && stragglerShare <= this.thresholds.shareWarnAtFloor) return null;
1125
1263
  const speculativeShare = useSpeculative ? stage.speculativeTasks / stage.taskCount : 0;
1126
1264
  // Same absolute delta impact-estimator.ts's straggler/stageShape case reports as savings: a
1127
1265
  // high straggler/speculative share on a stage whose tasks barely vary models near-zero
1128
1266
  // savings, so it must not outrank 'info'. Clipped the same way before the floor check.
1129
- const wasteMs = Math.max(0, stage.taskDurationMax - stage.taskDurationP50);
1130
- const appDurationMs = computeAppDurationMs(ctx);
1131
- const floorWasteMs = clippedWasteMs(wasteMs, stage.id, ctx);
1267
+ const longestTaskAfterFixMs = stragglerFixLongestTaskMs(stage);
1268
+ const singleDelta = Math.max(0, stage.taskDurationMax - longestTaskAfterFixMs);
1269
+ const wasteMs = tailRecoveryMs(stage, singleDelta);
1270
+ const floorWasteMs = clippedWasteMs(wasteMs, stage.id, ctx, tailRemovedWorkMs(stage, singleDelta), longestTaskAfterFixMs);
1132
1271
  const meetsWarnFloor = meetsRuntimeFloor(floorWasteMs, appDurationMs, this.thresholds.floorPctWarn);
1133
1272
  const meetsCritFloor = meetsRuntimeFloor(floorWasteMs, appDurationMs, this.thresholds.floorPctCrit);
1273
+ // The lower gate needs positive evidence the tail matters: meetsRuntimeFloor passes by
1274
+ // default when the app's duration is unknown (an incomplete run), which isn't that.
1275
+ const stragglerShareFires = stragglerShare > this.thresholds.shareWarn
1276
+ || (stragglerShare > this.thresholds.shareWarnAtFloor && appDurationMs != null && meetsWarnFloor);
1277
+ if (!useSpeculative && !stragglerShareFires) return null;
1134
1278
  const speculativeTier = speculativeShare >= this.thresholds.critPct && meetsCritFloor ? 'critical'
1135
1279
  : speculativeShare >= this.thresholds.warnPct && meetsWarnFloor ? 'warning' : 'info';
1136
1280
  // Straggler share has no dedicated critical tier per detector-contract.md; only warning.
1137
- const stragglerTier = stragglerShare > this.thresholds.shareWarn && meetsWarnFloor ? 'warning' : 'info';
1281
+ const stragglerTier = stragglerShareFires && meetsWarnFloor ? 'warning' : 'info';
1138
1282
  // Fixed fallback: overwritten by deriveImpactBand when this finding gets a real wallClock
1139
1283
  // estimate (the common case). Only surfaces on the rare occupancy-sweep miss.
1140
1284
  const impactBand = 'info';
@@ -1163,7 +1307,7 @@ export const DETECTORS = [
1163
1307
  },
1164
1308
  {
1165
1309
  type: 'speculationWaste', scope: 'stage', order: 71, fixEffort: 'config', version: 1,
1166
- docAnchor: '#bottleneck-straggler',
1310
+ docAnchor: '#bottleneck-speculation-waste',
1167
1311
  thresholds: { minWasted: 5, minWasteMs: 60000 },
1168
1312
  detect(
1169
1313
 
@@ -1209,12 +1353,17 @@ export const DETECTORS = [
1209
1353
  {
1210
1354
  type: 'tinyTask', scope: 'stage', order: 80, fixEffort: 'code', version: 1,
1211
1355
  docAnchor: '#bottleneck-tiny-tasks',
1212
- thresholds: { minTasks: 100, maxP50: 500, maxP95: 1000 },
1356
+ // stageFloorPct: the tiered detectors' 0.5% runtime floor. Coalescing can't save more than the
1357
+ // stage's own duration, so on a shorter stage every finding graded info: on the 14 real logs
1358
+ // 132 of 151 tinyTask findings. The tasks are still tiny there; the floor is why they're dropped.
1359
+ thresholds: { minTasks: 100, maxP50: 500, maxP95: 1000, stageFloorPct: 0.005 },
1213
1360
  detect(
1214
-
1361
+
1215
1362
  stage ,
1363
+ ctx ,
1216
1364
  ) {
1217
1365
  if (stage.taskCount < this.thresholds.minTasks) return null;
1366
+ if (stageBelowRuntimeFloor(stage, ctx, this.thresholds.stageFloorPct)) return null;
1218
1367
  if (stage.taskDurationP50 > this.thresholds.maxP50 || stage.taskDurationP95 > this.thresholds.maxP95) return null;
1219
1368
  const coalesceTo = Math.max(1, Math.round(stage.taskCount / 10));
1220
1369
  const fix = stage.shuffleReadBytes > 0
@@ -1250,23 +1399,42 @@ export const DETECTORS = [
1250
1399
 
1251
1400
  ctx ,
1252
1401
  ) {
1253
- const { app, stages } = ctx;
1402
+ const { app, stages, executorsAdded, executorsRemoved } = ctx;
1254
1403
  // Nullish (not falsy) check: a literal startTime:0 must not be treated as "missing".
1255
1404
  if (!app || app.startTime == null || stages.size === 0) return null;
1256
- let firstTaskLaunch = Infinity;
1405
+ let firstStageSubmitted = Infinity;
1257
1406
  for (const stage of stages.values()) {
1258
- if (stage.submittedAt > 0 && stage.submittedAt < firstTaskLaunch) firstTaskLaunch = stage.submittedAt;
1407
+ if (stage.submittedAt > 0 && stage.submittedAt < firstStageSubmitted) firstStageSubmitted = stage.submittedAt;
1259
1408
  }
1260
1409
  // No stage ever recorded a submission timestamp: no basis to measure a startup gap against.
1261
- // Exposed now that a literal app.startTime:0 no longer short-circuits this detector entirely.
1262
- if (!Number.isFinite(firstTaskLaunch)) return null;
1263
- const gapSeconds = (firstTaskLaunch - app.startTime) / 1000;
1410
+ if (!Number.isFinite(firstStageSubmitted)) return null;
1411
+ // The wait is from the first runnable stage to the first executor, not from app start: the
1412
+ // driver's own startup before its first job (36-47s on every real log, whatever the
1413
+ // executors did) isn't something executors could shorten. On the 9 real logs that fired,
1414
+ // the first executor arrived 146-192s after the first stage on two whose old gap read ~40s,
1415
+ // and 2s after it on one the old gap flagged critical at 42s.
1416
+ // An executor added before the first stage only counts if it was still alive at submission:
1417
+ // with dynamic allocation scaling to zero, early executors can idle out before any job runs.
1418
+ const removedAt = new Map ();
1419
+ for (const ev of executorsRemoved) removedAt.set(ev.executorId, ev.timestamp);
1420
+ let firstExecutorAdded = Infinity;
1421
+ for (const e of executorsAdded) {
1422
+ if (!(e.timestamp > 0)) continue;
1423
+ if (e.timestamp <= firstStageSubmitted) {
1424
+ const removed = removedAt.get(e.executorId);
1425
+ if (removed == null || removed > firstStageSubmitted) return null;
1426
+ } else if (e.timestamp < firstExecutorAdded) {
1427
+ firstExecutorAdded = e.timestamp;
1428
+ }
1429
+ }
1430
+ if (!Number.isFinite(firstExecutorAdded)) return null;
1431
+ const gapSeconds = (firstExecutorAdded - firstStageSubmitted) / 1000;
1264
1432
  if (gapSeconds <= this.thresholds.gapSeconds) return null;
1265
1433
  const value = Math.round(gapSeconds);
1266
1434
  return {
1267
1435
  type: 'coldStart', stageId: null, impactBand: 'warning',
1268
1436
  metric: 'startupGapSeconds', value,
1269
- recommendation: `The first task waited ${value}s for executors to become available: keep a warm pool of idle executors, or if using dynamic allocation, raise the minimum/initial executor count so it doesn't scale up from zero.`,
1437
+ recommendation: `The first stage waited ${value}s for an executor to become available: keep a warm pool of idle executors, or if using dynamic allocation, raise the minimum/initial executor count so it doesn't scale up from zero.`,
1270
1438
  };
1271
1439
  },
1272
1440
  },
@@ -1439,7 +1607,7 @@ export const DETECTORS = [
1439
1607
  // block-access events, so a literal cache hit rate isn't derivable). Two per-RDD tiered
1440
1608
  // checks over rddInfo snapshots: partial caching and disk spillover. An RDD can produce both.
1441
1609
  type: 'cacheUtilization', scope: 'app', order: 103, fixEffort: 'code', version: 1,
1442
- docAnchor: '#memory-model',
1610
+ docAnchor: '#bottleneck-cache-utilization',
1443
1611
  thresholds: {
1444
1612
  cachedRatioWarn: 0.50, cachedRatioInfo: 0.90,
1445
1613
  diskRatioWarn: 0.40, diskRatioInfo: 0.15,
@@ -1483,7 +1651,7 @@ export const DETECTORS = [
1483
1651
  // half of the "Wasted Cores Ratio" (idle-core half is memoryUtilization's idleCores).
1484
1652
  // NO_PREF stays in the denominator only: shuffle-read stages legitimately report it.
1485
1653
  type: 'coreLocality', scope: 'app', order: 103, fixEffort: 'config', version: 1,
1486
- docAnchor: '#bottleneck-utilization',
1654
+ docAnchor: '#bottleneck-core-locality',
1487
1655
  thresholds: { minTasks: 50, warnRatio: 0.15, critRatio: 0.35 },
1488
1656
  detect(
1489
1657
 
@@ -1513,6 +1681,7 @@ export const DETECTORS = [
1513
1681
  // re-provisioning, not normal scale-down). Reuses utilization's add/remove matching, but
1514
1682
  // measures lifetime against a threshold instead of aggregate active-time.
1515
1683
  type: 'autoscalingChurn', scope: 'app', order: 103, fixEffort: 'config', version: 1,
1684
+ docAnchor: '#bottleneck-autoscaling-churn',
1516
1685
  thresholds: { shortLivedMs: 120_000, warningPct: 0.30, criticalPct: 0.60, minExecutors: 5 },
1517
1686
  detect(
1518
1687
 
@@ -1554,7 +1723,7 @@ export const DETECTORS = [
1554
1723
  // Cross-execution relation reuse: flags an input relation scanned by two or more SQL
1555
1724
  // executions in one run, firing on real relation names (parquet:..., jdbc:...).
1556
1725
  type: 'cachingOpportunity', scope: 'app', order: 105, fixEffort: 'code', version: 1,
1557
- docAnchor: '#bottleneck-utilization',
1726
+ docAnchor: '#bottleneck-caching-opportunity',
1558
1727
  thresholds: { minExecutions: 2 },
1559
1728
  detect( ctx ) {
1560
1729
  const sql = ctx.sql;
@@ -1581,7 +1750,7 @@ export const DETECTORS = [
1581
1750
  // Dedupe relations within one execution (self-joins count once), summing read bytes per relation.
1582
1751
  const perExec = new Map ();
1583
1752
  walkPlanTree(exec.planTree, (node) => {
1584
- const rid = scanRelationId(node.name ?? '', node.detail ?? '');
1753
+ const rid = relationIdOf(node);
1585
1754
  if (!rid) return;
1586
1755
  const bytesMetric = (node.metrics ?? []).find(m => m.name === FILES_READ_BYTES);
1587
1756
  perExec.set(rid, (perExec.get(rid) ?? 0) + (bytesMetric ? bytesMetric.value : 0));
@@ -1845,9 +2014,13 @@ export const DETECTORS = [
1845
2014
  {
1846
2015
  type: 'duplicatePlanSubtree', scope: 'sql', order: 130, fixEffort: 'code', version: 2,
1847
2016
  docAnchor: '#bottleneck-duplicate-plan-subtree',
1848
- thresholds: { minSubtreeSize: 3, minOccurrences: 2 },
2017
+ // stageFloorPct: the 0.5% runtime floor. The claim counts at most each linked stage's own
2018
+ // task-active time, so a repeat whose stages together lasted less than this share of the run
2019
+ // graded info: 340 of 545 findings on the 14 real logs. The repeat is still in the plan; the
2020
+ // floor is why they're dropped. A repeat with no linked stage time is kept.
2021
+ thresholds: { minSubtreeSize: 3, minOccurrences: 2, stageFloorPct: 0.005 },
1849
2022
  detect(
1850
-
2023
+
1851
2024
  sqlExec ,
1852
2025
  ctx ,
1853
2026
  ) {
@@ -1855,27 +2028,44 @@ export const DETECTORS = [
1855
2028
  const groups = findDuplicateSubtrees(sqlExec.planTree, this.thresholds);
1856
2029
  if (groups.length === 0) return null;
1857
2030
  const fallbackStageIds = stageIdsForSqlExec(sqlExec.id, ctx.stages);
1858
- return groups.map((g) => {
2031
+ const executionNodes = [];
2032
+ walkPlanTree(sqlExec.planTree, (node) => executionNodes.push(node));
2033
+ const operatorsByStage = operatorCountByStage(executionNodes);
2034
+ const appDurationMs = computeAppDurationMs(ctx);
2035
+ const findings = groups.map((g) => {
1859
2036
  const nodes = [];
1860
2037
  for (const n of g.nodes) walkPlanTree(n, (node) => nodes.push(node));
1861
2038
  const stageIds = unionStageIds(nodes, fallbackStageIds);
2039
+ let stagesMs = 0;
2040
+ for (const id of stageIds) {
2041
+ const stage = ctx.stages.get(id);
2042
+ if (stage) stagesMs += Math.max(0, (stage.completedAt ?? 0) - (stage.submittedAt ?? 0));
2043
+ }
2044
+ if (appDurationMs != null && stagesMs > 0 && stagesMs < appDurationMs * this.thresholds.stageFloorPct) return null;
2045
+ const occurrencesIdentical = occurrencesHaveIdenticalDetails(g.nodes);
2046
+ const stageShares = stageOperatorShares(nodes, operatorsByStage);
1862
2047
  // resolvePlanTree always sets id; safe downstream of it.
1863
2048
  const planNodeIds = nodes.map((n) => n.id ).filter(Boolean);
1864
2049
  const touching = g.sampleRelation ? ` (touching ${g.sampleRelation})` : '';
2050
+ const differing = occurrencesIdentical ? '' : ' Their filters, columns or scanned tables differ, so the repeats may compute different data.';
1865
2051
  return {
1866
2052
  type: 'duplicatePlanSubtree', executionId: sqlExec.id, stageIds, planNodeIds,
2053
+ stageShares, occurrencesIdentical,
1867
2054
  // Fixed fallback: overwritten by deriveImpactBand when this finding gets a real
1868
- // wallClock estimate (the common case). Only surfaces on the rare occupancy-sweep miss.
1869
- impactBand: 'warning', metric: 'subtreeOccurrences', value: g.occurrences,
2055
+ // wallClock estimate. Repeats with differing details get no estimate (nothing is
2056
+ // known to be recomputed), and neither does a subtree with no stage of its own.
2057
+ impactBand: occurrencesIdentical && Object.keys(stageShares).length > 0 ? 'warning' : 'info',
2058
+ metric: 'subtreeOccurrences', value: g.occurrences,
1870
2059
  rootName: g.rootName, subtreeSize: g.subtreeSize, sampleRelation: g.sampleRelation,
1871
2060
  groupIndex: g.groupIndex,
1872
- confidence: duplicateSubtreeConfidence(g.subtreeSize, g.occurrences, this.thresholds),
2061
+ confidence: occurrencesIdentical ? duplicateSubtreeConfidence(g.subtreeSize, g.occurrences, this.thresholds) : 'low',
1873
2062
  validationRequired: 'Duplicate-subtree matching compares operator names and metric names only, not literal values or expr IDs: confirm the repeated work is real in the Spark SQL plan tab before acting.',
1874
- recommendation: g.isExchangeRoot
2063
+ recommendation: (g.isExchangeRoot
1875
2064
  ? `A ${g.subtreeSize}-node subtree rooted at ${pathBasename(g.rootName)} repeats ${g.occurrences}x in this plan${touching}: this looks like a possible missed exchange reuse; check whether the same shuffle could be computed once and reused.`
1876
- : `A ${g.subtreeSize}-node subtree rooted at ${pathBasename(g.rootName)} repeats ${g.occurrences}x in this plan${touching}: consider caching/persisting the shared computation or check for a duplicated query branch.`,
2065
+ : `A ${g.subtreeSize}-node subtree rooted at ${pathBasename(g.rootName)} repeats ${g.occurrences}x in this plan${touching}: consider caching/persisting the shared computation or check for a duplicated query branch.`) + differing,
1877
2066
  };
1878
- });
2067
+ }).filter((f) => f !== null);
2068
+ return findings.length > 0 ? findings : null;
1879
2069
  },
1880
2070
  },
1881
2071
  {