sparkforensics-mcp 0.3.0 → 0.5.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (92) hide show
  1. package/bin/sparkforensics-mcp.mjs +41 -10
  2. package/package.json +1 -1
  3. package/vendor-core/allocation.js +106 -0
  4. package/vendor-core/analyzer.js +168 -60
  5. package/vendor-core/check-coverage.js +88 -0
  6. package/vendor-core/cli/budgets.js +54 -27
  7. package/vendor-core/cli/collect-run.js +84 -32
  8. package/vendor-core/cli/regression-budgets.js +83 -0
  9. package/vendor-core/cli/threshold-config.js +28 -0
  10. package/vendor-core/comparison-verdict.js +177 -0
  11. package/vendor-core/core-source-hash.txt +1 -0
  12. package/vendor-core/core-usage-locality.js +56 -2
  13. package/vendor-core/detector-docs.js +58 -0
  14. package/vendor-core/detectors.js +1094 -500
  15. package/vendor-core/docs-config.js +0 -36
  16. package/vendor-core/docs-content/chapters/nav-index.json +31 -0
  17. package/vendor-core/docs-content/detection/cache.md +3 -2
  18. package/vendor-core/docs-content/detection/cfg.md +9 -8
  19. package/vendor-core/docs-content/detection/chrn.md +1 -2
  20. package/vendor-core/docs-content/detection/cold.md +4 -2
  21. package/vendor-core/docs-content/detection/cstor.md +9 -0
  22. package/vendor-core/docs-content/detection/fail.md +3 -2
  23. package/vendor-core/docs-content/detection/gc.md +3 -2
  24. package/vendor-core/docs-content/detection/host.md +2 -1
  25. package/vendor-core/docs-content/detection/local.md +1 -1
  26. package/vendor-core/docs-content/detection/mem.md +5 -2
  27. package/vendor-core/docs-content/detection/plan.md +2 -1
  28. package/vendor-core/docs-content/detection/sfail.md +2 -1
  29. package/vendor-core/docs-content/detection/shape.md +5 -4
  30. package/vendor-core/docs-content/detection/skew.md +3 -1
  31. package/vendor-core/docs-content/detection/slow.md +2 -2
  32. package/vendor-core/docs-content/detection/spec.md +2 -3
  33. package/vendor-core/docs-content/detection/spill.md +1 -1
  34. package/vendor-core/docs-site-config.js +3 -0
  35. package/vendor-core/effective-conf.js +107 -0
  36. package/vendor-core/efficiency-model.js +8 -6
  37. package/vendor-core/event-handlers.js +321 -44
  38. package/vendor-core/event-schemas.js +23 -0
  39. package/vendor-core/evidence-report.js +432 -115
  40. package/vendor-core/export-data.js +79 -6
  41. package/vendor-core/finding-action-label.js +9 -88
  42. package/vendor-core/finding-filter-predicate.js +9 -0
  43. package/vendor-core/finding-generic-recommendation.js +26 -105
  44. package/vendor-core/finding-names.js +28 -45
  45. package/vendor-core/finding-presentation.js +368 -0
  46. package/vendor-core/finding-tag-help.js +110 -0
  47. package/vendor-core/finding-types.js +373 -0
  48. package/vendor-core/findings-of-type.js +11 -0
  49. package/vendor-core/format-utils.js +96 -30
  50. package/vendor-core/html-export.js +51 -0
  51. package/vendor-core/impact-band.js +21 -8
  52. package/vendor-core/impact-estimator.js +25 -520
  53. package/vendor-core/impact-format.js +115 -0
  54. package/vendor-core/impact-model.js +197 -0
  55. package/vendor-core/ingest.js +6 -2
  56. package/vendor-core/intervals.js +13 -0
  57. package/vendor-core/list-runs.js +7 -5
  58. package/vendor-core/load-vendored.js +70 -5
  59. package/vendor-core/mcp-server-factory.js +14 -10
  60. package/vendor-core/mcp-tools.js +105 -45
  61. package/vendor-core/model-assembler.js +35 -1
  62. package/vendor-core/occupancy.js +1 -1
  63. package/vendor-core/parser-worker.js +2 -2
  64. package/vendor-core/plan-graph-model.js +3 -2
  65. package/vendor-core/plan-node-detail.js +1 -1
  66. package/vendor-core/proxy.js +3 -1
  67. package/vendor-core/python-stage.js +25 -0
  68. package/vendor-core/recommendation-rollup.js +70 -3
  69. package/vendor-core/redact.js +96 -37
  70. package/vendor-core/remediation.js +20 -0
  71. package/vendor-core/run-comparison.js +73 -29
  72. package/vendor-core/run-interpretation.js +291 -0
  73. package/vendor-core/run-metrics.js +198 -0
  74. package/vendor-core/run-outcome.js +74 -0
  75. package/vendor-core/run-payload.js +17 -0
  76. package/vendor-core/run-shape.js +40 -0
  77. package/vendor-core/run-totals.js +24 -0
  78. package/vendor-core/run-verdict.js +352 -0
  79. package/vendor-core/scaling-sim.js +4 -5
  80. package/vendor-core/scorecard-estimates.js +63 -0
  81. package/vendor-core/session-snapshot.js +7 -0
  82. package/vendor-core/shs-schemas.js +2 -2
  83. package/vendor-core/spark-memory.js +17 -0
  84. package/vendor-core/sql-stages.js +11 -0
  85. package/vendor-core/stage-plan-nodes.js +18 -0
  86. package/vendor-core/stage-quantiles.js +6 -0
  87. package/vendor-core/threshold-overrides.js +160 -0
  88. package/vendor-core/threshold-summary.js +11 -33
  89. package/vendor-core/types.js +54 -42
  90. package/vendor-core/wall-clock.js +1 -12
  91. package/vendor-core/wasted-core-hours.js +12 -9
  92. package/vendor-core/write-targets.js +312 -0
@@ -14,6 +14,7 @@ import {
14
14
  DriverAccumUpdatesEventSchema,
15
15
  ExecutorAddedEventSchema,
16
16
  ExecutorRemovedEventSchema,
17
+ BlockUpdatedEventSchema,
17
18
 
18
19
 
19
20
  } from './event-schemas.js';
@@ -21,8 +22,9 @@ import { assertNever } from './assert-never.js';
21
22
  import { finalizeStage } from './stage-quantiles.js';
22
23
  import { MAX_FAILURE_DETAILS_PER_STAGE, extractTaskFailureDetail, taskFailureKey, } from './task-failure.js';
23
24
  import { computeRunAggregates } from './run-aggregates.js';
25
+ import { parseSparkMemoryMB } from './spark-memory.js';
24
26
 
25
-
27
+
26
28
 
27
29
 
28
30
  // Internal parser-state shapes: the real runtime objects the handlers build and mutate, not the
@@ -62,7 +64,22 @@ import { computeRunAggregates } from './run-aggregates.js';
62
64
 
63
65
 
64
66
 
67
+
68
+
69
+
70
+
65
71
 
72
+
73
+
74
+ // Live per-RDD block residency rebuilt from SparkListenerBlockUpdated. Keyed by partition and
75
+ // executor because a block's status is per BlockManager: replicas and re-caches on another
76
+ // executor are separate entries, as in Spark's own AppStatusListener.
77
+
78
+
79
+
80
+
81
+
82
+
66
83
 
67
84
 
68
85
 
@@ -113,6 +130,8 @@ const MAX_TASK_SAMPLES = 20;
113
130
 
114
131
 
115
132
 
133
+
134
+
116
135
 
117
136
 
118
137
 
@@ -133,12 +152,29 @@ const MAX_TASK_SAMPLES = 20;
133
152
 
134
153
 
135
154
 
155
+
156
+
136
157
 
137
158
 
138
159
 
139
160
 
140
161
 
141
162
 
163
+
164
+
165
+
166
+
167
+
168
+
169
+
170
+
171
+
172
+
173
+
174
+
175
+
176
+
177
+
142
178
 
143
179
 
144
180
 
@@ -147,6 +183,12 @@ const MAX_TASK_SAMPLES = 20;
147
183
 
148
184
 
149
185
 
186
+
187
+
188
+
189
+
190
+
191
+
150
192
 
151
193
 
152
194
 
@@ -176,8 +218,15 @@ const MAX_TASK_SAMPLES = 20;
176
218
 
177
219
 
178
220
 
221
+
222
+
223
+
179
224
 
180
225
 
226
+
227
+
228
+
229
+
181
230
 
182
231
 
183
232
 
@@ -365,8 +414,11 @@ export function createState() {
365
414
  jobs: new Map(),
366
415
  executors: { added: [], removed: [] },
367
416
  skippedLines: 0,
417
+ unreadableSqlStarts: new Set(),
368
418
  accumState: new Map(),
369
419
  rddInfo: new Map(),
420
+ rddBlocks: new Map(),
421
+ rddBlockUpdates: 0,
370
422
  taskAccumStages: new Map(),
371
423
  pendingAdaptiveUpdates: new Map(),
372
424
  resolvedPlanExecutions: new Set(),
@@ -404,23 +456,7 @@ export function normalizeSparkProperties(
404
456
  return map;
405
457
  }
406
458
 
407
- // Parse a Spark memory-size string to MiB. Spark's JVM-memory configs use bytesConf(ByteUnit.MiB),
408
- // so a bare number means MiB. A k/m/g/t suffix sets the unit (trailing "b" redundant); a lone "b"
409
- // ("10b") means bytes.
410
- export function parseSparkMemoryMB(value ) {
411
- if (value == null) return null;
412
- const m = String(value).trim().toLowerCase().match(/^([\d.]+)\s*([kmgt]?)(b?)$/);
413
- if (!m) return null;
414
- const n = parseFloat(m[1]);
415
- if (!Number.isFinite(n)) return null;
416
- switch (m[2]) {
417
- case 'k': return Math.round(n / 1024);
418
- case 'g': return Math.round(n * 1024);
419
- case 't': return Math.round(n * 1024 * 1024);
420
- case 'm': return Math.round(n);
421
- default: return m[3] === 'b' ? Math.round(n / (1024 * 1024)) : Math.round(n);
422
- }
423
- }
459
+ export { parseSparkMemoryMB };
424
460
 
425
461
  // Derive an allocated-resource summary from the Spark config map. Absent keys degrade to null,
426
462
  // not guessed defaults.
@@ -478,6 +514,7 @@ function snapshotEvidenceInputs(state ) {
478
514
  function appMessage(state ) {
479
515
  const evidenceInputs = snapshotEvidenceInputs(state);
480
516
  state.app .evidenceInputs = evidenceInputs;
517
+ state.app .rddBlockUpdates = state.rddBlockUpdates;
481
518
  return {
482
519
  type: 'app',
483
520
  data: { ...state.app , rddInfo: snapshotRddInfo(state.rddInfo) },
@@ -504,25 +541,11 @@ function internTaskFailure(stage , endReason
504
541
  return detail;
505
542
  }
506
543
 
507
- export function accumulateTask(event , state ) {
508
- const stageId = event['Stage ID'];
509
- const stage = state.stages.get(stageId);
510
- if (!stage) return null;
511
- // Late TaskEnd for a stage whose StageCompleted already freed taskAttempts (finalizeStage): its
512
- // stats are already baked into the finalized stage, don't re-add.
513
- if (stage.taskAttempts === null) return null;
514
-
515
- state.evidenceInputs.taskRecords++;
516
-
517
- const accumulables = event['Task Info']?.Accumulables ?? [];
518
- for (const acc of accumulables) {
519
- if (!state.taskAccumStages.has(acc.ID)) state.taskAccumStages.set(acc.ID, new Set());
520
- state.taskAccumStages.get(acc.ID) .add(stageId);
521
- }
544
+ // 'Task Info' and its Failed/Killed/Speculative fields are optional in the schema; Partial<>
545
+ // lets the {} fallback type-check while reads below default via ??/||.
546
+
522
547
 
523
- // 'Task Info' and its Failed/Killed/Speculative fields are optional in the schema; Partial<>
524
- // lets the {} fallback type-check while reads below default via ??/||.
525
-
548
+ function taskRecordOf(event , failure ) {
526
549
  const info = event['Task Info'] ?? {};
527
550
  const m = event['Task Metrics'] ?? {};
528
551
  const sr = m['Shuffle Read Metrics'] ?? {};
@@ -533,14 +556,14 @@ export function accumulateTask(event , state
533
556
  const duration = (info['Finish Time'] ?? 0) - (info['Launch Time'] ?? 0);
534
557
  const failed = !!(info['Failed'] || info['Killed']);
535
558
 
536
- const record = {
559
+ return {
537
560
  duration, failed,
538
561
  taskId: info['Task ID'] ?? null,
539
562
  attemptNumber: info['Attempt Number'] ?? 0,
540
563
  launchTime: info['Launch Time'] ?? 0,
541
564
  finishTime: info['Finish Time'] ?? 0,
542
565
  reason: event['Task End Reason']?.['Reason'] ?? null,
543
- failure: failed ? internTaskFailure(stage, event['Task End Reason']) : null,
566
+ failure,
544
567
  speculative: info['Speculative'] === true,
545
568
  host: info['Host'] ?? '',
546
569
  executorId: info['Executor ID'] ?? '',
@@ -556,7 +579,37 @@ export function accumulateTask(event , state
556
579
  executorCpuTime: m['Executor CPU Time'] ?? 0,
557
580
  inputBytes: inp['Bytes Read'] ?? 0,
558
581
  outputBytes: out['Bytes Written'] ?? 0,
582
+ outputRecords: out['Records Written'] ?? null,
559
583
  };
584
+ }
585
+
586
+ export function accumulateTask(event , state ) {
587
+ const stageId = event['Stage ID'];
588
+ const stage = state.stages.get(stageId);
589
+ if (!stage) return null;
590
+ // Late TaskEnd for a stage whose StageCompleted already freed taskAttempts (finalizeStage): the
591
+ // finalized stage's figures stay as posted. A losing speculative attempt adds the wasted time the
592
+ // finalized stage never saw; any other task of a failed or earlier attempt is work only the
593
+ // metrics block reads, from lateAttemptWork.
594
+ if (stage.taskAttempts === null) {
595
+ const earlierAttempt = (event['Stage Attempt ID'] ?? 0) !== stage.stageAttemptId;
596
+ if (!accountLateSpeculativeLoser(event, stage) && (stage.stageFailureReason != null || earlierAttempt)) {
597
+ stage.lateAttemptWork = mergeAttemptTotals(taskAttemptTotals(taskRecordOf(event, null)), stage.lateAttemptWork);
598
+ }
599
+ return null;
600
+ }
601
+
602
+ state.evidenceInputs.taskRecords++;
603
+
604
+ const accumulables = event['Task Info']?.Accumulables ?? [];
605
+ for (const acc of accumulables) {
606
+ if (!state.taskAccumStages.has(acc.ID)) state.taskAccumStages.set(acc.ID, new Set());
607
+ state.taskAccumStages.get(acc.ID) .add(stageId);
608
+ }
609
+
610
+ const info = event['Task Info'] ?? {};
611
+ const failed = !!(info['Failed'] || info['Killed']);
612
+ const record = taskRecordOf(event, failed ? internTaskFailure(stage, event['Task End Reason']) : null);
560
613
 
561
614
  // Dedupe only when Index is present (always true for real logs). Without it every event is a
562
615
  // distinct task, preserving behavior for fixtures that omit Index.
@@ -566,6 +619,7 @@ export function accumulateTask(event , state
566
619
 
567
620
  if (!existing) {
568
621
  stage.taskAttempts.set(key, record);
622
+ if (record.speculative && !record.failed) stage.speculativeWinners.add(key);
569
623
  } else if (existing.failed && !record.failed) {
570
624
  // A retry succeeded where the earlier attempt failed: the earlier attempt's time was wasted.
571
625
  // Spark marks only the speculative COPY's Speculative flag, never the original it raced, so
@@ -580,8 +634,11 @@ export function accumulateTask(event , state
580
634
  stage.retryTaskSamples.push(taskRecordToSample(existing));
581
635
  }
582
636
  }
637
+ stage.lateAttemptWork = mergeAttemptTotals(discardedAttemptTotals(existing), stage.lateAttemptWork);
583
638
  stage.taskAttempts.set(key, record);
639
+ if (record.speculative) stage.speculativeWinners.add(key);
584
640
  } else {
641
+ stage.lateAttemptWork = mergeAttemptTotals(discardedAttemptTotals(record), stage.lateAttemptWork);
585
642
  // Non-winning duplicate (both failed, or a race where a winner is
586
643
  // already recorded): its time is waste, its metrics are discarded.
587
644
  if (existing.speculative || record.speculative) {
@@ -599,6 +656,23 @@ export function accumulateTask(event , state
599
656
  return null;
600
657
  }
601
658
 
659
+ // Spark kills the losing copy of a speculative race only once the stage finishes ("Stage
660
+ // cancelled: Stage finished"), so that loser's TaskEnd normally lands after StageCompleted. Count
661
+ // its time as speculation waste, pairing it the same way accumulateTask does: the late attempt is
662
+ // the speculative copy itself, or the original that a speculative winner beat. Every other stat
663
+ // of a late attempt stays excluded, as the finalized stage already posted them.
664
+ function accountLateSpeculativeLoser(event , stage ) {
665
+ const info = event['Task Info'];
666
+ if (info?.['Index'] == null) return false;
667
+ const key = `${event['Stage Attempt ID'] ?? 0}:${info['Index']}`;
668
+ if (info['Speculative'] !== true && !stage.speculativeWinners.has(key)) return false;
669
+ stage.speculationWasteMs += (info['Finish Time'] ?? 0) - (info['Launch Time'] ?? 0);
670
+ stage.speculationWastedAttempts++;
671
+ stage.lateSpeculationWaste = true;
672
+ stage.lateAttemptWork = mergeAttemptTotals(discardedAttemptTotals(taskRecordOf(event, null)), stage.lateAttemptWork);
673
+ return true;
674
+ }
675
+
602
676
  export function resolvePlanTree(
603
677
  rootInfo ,
604
678
  accumMap ,
@@ -803,10 +877,63 @@ export function endJob(event , state
803
877
  return { type: 'job', data: { ...job } };
804
878
  }
805
879
 
880
+ const SUMMED_ATTEMPT_FIELDS = [
881
+ 'taskCount', 'failedTasks', 'wastedAttempts', 'executorRunTime', 'executorCpuTime', 'jvmGCTime',
882
+ 'memoryBytesSpilled', 'diskBytesSpilled', 'shuffleReadBytes', 'shuffleWriteBytes', 'inputBytes', 'outputBytes',
883
+ ] ;
884
+
885
+ const addNullable = (a , b ) => (a == null ? b : b == null ? a : a + b);
886
+
887
+ function mergeAttemptTotals(a , b ) {
888
+ if (b == null) return a;
889
+ const totals = {
890
+ outputRecords: addNullable(a.outputRecords, b.outputRecords),
891
+ peakExecutionMemoryMax: Math.max(a.peakExecutionMemoryMax, b.peakExecutionMemoryMax),
892
+ durationMs: addNullable(a.durationMs, b.durationMs),
893
+ } ;
894
+ for (const field of SUMMED_ATTEMPT_FIELDS) totals[field] = a[field] + b[field];
895
+ return totals;
896
+ }
897
+
898
+ // One task's work; a late task adds no stage duration.
899
+ function taskAttemptTotals(t ) {
900
+ return {
901
+ taskCount: 1, failedTasks: t.failed ? 1 : 0, wastedAttempts: 0,
902
+ executorRunTime: t.executorRunTime, executorCpuTime: t.executorCpuTime, jvmGCTime: t.gcTime,
903
+ memoryBytesSpilled: t.memSpilled, diskBytesSpilled: t.diskSpilled,
904
+ shuffleReadBytes: t.shuffleRead, shuffleWriteBytes: t.shuffleWrite,
905
+ inputBytes: t.inputBytes, outputBytes: t.outputBytes, outputRecords: t.outputRecords,
906
+ peakExecutionMemoryMax: t.peakExecMem, durationMs: null,
907
+ };
908
+ }
909
+
910
+ // A task attempt the stage's figures drop because another attempt of the same task won (a retry
911
+ // after a failure, a speculative twin): its CPU, run time and I/O were still spent, so the
912
+ // metrics block counts them, but the task itself is already counted once.
913
+ function discardedAttemptTotals(t ) {
914
+ return { ...taskAttemptTotals(t), taskCount: 0, failedTasks: 0 };
915
+ }
916
+
917
+ // The replaced record's finalized attempt added to the attempts it had already folded. An attempt resubmitted before its StageCompleted was never finalized, so its
918
+ // tasks are not counted.
919
+ function foldEarlierAttempts(replaced ) {
920
+ if (!replaced) return null;
921
+ if (replaced.taskAttempts !== null) return replaced.earlierAttempts;
922
+ const attempt = {
923
+ ...Object.fromEntries(SUMMED_ATTEMPT_FIELDS.map((field) => [field, replaced[field]])) ,
924
+ outputRecords: replaced.outputRecords,
925
+ peakExecutionMemoryMax: replaced.peakExecutionMemoryMax ?? 0,
926
+ durationMs: replaced.submittedAt > 0 && replaced.completedAt >= replaced.submittedAt
927
+ ? replaced.completedAt - replaced.submittedAt : null,
928
+ };
929
+ return mergeAttemptTotals(attempt, replaced.earlierAttempts);
930
+ }
931
+
806
932
  export function submitStage(event , state ) {
807
933
  state.evidenceInputs.stageSubmissions++;
808
934
  const info = event['Stage Info'];
809
935
  const id = info['Stage ID'];
936
+ const replaced = state.stages.get(id);
810
937
  state.stages.set(id, {
811
938
  id, name: info['Stage Name'] ?? '', details: info['Details'] ?? '',
812
939
  submittedAt: info['Submission Time'] ?? 0, completedAt: 0,
@@ -814,13 +941,18 @@ export function submitStage(event , st
814
941
  shuffleReadBytes: 0, shuffleWriteBytes: 0, fetchWaitTime: 0,
815
942
  memoryBytesSpilled: 0, diskBytesSpilled: 0,
816
943
  jvmGCTime: 0, executorRunTime: 0, executorCpuTime: 0,
817
- inputBytes: 0, outputBytes: 0,
944
+ inputBytes: 0, outputBytes: 0, outputRecords: null,
818
945
  sqlExecutionId: state.stageToSqlExec.get(id) ?? null,
819
946
  parentIds: info['Parent IDs'] ?? [],
820
947
  hostStats: new Map(),
821
948
  speculativeTasks: 0,
822
949
  failureReasons: new Map(),
823
950
  stageFailureReason: null,
951
+ stageAttemptId: info['Stage Attempt ID'] ?? 0,
952
+ stageAttempts: (replaced?.stageAttempts ?? 0) + 1,
953
+ failedStageAttempts: replaced?.failedStageAttempts ?? 0,
954
+ earlierAttempts: foldEarlierAttempts(replaced),
955
+ lateAttemptWork: replaced?.lateAttemptWork ?? null,
824
956
  taskAttempts: new Map(),
825
957
  failureDetails: new Map(),
826
958
  retryTaskSamples: [],
@@ -828,6 +960,8 @@ export function submitStage(event , st
828
960
  wastedAttempts: 0,
829
961
  speculationWasteMs: 0,
830
962
  speculationWastedAttempts: 0,
963
+ speculativeWinners: new Set(),
964
+ lateSpeculationWaste: false,
831
965
  executorMetrics: new Map(),
832
966
  });
833
967
  mergeStageRddInfo(info, id, state);
@@ -847,11 +981,16 @@ export function mergeStageRddInfo(
847
981
  const prev = state.rddInfo.get(rddId);
848
982
  const stageIds = prev?.stageIds ?? new Set ();
849
983
  stageIds.add(id);
984
+ // Block updates are the authoritative source once seen: never let a later RDD Info snapshot
985
+ // (0 on Spark 2.3+) replace them, nor the NONE level an unpersist() leaves on ancestor RDDs
986
+ // listed by later stages.
987
+ const fromBlocks = prev?.storageSource === 'blockUpdates';
988
+ const persisted = Boolean(sl['Use Disk'] || sl['Use Memory']);
850
989
  state.rddInfo.set(rddId, {
851
990
  id: rddId,
852
991
  name: rdd['Name'] ?? '',
853
992
  callsite: rdd['Callsite'] ?? '',
854
- storageLevel: {
993
+ storageLevel: fromBlocks && !persisted ? prev.storageLevel : {
855
994
  useDisk: sl['Use Disk'] ?? false,
856
995
  useMemory: sl['Use Memory'] ?? false,
857
996
  deserialized: sl['Deserialized'] ?? false,
@@ -861,14 +1000,88 @@ export function mergeStageRddInfo(
861
1000
  // Merge forward, don't overwrite: an RDD cached for the first time in THIS stage legitimately
862
1001
  // reports 0 (snapshot reflects BlockManager state at submission). Keep the last real value on
863
1002
  // a resubmission instead of regressing to 0.
864
- numCachedPartitions: rdd['Number of Cached Partitions'] || prev?.numCachedPartitions || 0,
865
- memorySize: rdd['Memory Size'] || prev?.memorySize || 0,
866
- diskSize: rdd['Disk Size'] || prev?.diskSize || 0,
1003
+ numCachedPartitions: fromBlocks ? prev.numCachedPartitions : rdd['Number of Cached Partitions'] || prev?.numCachedPartitions || 0,
1004
+ memorySize: fromBlocks ? prev.memorySize : rdd['Memory Size'] || prev?.memorySize || 0,
1005
+ diskSize: fromBlocks ? prev.diskSize : rdd['Disk Size'] || prev?.diskSize || 0,
1006
+ storageSource: prev?.storageSource ?? 'rddInfo',
867
1007
  stageIds,
868
1008
  });
869
1009
  }
870
1010
  }
871
1011
 
1012
+ const RDD_BLOCK_ID = /^rdd_(\d+)_(\d+)$/;
1013
+
1014
+ function dropRddBlock(rdd , key ) {
1015
+ const prev = rdd.blocks.get(key);
1016
+ if (!prev) return;
1017
+ rdd.memorySize -= prev.memorySize;
1018
+ rdd.diskSize -= prev.diskSize;
1019
+ const replicas = (rdd.replicasByPartition.get(prev.partition) ?? 1) - 1;
1020
+ if (replicas > 0) rdd.replicasByPartition.set(prev.partition, replicas);
1021
+ else rdd.replicasByPartition.delete(prev.partition);
1022
+ rdd.blocks.delete(key);
1023
+ }
1024
+
1025
+ /**
1026
+ * Folds one SparkListenerBlockUpdated into its RDD's live residency, then publishes the RDD's
1027
+ * peak state to rddInfo: the most partitions resident at once, with the memory/disk bytes at the
1028
+ * latest moment that peak held. A peak rather than the final state, because an unpersist() (or
1029
+ * the app's own cleanup) removes every block before the log ends, and a final snapshot would
1030
+ * read as "nothing was cached". Ties refresh, so a partition dropping from memory to disk after
1031
+ * the peak still shows up in diskSize. Non-RDD blocks (broadcast, shuffle, task results) are ignored.
1032
+ */
1033
+ export function recordBlockUpdate(event , state ) {
1034
+ const info = event['Block Updated Info'];
1035
+ const match = RDD_BLOCK_ID.exec(info['Block ID']);
1036
+ if (!match) return null;
1037
+ state.rddBlockUpdates++;
1038
+ const rddId = Number(match[1]);
1039
+ const partition = Number(match[2]);
1040
+ const sl = info['Storage Level'] ?? {};
1041
+ // Spark's StorageLevel.isValid: a removal or eviction reports level NONE.
1042
+ const resident = Boolean(sl['Use Memory'] || sl['Use Disk']) && (sl['Replication'] ?? 1) > 0;
1043
+
1044
+ let rdd = state.rddBlocks.get(rddId);
1045
+ if (!rdd) {
1046
+ rdd = { blocks: new Map(), replicasByPartition: new Map(), memorySize: 0, diskSize: 0, peakCachedPartitions: 0 };
1047
+ state.rddBlocks.set(rddId, rdd);
1048
+ }
1049
+ const key = `${partition}@${info['Block Manager ID']?.['Executor ID'] ?? ''}`;
1050
+ dropRddBlock(rdd, key);
1051
+ if (resident) {
1052
+ // Sizes count only where the level says the block lives, as Spark's AppStatusListener does: a
1053
+ // drop from memory to disk reports Use Memory false but still carries the dropped bytes as
1054
+ // Memory Size (BlockManager reports max(memSize, droppedMemorySize)).
1055
+ const block = {
1056
+ partition,
1057
+ memorySize: sl['Use Memory'] ? info['Memory Size'] ?? 0 : 0,
1058
+ diskSize: sl['Use Disk'] ? info['Disk Size'] ?? 0 : 0,
1059
+ };
1060
+ rdd.blocks.set(key, block);
1061
+ rdd.memorySize += block.memorySize;
1062
+ rdd.diskSize += block.diskSize;
1063
+ rdd.replicasByPartition.set(partition, (rdd.replicasByPartition.get(partition) ?? 0) + 1);
1064
+ }
1065
+
1066
+ const record = state.rddInfo.get(rddId) ?? {
1067
+ // A block reported before any stage listed its RDD: name and partition count arrive with
1068
+ // the next StageSubmitted (mergeStageRddInfo keeps the block-derived sizes).
1069
+ id: rddId, name: '', callsite: '',
1070
+ storageLevel: { useDisk: Boolean(sl['Use Disk']), useMemory: Boolean(sl['Use Memory']), deserialized: false, replication: sl['Replication'] ?? 1 },
1071
+ numPartitions: 0, numCachedPartitions: 0, memorySize: 0, diskSize: 0,
1072
+ storageSource: 'blockUpdates' , stageIds: new Set (),
1073
+ };
1074
+ record.storageSource = 'blockUpdates';
1075
+ if (rdd.replicasByPartition.size >= rdd.peakCachedPartitions) {
1076
+ rdd.peakCachedPartitions = rdd.replicasByPartition.size;
1077
+ record.numCachedPartitions = rdd.peakCachedPartitions;
1078
+ record.memorySize = rdd.memorySize;
1079
+ record.diskSize = rdd.diskSize;
1080
+ }
1081
+ state.rddInfo.set(rddId, record);
1082
+ return null;
1083
+ }
1084
+
872
1085
  // Spark logs these AFTER SparkListenerStageCompleted, so finalizeStage has already posted the
873
1086
  // stage with an empty executorMetrics Map. Keep accumulating worker-side, then re-post all maps
874
1087
  // once via `stageExecutorMetrics` just before `done` so main-thread stages are patched before analyze().
@@ -982,6 +1195,14 @@ export function removeExecutor(event
982
1195
  reason: event['Removed Reason'] ?? '',
983
1196
  };
984
1197
  state.executors.removed.push(ev);
1198
+ // Blocks lost with their executor get no BlockUpdated; drop them as Spark's AppStatusListener
1199
+ // does, so a partition re-cached elsewhere isn't counted twice.
1200
+ const suffix = `@${ev.executorId}`;
1201
+ for (const rdd of state.rddBlocks.values()) {
1202
+ for (const key of rdd.blocks.keys()) {
1203
+ if (key.endsWith(suffix)) dropRddBlock(rdd, key);
1204
+ }
1205
+ }
985
1206
  return { type: 'executor', data: ev };
986
1207
  }
987
1208
 
@@ -1025,6 +1246,8 @@ export function processEvent(event , state ) {
1025
1246
  // every stage-duration figure becomes the epoch timestamp itself (a "47-year" stage).
1026
1247
  if (!stage.submittedAt && info['Submission Time'] != null) stage.submittedAt = info['Submission Time'];
1027
1248
  stage.stageFailureReason = info['Failure Reason'] ?? null;
1249
+ // A duplicate StageCompleted of an already-finalized attempt is not another failed attempt.
1250
+ if (stage.stageFailureReason != null && stage.taskAttempts !== null) stage.failedStageAttempts++;
1028
1251
  // finalizeStage keeps its `stage` parameter typed as a loose Record (see that module); bridge
1029
1252
  // StageRecord's more precise shape across that boundary with an explicit cast.
1030
1253
  return finalizeStage(
@@ -1058,6 +1281,9 @@ export function processEvent(event , state ) {
1058
1281
  case 'SparkListenerExecutorRemoved':
1059
1282
  return removeExecutor(event, state);
1060
1283
 
1284
+ case 'SparkListenerBlockUpdated':
1285
+ return recordBlockUpdate(event, state);
1286
+
1061
1287
  default:
1062
1288
  return assertNever(event);
1063
1289
  }
@@ -1215,12 +1441,32 @@ export function dispatchLine(
1215
1441
  parseAndDispatch(line, state, emit);
1216
1442
  }
1217
1443
 
1444
+ // With logBlockUpdates on, most BlockUpdated lines are broadcast/shuffle blocks recordBlockUpdate
1445
+ // ignores. Spark writes "Event" first and "Block ID" as a plain string, so those lines can be
1446
+ // dropped on a substring test before paying for JSON.parse.
1447
+ const BLOCK_UPDATED_PREFIX = '{"Event":"SparkListenerBlockUpdated",';
1448
+ const RDD_BLOCK_ID_FRAGMENT = '"Block ID":"rdd_';
1449
+
1450
+ const SQL_START_EVENT = 'org.apache.spark.sql.execution.ui.SparkListenerSQLExecutionStart';
1451
+ const SQL_START_ID = /"executionId":(\d+)/;
1452
+
1453
+ // Records the execution id of a skipped SQL start line, when the line's head still names it: a
1454
+ // line cut off mid-plan fails JSON.parse but keeps its leading fields.
1455
+ function noteUnreadableSqlStart(line , state ) {
1456
+ const head = line.slice(0, 300);
1457
+ if (!head.includes(`${SQL_START_EVENT}"`)) return;
1458
+ const id = SQL_START_ID.exec(head);
1459
+ if (id) state.unreadableSqlStarts.add(Number(id[1]));
1460
+ }
1461
+
1218
1462
  function parseAndDispatch(line , state , emit ) {
1463
+ if (line.startsWith(BLOCK_UPDATED_PREFIX) && !line.includes(RDD_BLOCK_ID_FRAGMENT)) return;
1219
1464
  let parsed ;
1220
1465
  try {
1221
1466
  parsed = parseTaskEnd(line) ?? JSON.parse(stripPlanDescription(line));
1222
1467
  } catch {
1223
1468
  state.skippedLines++;
1469
+ noteUnreadableSqlStart(line, state);
1224
1470
  return;
1225
1471
  }
1226
1472
  // A real Spark event log carries many event types this tool never modeled (BlockManagerAdded,
@@ -1233,6 +1479,7 @@ function parseAndDispatch(line , state , emit
1233
1479
  const result = SparkEventSchema.safeParse(parsed);
1234
1480
  if (!result.success) {
1235
1481
  state.skippedLines++;
1482
+ if (eventType === SQL_START_EVENT) noteUnreadableSqlStart(line, state);
1236
1483
  return;
1237
1484
  }
1238
1485
  if (result.data.Event === 'org.apache.spark.sql.execution.ui.SparkListenerSQLExecutionEnd') {
@@ -1265,13 +1512,43 @@ export function collectStageExecutorMetrics(state )
1265
1512
  return out;
1266
1513
  }
1267
1514
 
1515
+ // Speculation totals of every stage a late TaskEnd added waste to (accountLateSpeculativeLoser),
1516
+ // re-posted once before `done`: the stage message posted at completion carried the earlier totals.
1517
+ export function collectLateSpeculationWaste(
1518
+ state ,
1519
+ ) {
1520
+ const out = new Map ();
1521
+ for (const [id, stage] of state.stages) {
1522
+ if (stage.lateSpeculationWaste) {
1523
+ out.set(id, { speculationWasteMs: stage.speculationWasteMs, speculationWastedAttempts: stage.speculationWastedAttempts });
1524
+ }
1525
+ }
1526
+ return out;
1527
+ }
1528
+
1529
+ // Late work of every stage a failed attempt's late TaskEnd added to, re-posted once before `done`:
1530
+ // the stage message posted at completion predates it.
1531
+ export function collectLateAttemptWork(state ) {
1532
+ const out = new Map ();
1533
+ for (const [id, stage] of state.stages) {
1534
+ if (stage.lateAttemptWork != null) out.set(id, stage.lateAttemptWork);
1535
+ }
1536
+ return out;
1537
+ }
1538
+
1268
1539
  export function emitParseCompletion(state , emit , linesProcessed ) {
1269
1540
  // Executions that never ended keep their latest AQE update, as they did before it was deferred.
1270
1541
  for (const executionId of [...state.pendingAdaptiveUpdates.keys()]) flushAdaptiveUpdate(executionId, state, emit);
1271
1542
  emit({ type: 'progress', pct: 1, linesProcessed });
1543
+ emit({ type: 'stageLateAttemptWork', data: collectLateAttemptWork(state) });
1272
1544
  emit({ type: 'runAggregates', data: computeRunAggregates(state.taskStore) });
1545
+ emit({ type: 'stageSpeculationWaste', data: collectLateSpeculationWaste(state) });
1273
1546
  emit({ type: 'stageExecutorMetrics', data: collectStageExecutorMetrics(state) });
1274
1547
  emit(appMessage(state));
1275
- emit({ type: 'done', skippedLines: state.skippedLines });
1548
+ emit({
1549
+ type: 'done',
1550
+ skippedLines: state.skippedLines,
1551
+ ...(state.unreadableSqlStarts.size > 0 ? { unreadableSqlExecutions: [...state.unreadableSqlStarts].sort((a, b) => a - b) } : {}),
1552
+ });
1276
1553
  state.accumState.clear();
1277
1554
  }
@@ -201,6 +201,7 @@ export const StageSubmittedEventSchema = z.object({
201
201
  Event: z.literal('SparkListenerStageSubmitted'),
202
202
  'Stage Info': z.object({
203
203
  'Stage ID': z.number(),
204
+ 'Stage Attempt ID': z.number().optional(),
204
205
  'Stage Name': z.string().optional(),
205
206
  Details: z.string().optional(),
206
207
  'Submission Time': z.number().optional(),
@@ -209,6 +210,26 @@ export const StageSubmittedEventSchema = z.object({
209
210
  }),
210
211
  });
211
212
 
213
+ // recordBlockUpdate. Written only with spark.eventLog.logBlockUpdates.enabled=true; one event per
214
+ // block status change reported to the driver's BlockManagerMaster (a removal or eviction carries
215
+ // storage level NONE with zero sizes). Only 'rdd_<rddId>_<partition>' blocks are read.
216
+ export const BlockUpdatedEventSchema = z.object({
217
+ Event: z.literal('SparkListenerBlockUpdated'),
218
+ 'Block Updated Info': z.object({
219
+ 'Block Manager ID': z.object({
220
+ 'Executor ID': z.string().optional(),
221
+ }).optional(),
222
+ 'Block ID': z.string(),
223
+ 'Storage Level': z.object({
224
+ 'Use Disk': z.boolean().optional(),
225
+ 'Use Memory': z.boolean().optional(),
226
+ Replication: z.number().optional(),
227
+ }).optional(),
228
+ 'Memory Size': z.number().optional(),
229
+ 'Disk Size': z.number().optional(),
230
+ }),
231
+ });
232
+
212
233
  // inline StageCompleted case.
213
234
  export const StageCompletedEventSchema = z.object({
214
235
  Event: z.literal('SparkListenerStageCompleted'),
@@ -291,6 +312,7 @@ export const TaskEndEventSchema = z.object({
291
312
  }).optional(),
292
313
  'Output Metrics': z.object({
293
314
  'Bytes Written': z.number().optional(),
315
+ 'Records Written': z.number().optional(),
294
316
  }).optional(),
295
317
  }).optional(),
296
318
  });
@@ -345,5 +367,6 @@ export const SparkEventSchema = z.discriminatedUnion('Event', [
345
367
  DriverAccumUpdatesEventSchema,
346
368
  ExecutorAddedEventSchema,
347
369
  ExecutorRemovedEventSchema,
370
+ BlockUpdatedEventSchema,
348
371
  ]);
349
372