sparkforensics-mcp 0.2.4 → 0.4.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -0
- package/README.md +7 -1
- package/bin/sparkforensics-mcp.mjs +41 -10
- package/package.json +1 -1
- package/vendor-core/analyzer.js +156 -48
- package/vendor-core/check-coverage.js +88 -0
- package/vendor-core/cli/budgets.js +31 -18
- package/vendor-core/cli/collect-run.js +76 -31
- package/vendor-core/cli/native-zstd.js +2 -2
- package/vendor-core/cli/threshold-config.js +28 -0
- package/vendor-core/comparison-verdict.js +177 -0
- package/vendor-core/core-source-hash.txt +1 -0
- package/vendor-core/core-usage-locality.js +56 -2
- package/vendor-core/detector-docs.js +58 -0
- package/vendor-core/detectors.js +933 -459
- package/vendor-core/docs-config.js +0 -36
- package/vendor-core/docs-content/chapters/nav-index.json +31 -0
- package/vendor-core/docs-content/detection/cstor.md +9 -0
- package/vendor-core/docs-content/detection/fail.md +6 -2
- package/vendor-core/docs-site-config.js +3 -0
- package/vendor-core/event-handlers.js +191 -6
- package/vendor-core/event-schemas.js +29 -0
- package/vendor-core/evidence-report.js +440 -112
- package/vendor-core/export-data.js +79 -6
- package/vendor-core/finding-action-label.js +9 -88
- package/vendor-core/finding-filter-predicate.js +9 -0
- package/vendor-core/finding-generic-recommendation.js +6 -104
- package/vendor-core/finding-names.js +21 -45
- package/vendor-core/finding-presentation.js +333 -0
- package/vendor-core/finding-tag-help.js +110 -0
- package/vendor-core/finding-types.js +361 -0
- package/vendor-core/findings-of-type.js +11 -0
- package/vendor-core/format-utils.js +92 -27
- package/vendor-core/html-export.js +51 -0
- package/vendor-core/impact-band.js +21 -8
- package/vendor-core/impact-estimator.js +8 -521
- package/vendor-core/impact-format.js +114 -0
- package/vendor-core/impact-model.js +175 -0
- package/vendor-core/ingest.js +2 -0
- package/vendor-core/intervals.js +13 -0
- package/vendor-core/list-runs.js +2 -3
- package/vendor-core/load-vendored.js +70 -5
- package/vendor-core/mcp-server-factory.js +14 -10
- package/vendor-core/mcp-tools.js +105 -45
- package/vendor-core/model-assembler.js +12 -0
- package/vendor-core/occupancy.js +1 -1
- package/vendor-core/parser-worker.js +22 -5
- package/vendor-core/plan-graph-model.js +3 -2
- package/vendor-core/plan-node-detail.js +1 -1
- package/vendor-core/recommendation-rollup.js +63 -3
- package/vendor-core/redact.js +68 -28
- package/vendor-core/run-comparison.js +40 -7
- package/vendor-core/run-interpretation.js +290 -0
- package/vendor-core/run-outcome.js +74 -0
- package/vendor-core/run-payload.js +17 -0
- package/vendor-core/run-shape.js +40 -0
- package/vendor-core/run-verdict.js +353 -0
- package/vendor-core/scaling-sim.js +4 -5
- package/vendor-core/scorecard-estimates.js +62 -0
- package/vendor-core/shs-fetch.js +175 -65
- package/vendor-core/shs-load.js +1 -1
- package/vendor-core/sql-stages.js +11 -0
- package/vendor-core/stage-quantiles.js +14 -0
- package/vendor-core/task-failure.js +151 -0
- package/vendor-core/threshold-overrides.js +160 -0
- package/vendor-core/threshold-summary.js +11 -33
- package/vendor-core/types.js +6 -42
- package/vendor-core/vendor/fflate.js +1 -1
- package/vendor-core/wall-clock.js +1 -12
- package/vendor-core/wasted-core-hours.js +2 -2
- package/vendor-core/zip-archive.js +167 -0
|
@@ -14,11 +14,13 @@ import {
|
|
|
14
14
|
DriverAccumUpdatesEventSchema,
|
|
15
15
|
ExecutorAddedEventSchema,
|
|
16
16
|
ExecutorRemovedEventSchema,
|
|
17
|
+
BlockUpdatedEventSchema,
|
|
17
18
|
|
|
18
19
|
|
|
19
20
|
} from './event-schemas.js';
|
|
20
21
|
import { assertNever } from './assert-never.js';
|
|
21
22
|
import { finalizeStage } from './stage-quantiles.js';
|
|
23
|
+
import { MAX_FAILURE_DETAILS_PER_STAGE, extractTaskFailureDetail, taskFailureKey, } from './task-failure.js';
|
|
22
24
|
import { computeRunAggregates } from './run-aggregates.js';
|
|
23
25
|
|
|
24
26
|
|
|
@@ -61,7 +63,22 @@ import { computeRunAggregates } from './run-aggregates.js';
|
|
|
61
63
|
|
|
62
64
|
|
|
63
65
|
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
|
|
64
70
|
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
// Live per-RDD block residency rebuilt from SparkListenerBlockUpdated. Keyed by partition and
|
|
74
|
+
// executor because a block's status is per BlockManager: replicas and re-caches on another
|
|
75
|
+
// executor are separate entries, as in Spark's own AppStatusListener.
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
|
|
81
|
+
|
|
65
82
|
|
|
66
83
|
|
|
67
84
|
|
|
@@ -94,6 +111,9 @@ const MAX_TASK_SAMPLES = 20;
|
|
|
94
111
|
|
|
95
112
|
|
|
96
113
|
|
|
114
|
+
|
|
115
|
+
|
|
116
|
+
|
|
97
117
|
|
|
98
118
|
|
|
99
119
|
|
|
@@ -136,11 +156,19 @@ const MAX_TASK_SAMPLES = 20;
|
|
|
136
156
|
|
|
137
157
|
|
|
138
158
|
|
|
159
|
+
|
|
160
|
+
|
|
139
161
|
|
|
140
162
|
|
|
141
163
|
|
|
142
164
|
|
|
143
165
|
|
|
166
|
+
|
|
167
|
+
|
|
168
|
+
|
|
169
|
+
|
|
170
|
+
|
|
171
|
+
|
|
144
172
|
|
|
145
173
|
|
|
146
174
|
|
|
@@ -172,6 +200,10 @@ const MAX_TASK_SAMPLES = 20;
|
|
|
172
200
|
|
|
173
201
|
|
|
174
202
|
|
|
203
|
+
|
|
204
|
+
|
|
205
|
+
|
|
206
|
+
|
|
175
207
|
|
|
176
208
|
|
|
177
209
|
|
|
@@ -361,6 +393,8 @@ export function createState() {
|
|
|
361
393
|
skippedLines: 0,
|
|
362
394
|
accumState: new Map(),
|
|
363
395
|
rddInfo: new Map(),
|
|
396
|
+
rddBlocks: new Map(),
|
|
397
|
+
rddBlockUpdates: 0,
|
|
364
398
|
taskAccumStages: new Map(),
|
|
365
399
|
pendingAdaptiveUpdates: new Map(),
|
|
366
400
|
resolvedPlanExecutions: new Set(),
|
|
@@ -472,6 +506,7 @@ function snapshotEvidenceInputs(state ) {
|
|
|
472
506
|
function appMessage(state ) {
|
|
473
507
|
const evidenceInputs = snapshotEvidenceInputs(state);
|
|
474
508
|
state.app .evidenceInputs = evidenceInputs;
|
|
509
|
+
state.app .rddBlockUpdates = state.rddBlockUpdates;
|
|
475
510
|
return {
|
|
476
511
|
type: 'app',
|
|
477
512
|
data: { ...state.app , rddInfo: snapshotRddInfo(state.rddInfo) },
|
|
@@ -485,13 +520,30 @@ function taskRecordToSample(t ) {
|
|
|
485
520
|
};
|
|
486
521
|
}
|
|
487
522
|
|
|
523
|
+
// One shared detail object per distinct failure, so a stage with thousands of failed attempts
|
|
524
|
+
// holds each bounded stack excerpt once. Past the cap an attempt keeps only its reason tag.
|
|
525
|
+
function internTaskFailure(stage , endReason ) {
|
|
526
|
+
const detail = extractTaskFailureDetail(endReason);
|
|
527
|
+
if (!detail || !stage.failureDetails) return null;
|
|
528
|
+
const key = taskFailureKey(detail);
|
|
529
|
+
const known = stage.failureDetails.get(key);
|
|
530
|
+
if (known) return known;
|
|
531
|
+
if (stage.failureDetails.size >= MAX_FAILURE_DETAILS_PER_STAGE) return null;
|
|
532
|
+
stage.failureDetails.set(key, detail);
|
|
533
|
+
return detail;
|
|
534
|
+
}
|
|
535
|
+
|
|
488
536
|
export function accumulateTask(event , state ) {
|
|
489
537
|
const stageId = event['Stage ID'];
|
|
490
538
|
const stage = state.stages.get(stageId);
|
|
491
539
|
if (!stage) return null;
|
|
492
540
|
// Late TaskEnd for a stage whose StageCompleted already freed taskAttempts (finalizeStage): its
|
|
493
|
-
// stats are already baked into the finalized stage, don't re-add.
|
|
494
|
-
|
|
541
|
+
// stats are already baked into the finalized stage, don't re-add. The one exception is a losing
|
|
542
|
+
// speculative attempt, whose wasted time the finalized stage never saw.
|
|
543
|
+
if (stage.taskAttempts === null) {
|
|
544
|
+
accountLateSpeculativeLoser(event, stage);
|
|
545
|
+
return null;
|
|
546
|
+
}
|
|
495
547
|
|
|
496
548
|
state.evidenceInputs.taskRecords++;
|
|
497
549
|
|
|
@@ -521,6 +573,7 @@ export function accumulateTask(event , state
|
|
|
521
573
|
launchTime: info['Launch Time'] ?? 0,
|
|
522
574
|
finishTime: info['Finish Time'] ?? 0,
|
|
523
575
|
reason: event['Task End Reason']?.['Reason'] ?? null,
|
|
576
|
+
failure: failed ? internTaskFailure(stage, event['Task End Reason']) : null,
|
|
524
577
|
speculative: info['Speculative'] === true,
|
|
525
578
|
host: info['Host'] ?? '',
|
|
526
579
|
executorId: info['Executor ID'] ?? '',
|
|
@@ -546,6 +599,7 @@ export function accumulateTask(event , state
|
|
|
546
599
|
|
|
547
600
|
if (!existing) {
|
|
548
601
|
stage.taskAttempts.set(key, record);
|
|
602
|
+
if (record.speculative && !record.failed) stage.speculativeWinners.add(key);
|
|
549
603
|
} else if (existing.failed && !record.failed) {
|
|
550
604
|
// A retry succeeded where the earlier attempt failed: the earlier attempt's time was wasted.
|
|
551
605
|
// Spark marks only the speculative COPY's Speculative flag, never the original it raced, so
|
|
@@ -561,6 +615,7 @@ export function accumulateTask(event , state
|
|
|
561
615
|
}
|
|
562
616
|
}
|
|
563
617
|
stage.taskAttempts.set(key, record);
|
|
618
|
+
if (record.speculative) stage.speculativeWinners.add(key);
|
|
564
619
|
} else {
|
|
565
620
|
// Non-winning duplicate (both failed, or a race where a winner is
|
|
566
621
|
// already recorded): its time is waste, its metrics are discarded.
|
|
@@ -579,6 +634,21 @@ export function accumulateTask(event , state
|
|
|
579
634
|
return null;
|
|
580
635
|
}
|
|
581
636
|
|
|
637
|
+
// Spark kills the losing copy of a speculative race only once the stage finishes ("Stage
|
|
638
|
+
// cancelled: Stage finished"), so that loser's TaskEnd normally lands after StageCompleted. Count
|
|
639
|
+
// its time as speculation waste, pairing it the same way accumulateTask does: the late attempt is
|
|
640
|
+
// the speculative copy itself, or the original that a speculative winner beat. Every other stat
|
|
641
|
+
// of a late attempt stays excluded, as the finalized stage already posted them.
|
|
642
|
+
function accountLateSpeculativeLoser(event , stage ) {
|
|
643
|
+
const info = event['Task Info'];
|
|
644
|
+
if (info?.['Index'] == null) return;
|
|
645
|
+
const key = `${event['Stage Attempt ID'] ?? 0}:${info['Index']}`;
|
|
646
|
+
if (info['Speculative'] !== true && !stage.speculativeWinners.has(key)) return;
|
|
647
|
+
stage.speculationWasteMs += (info['Finish Time'] ?? 0) - (info['Launch Time'] ?? 0);
|
|
648
|
+
stage.speculationWastedAttempts++;
|
|
649
|
+
stage.lateSpeculationWaste = true;
|
|
650
|
+
}
|
|
651
|
+
|
|
582
652
|
export function resolvePlanTree(
|
|
583
653
|
rootInfo ,
|
|
584
654
|
accumMap ,
|
|
@@ -802,11 +872,14 @@ export function submitStage(event , st
|
|
|
802
872
|
failureReasons: new Map(),
|
|
803
873
|
stageFailureReason: null,
|
|
804
874
|
taskAttempts: new Map(),
|
|
875
|
+
failureDetails: new Map(),
|
|
805
876
|
retryTaskSamples: [],
|
|
806
877
|
retryWasteMs: 0,
|
|
807
878
|
wastedAttempts: 0,
|
|
808
879
|
speculationWasteMs: 0,
|
|
809
880
|
speculationWastedAttempts: 0,
|
|
881
|
+
speculativeWinners: new Set(),
|
|
882
|
+
lateSpeculationWaste: false,
|
|
810
883
|
executorMetrics: new Map(),
|
|
811
884
|
});
|
|
812
885
|
mergeStageRddInfo(info, id, state);
|
|
@@ -826,11 +899,16 @@ export function mergeStageRddInfo(
|
|
|
826
899
|
const prev = state.rddInfo.get(rddId);
|
|
827
900
|
const stageIds = prev?.stageIds ?? new Set ();
|
|
828
901
|
stageIds.add(id);
|
|
902
|
+
// Block updates are the authoritative source once seen: never let a later RDD Info snapshot
|
|
903
|
+
// (0 on Spark 2.3+) replace them, nor the NONE level an unpersist() leaves on ancestor RDDs
|
|
904
|
+
// listed by later stages.
|
|
905
|
+
const fromBlocks = prev?.storageSource === 'blockUpdates';
|
|
906
|
+
const persisted = Boolean(sl['Use Disk'] || sl['Use Memory']);
|
|
829
907
|
state.rddInfo.set(rddId, {
|
|
830
908
|
id: rddId,
|
|
831
909
|
name: rdd['Name'] ?? '',
|
|
832
910
|
callsite: rdd['Callsite'] ?? '',
|
|
833
|
-
storageLevel: {
|
|
911
|
+
storageLevel: fromBlocks && !persisted ? prev.storageLevel : {
|
|
834
912
|
useDisk: sl['Use Disk'] ?? false,
|
|
835
913
|
useMemory: sl['Use Memory'] ?? false,
|
|
836
914
|
deserialized: sl['Deserialized'] ?? false,
|
|
@@ -840,14 +918,88 @@ export function mergeStageRddInfo(
|
|
|
840
918
|
// Merge forward, don't overwrite: an RDD cached for the first time in THIS stage legitimately
|
|
841
919
|
// reports 0 (snapshot reflects BlockManager state at submission). Keep the last real value on
|
|
842
920
|
// a resubmission instead of regressing to 0.
|
|
843
|
-
numCachedPartitions: rdd['Number of Cached Partitions'] || prev?.numCachedPartitions || 0,
|
|
844
|
-
memorySize: rdd['Memory Size'] || prev?.memorySize || 0,
|
|
845
|
-
diskSize: rdd['Disk Size'] || prev?.diskSize || 0,
|
|
921
|
+
numCachedPartitions: fromBlocks ? prev.numCachedPartitions : rdd['Number of Cached Partitions'] || prev?.numCachedPartitions || 0,
|
|
922
|
+
memorySize: fromBlocks ? prev.memorySize : rdd['Memory Size'] || prev?.memorySize || 0,
|
|
923
|
+
diskSize: fromBlocks ? prev.diskSize : rdd['Disk Size'] || prev?.diskSize || 0,
|
|
924
|
+
storageSource: prev?.storageSource ?? 'rddInfo',
|
|
846
925
|
stageIds,
|
|
847
926
|
});
|
|
848
927
|
}
|
|
849
928
|
}
|
|
850
929
|
|
|
930
|
+
const RDD_BLOCK_ID = /^rdd_(\d+)_(\d+)$/;
|
|
931
|
+
|
|
932
|
+
function dropRddBlock(rdd , key ) {
|
|
933
|
+
const prev = rdd.blocks.get(key);
|
|
934
|
+
if (!prev) return;
|
|
935
|
+
rdd.memorySize -= prev.memorySize;
|
|
936
|
+
rdd.diskSize -= prev.diskSize;
|
|
937
|
+
const replicas = (rdd.replicasByPartition.get(prev.partition) ?? 1) - 1;
|
|
938
|
+
if (replicas > 0) rdd.replicasByPartition.set(prev.partition, replicas);
|
|
939
|
+
else rdd.replicasByPartition.delete(prev.partition);
|
|
940
|
+
rdd.blocks.delete(key);
|
|
941
|
+
}
|
|
942
|
+
|
|
943
|
+
/**
|
|
944
|
+
* Folds one SparkListenerBlockUpdated into its RDD's live residency, then publishes the RDD's
|
|
945
|
+
* peak state to rddInfo: the most partitions resident at once, with the memory/disk bytes at the
|
|
946
|
+
* latest moment that peak held. A peak rather than the final state, because an unpersist() (or
|
|
947
|
+
* the app's own cleanup) removes every block before the log ends, and a final snapshot would
|
|
948
|
+
* read as "nothing was cached". Ties refresh, so a partition dropping from memory to disk after
|
|
949
|
+
* the peak still shows up in diskSize. Non-RDD blocks (broadcast, shuffle, task results) are ignored.
|
|
950
|
+
*/
|
|
951
|
+
export function recordBlockUpdate(event , state ) {
|
|
952
|
+
const info = event['Block Updated Info'];
|
|
953
|
+
const match = RDD_BLOCK_ID.exec(info['Block ID']);
|
|
954
|
+
if (!match) return null;
|
|
955
|
+
state.rddBlockUpdates++;
|
|
956
|
+
const rddId = Number(match[1]);
|
|
957
|
+
const partition = Number(match[2]);
|
|
958
|
+
const sl = info['Storage Level'] ?? {};
|
|
959
|
+
// Spark's StorageLevel.isValid: a removal or eviction reports level NONE.
|
|
960
|
+
const resident = Boolean(sl['Use Memory'] || sl['Use Disk']) && (sl['Replication'] ?? 1) > 0;
|
|
961
|
+
|
|
962
|
+
let rdd = state.rddBlocks.get(rddId);
|
|
963
|
+
if (!rdd) {
|
|
964
|
+
rdd = { blocks: new Map(), replicasByPartition: new Map(), memorySize: 0, diskSize: 0, peakCachedPartitions: 0 };
|
|
965
|
+
state.rddBlocks.set(rddId, rdd);
|
|
966
|
+
}
|
|
967
|
+
const key = `${partition}@${info['Block Manager ID']?.['Executor ID'] ?? ''}`;
|
|
968
|
+
dropRddBlock(rdd, key);
|
|
969
|
+
if (resident) {
|
|
970
|
+
// Sizes count only where the level says the block lives, as Spark's AppStatusListener does: a
|
|
971
|
+
// drop from memory to disk reports Use Memory false but still carries the dropped bytes as
|
|
972
|
+
// Memory Size (BlockManager reports max(memSize, droppedMemorySize)).
|
|
973
|
+
const block = {
|
|
974
|
+
partition,
|
|
975
|
+
memorySize: sl['Use Memory'] ? info['Memory Size'] ?? 0 : 0,
|
|
976
|
+
diskSize: sl['Use Disk'] ? info['Disk Size'] ?? 0 : 0,
|
|
977
|
+
};
|
|
978
|
+
rdd.blocks.set(key, block);
|
|
979
|
+
rdd.memorySize += block.memorySize;
|
|
980
|
+
rdd.diskSize += block.diskSize;
|
|
981
|
+
rdd.replicasByPartition.set(partition, (rdd.replicasByPartition.get(partition) ?? 0) + 1);
|
|
982
|
+
}
|
|
983
|
+
|
|
984
|
+
const record = state.rddInfo.get(rddId) ?? {
|
|
985
|
+
// A block reported before any stage listed its RDD: name and partition count arrive with
|
|
986
|
+
// the next StageSubmitted (mergeStageRddInfo keeps the block-derived sizes).
|
|
987
|
+
id: rddId, name: '', callsite: '',
|
|
988
|
+
storageLevel: { useDisk: Boolean(sl['Use Disk']), useMemory: Boolean(sl['Use Memory']), deserialized: false, replication: sl['Replication'] ?? 1 },
|
|
989
|
+
numPartitions: 0, numCachedPartitions: 0, memorySize: 0, diskSize: 0,
|
|
990
|
+
storageSource: 'blockUpdates' , stageIds: new Set (),
|
|
991
|
+
};
|
|
992
|
+
record.storageSource = 'blockUpdates';
|
|
993
|
+
if (rdd.replicasByPartition.size >= rdd.peakCachedPartitions) {
|
|
994
|
+
rdd.peakCachedPartitions = rdd.replicasByPartition.size;
|
|
995
|
+
record.numCachedPartitions = rdd.peakCachedPartitions;
|
|
996
|
+
record.memorySize = rdd.memorySize;
|
|
997
|
+
record.diskSize = rdd.diskSize;
|
|
998
|
+
}
|
|
999
|
+
state.rddInfo.set(rddId, record);
|
|
1000
|
+
return null;
|
|
1001
|
+
}
|
|
1002
|
+
|
|
851
1003
|
// Spark logs these AFTER SparkListenerStageCompleted, so finalizeStage has already posted the
|
|
852
1004
|
// stage with an empty executorMetrics Map. Keep accumulating worker-side, then re-post all maps
|
|
853
1005
|
// once via `stageExecutorMetrics` just before `done` so main-thread stages are patched before analyze().
|
|
@@ -961,6 +1113,14 @@ export function removeExecutor(event
|
|
|
961
1113
|
reason: event['Removed Reason'] ?? '',
|
|
962
1114
|
};
|
|
963
1115
|
state.executors.removed.push(ev);
|
|
1116
|
+
// Blocks lost with their executor get no BlockUpdated; drop them as Spark's AppStatusListener
|
|
1117
|
+
// does, so a partition re-cached elsewhere isn't counted twice.
|
|
1118
|
+
const suffix = `@${ev.executorId}`;
|
|
1119
|
+
for (const rdd of state.rddBlocks.values()) {
|
|
1120
|
+
for (const key of rdd.blocks.keys()) {
|
|
1121
|
+
if (key.endsWith(suffix)) dropRddBlock(rdd, key);
|
|
1122
|
+
}
|
|
1123
|
+
}
|
|
964
1124
|
return { type: 'executor', data: ev };
|
|
965
1125
|
}
|
|
966
1126
|
|
|
@@ -1037,6 +1197,9 @@ export function processEvent(event , state ) {
|
|
|
1037
1197
|
case 'SparkListenerExecutorRemoved':
|
|
1038
1198
|
return removeExecutor(event, state);
|
|
1039
1199
|
|
|
1200
|
+
case 'SparkListenerBlockUpdated':
|
|
1201
|
+
return recordBlockUpdate(event, state);
|
|
1202
|
+
|
|
1040
1203
|
default:
|
|
1041
1204
|
return assertNever(event);
|
|
1042
1205
|
}
|
|
@@ -1194,7 +1357,14 @@ export function dispatchLine(
|
|
|
1194
1357
|
parseAndDispatch(line, state, emit);
|
|
1195
1358
|
}
|
|
1196
1359
|
|
|
1360
|
+
// With logBlockUpdates on, most BlockUpdated lines are broadcast/shuffle blocks recordBlockUpdate
|
|
1361
|
+
// ignores. Spark writes "Event" first and "Block ID" as a plain string, so those lines can be
|
|
1362
|
+
// dropped on a substring test before paying for JSON.parse.
|
|
1363
|
+
const BLOCK_UPDATED_PREFIX = '{"Event":"SparkListenerBlockUpdated",';
|
|
1364
|
+
const RDD_BLOCK_ID_FRAGMENT = '"Block ID":"rdd_';
|
|
1365
|
+
|
|
1197
1366
|
function parseAndDispatch(line , state , emit ) {
|
|
1367
|
+
if (line.startsWith(BLOCK_UPDATED_PREFIX) && !line.includes(RDD_BLOCK_ID_FRAGMENT)) return;
|
|
1198
1368
|
let parsed ;
|
|
1199
1369
|
try {
|
|
1200
1370
|
parsed = parseTaskEnd(line) ?? JSON.parse(stripPlanDescription(line));
|
|
@@ -1244,11 +1414,26 @@ export function collectStageExecutorMetrics(state )
|
|
|
1244
1414
|
return out;
|
|
1245
1415
|
}
|
|
1246
1416
|
|
|
1417
|
+
// Speculation totals of every stage a late TaskEnd added waste to (accountLateSpeculativeLoser),
|
|
1418
|
+
// re-posted once before `done`: the stage message posted at completion carried the earlier totals.
|
|
1419
|
+
export function collectLateSpeculationWaste(
|
|
1420
|
+
state ,
|
|
1421
|
+
) {
|
|
1422
|
+
const out = new Map ();
|
|
1423
|
+
for (const [id, stage] of state.stages) {
|
|
1424
|
+
if (stage.lateSpeculationWaste) {
|
|
1425
|
+
out.set(id, { speculationWasteMs: stage.speculationWasteMs, speculationWastedAttempts: stage.speculationWastedAttempts });
|
|
1426
|
+
}
|
|
1427
|
+
}
|
|
1428
|
+
return out;
|
|
1429
|
+
}
|
|
1430
|
+
|
|
1247
1431
|
export function emitParseCompletion(state , emit , linesProcessed ) {
|
|
1248
1432
|
// Executions that never ended keep their latest AQE update, as they did before it was deferred.
|
|
1249
1433
|
for (const executionId of [...state.pendingAdaptiveUpdates.keys()]) flushAdaptiveUpdate(executionId, state, emit);
|
|
1250
1434
|
emit({ type: 'progress', pct: 1, linesProcessed });
|
|
1251
1435
|
emit({ type: 'runAggregates', data: computeRunAggregates(state.taskStore) });
|
|
1436
|
+
emit({ type: 'stageSpeculationWaste', data: collectLateSpeculationWaste(state) });
|
|
1252
1437
|
emit({ type: 'stageExecutorMetrics', data: collectStageExecutorMetrics(state) });
|
|
1253
1438
|
emit(appMessage(state));
|
|
1254
1439
|
emit({ type: 'done', skippedLines: state.skippedLines });
|
|
@@ -209,6 +209,26 @@ export const StageSubmittedEventSchema = z.object({
|
|
|
209
209
|
}),
|
|
210
210
|
});
|
|
211
211
|
|
|
212
|
+
// recordBlockUpdate. Written only with spark.eventLog.logBlockUpdates.enabled=true; one event per
|
|
213
|
+
// block status change reported to the driver's BlockManagerMaster (a removal or eviction carries
|
|
214
|
+
// storage level NONE with zero sizes). Only 'rdd_<rddId>_<partition>' blocks are read.
|
|
215
|
+
export const BlockUpdatedEventSchema = z.object({
|
|
216
|
+
Event: z.literal('SparkListenerBlockUpdated'),
|
|
217
|
+
'Block Updated Info': z.object({
|
|
218
|
+
'Block Manager ID': z.object({
|
|
219
|
+
'Executor ID': z.string().optional(),
|
|
220
|
+
}).optional(),
|
|
221
|
+
'Block ID': z.string(),
|
|
222
|
+
'Storage Level': z.object({
|
|
223
|
+
'Use Disk': z.boolean().optional(),
|
|
224
|
+
'Use Memory': z.boolean().optional(),
|
|
225
|
+
Replication: z.number().optional(),
|
|
226
|
+
}).optional(),
|
|
227
|
+
'Memory Size': z.number().optional(),
|
|
228
|
+
'Disk Size': z.number().optional(),
|
|
229
|
+
}),
|
|
230
|
+
});
|
|
231
|
+
|
|
212
232
|
// inline StageCompleted case.
|
|
213
233
|
export const StageCompletedEventSchema = z.object({
|
|
214
234
|
Event: z.literal('SparkListenerStageCompleted'),
|
|
@@ -238,6 +258,14 @@ export const TaskEndEventSchema = z.object({
|
|
|
238
258
|
'Stage Attempt ID': z.number().optional(),
|
|
239
259
|
'Task End Reason': z.object({
|
|
240
260
|
Reason: z.string().optional(),
|
|
261
|
+
// The error behind a failed attempt, read by extractTaskFailureDetail (task-failure.ts), which
|
|
262
|
+
// keeps only string values: unknown here so an unexpected shape drops the detail, never the task.
|
|
263
|
+
'Class Name': z.unknown().optional(),
|
|
264
|
+
Description: z.unknown().optional(),
|
|
265
|
+
'Full Stack Trace': z.unknown().optional(),
|
|
266
|
+
'Loss Reason': z.unknown().optional(),
|
|
267
|
+
Message: z.unknown().optional(),
|
|
268
|
+
'Kill Reason': z.unknown().optional(),
|
|
241
269
|
}).optional(),
|
|
242
270
|
'Task Info': z.object({
|
|
243
271
|
'Launch Time': z.number().optional(),
|
|
@@ -337,5 +365,6 @@ export const SparkEventSchema = z.discriminatedUnion('Event', [
|
|
|
337
365
|
DriverAccumUpdatesEventSchema,
|
|
338
366
|
ExecutorAddedEventSchema,
|
|
339
367
|
ExecutorRemovedEventSchema,
|
|
368
|
+
BlockUpdatedEventSchema,
|
|
340
369
|
]);
|
|
341
370
|
|