sparkforensics-mcp 0.4.0 → 0.5.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/package.json +1 -1
- package/vendor-core/allocation.js +106 -0
- package/vendor-core/analyzer.js +13 -13
- package/vendor-core/cli/budgets.js +23 -9
- package/vendor-core/cli/collect-run.js +11 -4
- package/vendor-core/cli/regression-budgets.js +83 -0
- package/vendor-core/comparison-verdict.js +22 -22
- package/vendor-core/core-source-hash.txt +1 -1
- package/vendor-core/detectors.js +202 -68
- package/vendor-core/docs-content/detection/cache.md +3 -2
- package/vendor-core/docs-content/detection/cfg.md +9 -8
- package/vendor-core/docs-content/detection/chrn.md +1 -2
- package/vendor-core/docs-content/detection/cold.md +4 -2
- package/vendor-core/docs-content/detection/fail.md +3 -2
- package/vendor-core/docs-content/detection/gc.md +3 -2
- package/vendor-core/docs-content/detection/host.md +2 -1
- package/vendor-core/docs-content/detection/local.md +1 -1
- package/vendor-core/docs-content/detection/mem.md +5 -2
- package/vendor-core/docs-content/detection/plan.md +2 -1
- package/vendor-core/docs-content/detection/sfail.md +2 -1
- package/vendor-core/docs-content/detection/shape.md +5 -4
- package/vendor-core/docs-content/detection/skew.md +3 -1
- package/vendor-core/docs-content/detection/slow.md +2 -2
- package/vendor-core/docs-content/detection/spec.md +2 -3
- package/vendor-core/docs-content/detection/spill.md +1 -1
- package/vendor-core/docs-site-config.js +1 -1
- package/vendor-core/effective-conf.js +107 -0
- package/vendor-core/efficiency-model.js +8 -6
- package/vendor-core/event-handlers.js +160 -47
- package/vendor-core/event-schemas.js +2 -0
- package/vendor-core/evidence-report.js +18 -10
- package/vendor-core/finding-generic-recommendation.js +20 -1
- package/vendor-core/finding-names.js +7 -0
- package/vendor-core/finding-presentation.js +61 -26
- package/vendor-core/finding-tag-help.js +1 -1
- package/vendor-core/finding-types.js +12 -0
- package/vendor-core/format-utils.js +4 -3
- package/vendor-core/impact-estimator.js +20 -2
- package/vendor-core/impact-format.js +14 -13
- package/vendor-core/impact-model.js +27 -5
- package/vendor-core/ingest.js +4 -2
- package/vendor-core/list-runs.js +5 -2
- package/vendor-core/mcp-tools.js +1 -1
- package/vendor-core/model-assembler.js +23 -1
- package/vendor-core/parser-worker.js +1 -1
- package/vendor-core/proxy.js +3 -1
- package/vendor-core/python-stage.js +25 -0
- package/vendor-core/recommendation-rollup.js +16 -9
- package/vendor-core/redact.js +51 -10
- package/vendor-core/remediation.js +20 -0
- package/vendor-core/run-comparison.js +43 -24
- package/vendor-core/run-interpretation.js +2 -1
- package/vendor-core/run-metrics.js +198 -0
- package/vendor-core/run-totals.js +24 -0
- package/vendor-core/run-verdict.js +3 -4
- package/vendor-core/scorecard-estimates.js +1 -0
- package/vendor-core/session-snapshot.js +7 -0
- package/vendor-core/shs-schemas.js +2 -2
- package/vendor-core/spark-memory.js +17 -0
- package/vendor-core/stage-plan-nodes.js +18 -0
- package/vendor-core/stage-quantiles.js +4 -0
- package/vendor-core/types.js +49 -1
- package/vendor-core/wasted-core-hours.js +10 -7
- package/vendor-core/write-targets.js +312 -0
|
@@ -22,8 +22,9 @@ import { assertNever } from './assert-never.js';
|
|
|
22
22
|
import { finalizeStage } from './stage-quantiles.js';
|
|
23
23
|
import { MAX_FAILURE_DETAILS_PER_STAGE, extractTaskFailureDetail, taskFailureKey, } from './task-failure.js';
|
|
24
24
|
import { computeRunAggregates } from './run-aggregates.js';
|
|
25
|
+
import { parseSparkMemoryMB } from './spark-memory.js';
|
|
25
26
|
|
|
26
|
-
|
|
27
|
+
|
|
27
28
|
|
|
28
29
|
|
|
29
30
|
// Internal parser-state shapes: the real runtime objects the handlers build and mutate, not the
|
|
@@ -129,6 +130,8 @@ const MAX_TASK_SAMPLES = 20;
|
|
|
129
130
|
|
|
130
131
|
|
|
131
132
|
|
|
133
|
+
|
|
134
|
+
|
|
132
135
|
|
|
133
136
|
|
|
134
137
|
|
|
@@ -149,12 +152,29 @@ const MAX_TASK_SAMPLES = 20;
|
|
|
149
152
|
|
|
150
153
|
|
|
151
154
|
|
|
155
|
+
|
|
156
|
+
|
|
152
157
|
|
|
153
158
|
|
|
154
159
|
|
|
155
160
|
|
|
156
161
|
|
|
157
162
|
|
|
163
|
+
|
|
164
|
+
|
|
165
|
+
|
|
166
|
+
|
|
167
|
+
|
|
168
|
+
|
|
169
|
+
|
|
170
|
+
|
|
171
|
+
|
|
172
|
+
|
|
173
|
+
|
|
174
|
+
|
|
175
|
+
|
|
176
|
+
|
|
177
|
+
|
|
158
178
|
|
|
159
179
|
|
|
160
180
|
|
|
@@ -198,6 +218,9 @@ const MAX_TASK_SAMPLES = 20;
|
|
|
198
218
|
|
|
199
219
|
|
|
200
220
|
|
|
221
|
+
|
|
222
|
+
|
|
223
|
+
|
|
201
224
|
|
|
202
225
|
|
|
203
226
|
|
|
@@ -391,6 +414,7 @@ export function createState() {
|
|
|
391
414
|
jobs: new Map(),
|
|
392
415
|
executors: { added: [], removed: [] },
|
|
393
416
|
skippedLines: 0,
|
|
417
|
+
unreadableSqlStarts: new Set(),
|
|
394
418
|
accumState: new Map(),
|
|
395
419
|
rddInfo: new Map(),
|
|
396
420
|
rddBlocks: new Map(),
|
|
@@ -432,23 +456,7 @@ export function normalizeSparkProperties(
|
|
|
432
456
|
return map;
|
|
433
457
|
}
|
|
434
458
|
|
|
435
|
-
|
|
436
|
-
// so a bare number means MiB. A k/m/g/t suffix sets the unit (trailing "b" redundant); a lone "b"
|
|
437
|
-
// ("10b") means bytes.
|
|
438
|
-
export function parseSparkMemoryMB(value ) {
|
|
439
|
-
if (value == null) return null;
|
|
440
|
-
const m = String(value).trim().toLowerCase().match(/^([\d.]+)\s*([kmgt]?)(b?)$/);
|
|
441
|
-
if (!m) return null;
|
|
442
|
-
const n = parseFloat(m[1]);
|
|
443
|
-
if (!Number.isFinite(n)) return null;
|
|
444
|
-
switch (m[2]) {
|
|
445
|
-
case 'k': return Math.round(n / 1024);
|
|
446
|
-
case 'g': return Math.round(n * 1024);
|
|
447
|
-
case 't': return Math.round(n * 1024 * 1024);
|
|
448
|
-
case 'm': return Math.round(n);
|
|
449
|
-
default: return m[3] === 'b' ? Math.round(n / (1024 * 1024)) : Math.round(n);
|
|
450
|
-
}
|
|
451
|
-
}
|
|
459
|
+
export { parseSparkMemoryMB };
|
|
452
460
|
|
|
453
461
|
// Derive an allocated-resource summary from the Spark config map. Absent keys degrade to null,
|
|
454
462
|
// not guessed defaults.
|
|
@@ -533,29 +541,11 @@ function internTaskFailure(stage , endReason
|
|
|
533
541
|
return detail;
|
|
534
542
|
}
|
|
535
543
|
|
|
536
|
-
|
|
537
|
-
|
|
538
|
-
|
|
539
|
-
if (!stage) return null;
|
|
540
|
-
// Late TaskEnd for a stage whose StageCompleted already freed taskAttempts (finalizeStage): its
|
|
541
|
-
// stats are already baked into the finalized stage, don't re-add. The one exception is a losing
|
|
542
|
-
// speculative attempt, whose wasted time the finalized stage never saw.
|
|
543
|
-
if (stage.taskAttempts === null) {
|
|
544
|
-
accountLateSpeculativeLoser(event, stage);
|
|
545
|
-
return null;
|
|
546
|
-
}
|
|
547
|
-
|
|
548
|
-
state.evidenceInputs.taskRecords++;
|
|
549
|
-
|
|
550
|
-
const accumulables = event['Task Info']?.Accumulables ?? [];
|
|
551
|
-
for (const acc of accumulables) {
|
|
552
|
-
if (!state.taskAccumStages.has(acc.ID)) state.taskAccumStages.set(acc.ID, new Set());
|
|
553
|
-
state.taskAccumStages.get(acc.ID) .add(stageId);
|
|
554
|
-
}
|
|
544
|
+
// 'Task Info' and its Failed/Killed/Speculative fields are optional in the schema; Partial<>
|
|
545
|
+
// lets the {} fallback type-check while reads below default via ??/||.
|
|
546
|
+
|
|
555
547
|
|
|
556
|
-
|
|
557
|
-
// lets the {} fallback type-check while reads below default via ??/||.
|
|
558
|
-
|
|
548
|
+
function taskRecordOf(event , failure ) {
|
|
559
549
|
const info = event['Task Info'] ?? {};
|
|
560
550
|
const m = event['Task Metrics'] ?? {};
|
|
561
551
|
const sr = m['Shuffle Read Metrics'] ?? {};
|
|
@@ -566,14 +556,14 @@ export function accumulateTask(event , state
|
|
|
566
556
|
const duration = (info['Finish Time'] ?? 0) - (info['Launch Time'] ?? 0);
|
|
567
557
|
const failed = !!(info['Failed'] || info['Killed']);
|
|
568
558
|
|
|
569
|
-
|
|
559
|
+
return {
|
|
570
560
|
duration, failed,
|
|
571
561
|
taskId: info['Task ID'] ?? null,
|
|
572
562
|
attemptNumber: info['Attempt Number'] ?? 0,
|
|
573
563
|
launchTime: info['Launch Time'] ?? 0,
|
|
574
564
|
finishTime: info['Finish Time'] ?? 0,
|
|
575
565
|
reason: event['Task End Reason']?.['Reason'] ?? null,
|
|
576
|
-
failure
|
|
566
|
+
failure,
|
|
577
567
|
speculative: info['Speculative'] === true,
|
|
578
568
|
host: info['Host'] ?? '',
|
|
579
569
|
executorId: info['Executor ID'] ?? '',
|
|
@@ -589,7 +579,37 @@ export function accumulateTask(event , state
|
|
|
589
579
|
executorCpuTime: m['Executor CPU Time'] ?? 0,
|
|
590
580
|
inputBytes: inp['Bytes Read'] ?? 0,
|
|
591
581
|
outputBytes: out['Bytes Written'] ?? 0,
|
|
582
|
+
outputRecords: out['Records Written'] ?? null,
|
|
592
583
|
};
|
|
584
|
+
}
|
|
585
|
+
|
|
586
|
+
export function accumulateTask(event , state ) {
|
|
587
|
+
const stageId = event['Stage ID'];
|
|
588
|
+
const stage = state.stages.get(stageId);
|
|
589
|
+
if (!stage) return null;
|
|
590
|
+
// Late TaskEnd for a stage whose StageCompleted already freed taskAttempts (finalizeStage): the
|
|
591
|
+
// finalized stage's figures stay as posted. A losing speculative attempt adds the wasted time the
|
|
592
|
+
// finalized stage never saw; any other task of a failed or earlier attempt is work only the
|
|
593
|
+
// metrics block reads, from lateAttemptWork.
|
|
594
|
+
if (stage.taskAttempts === null) {
|
|
595
|
+
const earlierAttempt = (event['Stage Attempt ID'] ?? 0) !== stage.stageAttemptId;
|
|
596
|
+
if (!accountLateSpeculativeLoser(event, stage) && (stage.stageFailureReason != null || earlierAttempt)) {
|
|
597
|
+
stage.lateAttemptWork = mergeAttemptTotals(taskAttemptTotals(taskRecordOf(event, null)), stage.lateAttemptWork);
|
|
598
|
+
}
|
|
599
|
+
return null;
|
|
600
|
+
}
|
|
601
|
+
|
|
602
|
+
state.evidenceInputs.taskRecords++;
|
|
603
|
+
|
|
604
|
+
const accumulables = event['Task Info']?.Accumulables ?? [];
|
|
605
|
+
for (const acc of accumulables) {
|
|
606
|
+
if (!state.taskAccumStages.has(acc.ID)) state.taskAccumStages.set(acc.ID, new Set());
|
|
607
|
+
state.taskAccumStages.get(acc.ID) .add(stageId);
|
|
608
|
+
}
|
|
609
|
+
|
|
610
|
+
const info = event['Task Info'] ?? {};
|
|
611
|
+
const failed = !!(info['Failed'] || info['Killed']);
|
|
612
|
+
const record = taskRecordOf(event, failed ? internTaskFailure(stage, event['Task End Reason']) : null);
|
|
593
613
|
|
|
594
614
|
// Dedupe only when Index is present (always true for real logs). Without it every event is a
|
|
595
615
|
// distinct task, preserving behavior for fixtures that omit Index.
|
|
@@ -614,9 +634,11 @@ export function accumulateTask(event , state
|
|
|
614
634
|
stage.retryTaskSamples.push(taskRecordToSample(existing));
|
|
615
635
|
}
|
|
616
636
|
}
|
|
637
|
+
stage.lateAttemptWork = mergeAttemptTotals(discardedAttemptTotals(existing), stage.lateAttemptWork);
|
|
617
638
|
stage.taskAttempts.set(key, record);
|
|
618
639
|
if (record.speculative) stage.speculativeWinners.add(key);
|
|
619
640
|
} else {
|
|
641
|
+
stage.lateAttemptWork = mergeAttemptTotals(discardedAttemptTotals(record), stage.lateAttemptWork);
|
|
620
642
|
// Non-winning duplicate (both failed, or a race where a winner is
|
|
621
643
|
// already recorded): its time is waste, its metrics are discarded.
|
|
622
644
|
if (existing.speculative || record.speculative) {
|
|
@@ -639,14 +661,16 @@ export function accumulateTask(event , state
|
|
|
639
661
|
// its time as speculation waste, pairing it the same way accumulateTask does: the late attempt is
|
|
640
662
|
// the speculative copy itself, or the original that a speculative winner beat. Every other stat
|
|
641
663
|
// of a late attempt stays excluded, as the finalized stage already posted them.
|
|
642
|
-
function accountLateSpeculativeLoser(event , stage )
|
|
664
|
+
function accountLateSpeculativeLoser(event , stage ) {
|
|
643
665
|
const info = event['Task Info'];
|
|
644
|
-
if (info?.['Index'] == null) return;
|
|
666
|
+
if (info?.['Index'] == null) return false;
|
|
645
667
|
const key = `${event['Stage Attempt ID'] ?? 0}:${info['Index']}`;
|
|
646
|
-
if (info['Speculative'] !== true && !stage.speculativeWinners.has(key)) return;
|
|
668
|
+
if (info['Speculative'] !== true && !stage.speculativeWinners.has(key)) return false;
|
|
647
669
|
stage.speculationWasteMs += (info['Finish Time'] ?? 0) - (info['Launch Time'] ?? 0);
|
|
648
670
|
stage.speculationWastedAttempts++;
|
|
649
671
|
stage.lateSpeculationWaste = true;
|
|
672
|
+
stage.lateAttemptWork = mergeAttemptTotals(discardedAttemptTotals(taskRecordOf(event, null)), stage.lateAttemptWork);
|
|
673
|
+
return true;
|
|
650
674
|
}
|
|
651
675
|
|
|
652
676
|
export function resolvePlanTree(
|
|
@@ -853,10 +877,63 @@ export function endJob(event , state
|
|
|
853
877
|
return { type: 'job', data: { ...job } };
|
|
854
878
|
}
|
|
855
879
|
|
|
880
|
+
const SUMMED_ATTEMPT_FIELDS = [
|
|
881
|
+
'taskCount', 'failedTasks', 'wastedAttempts', 'executorRunTime', 'executorCpuTime', 'jvmGCTime',
|
|
882
|
+
'memoryBytesSpilled', 'diskBytesSpilled', 'shuffleReadBytes', 'shuffleWriteBytes', 'inputBytes', 'outputBytes',
|
|
883
|
+
] ;
|
|
884
|
+
|
|
885
|
+
const addNullable = (a , b ) => (a == null ? b : b == null ? a : a + b);
|
|
886
|
+
|
|
887
|
+
function mergeAttemptTotals(a , b ) {
|
|
888
|
+
if (b == null) return a;
|
|
889
|
+
const totals = {
|
|
890
|
+
outputRecords: addNullable(a.outputRecords, b.outputRecords),
|
|
891
|
+
peakExecutionMemoryMax: Math.max(a.peakExecutionMemoryMax, b.peakExecutionMemoryMax),
|
|
892
|
+
durationMs: addNullable(a.durationMs, b.durationMs),
|
|
893
|
+
} ;
|
|
894
|
+
for (const field of SUMMED_ATTEMPT_FIELDS) totals[field] = a[field] + b[field];
|
|
895
|
+
return totals;
|
|
896
|
+
}
|
|
897
|
+
|
|
898
|
+
// One task's work; a late task adds no stage duration.
|
|
899
|
+
function taskAttemptTotals(t ) {
|
|
900
|
+
return {
|
|
901
|
+
taskCount: 1, failedTasks: t.failed ? 1 : 0, wastedAttempts: 0,
|
|
902
|
+
executorRunTime: t.executorRunTime, executorCpuTime: t.executorCpuTime, jvmGCTime: t.gcTime,
|
|
903
|
+
memoryBytesSpilled: t.memSpilled, diskBytesSpilled: t.diskSpilled,
|
|
904
|
+
shuffleReadBytes: t.shuffleRead, shuffleWriteBytes: t.shuffleWrite,
|
|
905
|
+
inputBytes: t.inputBytes, outputBytes: t.outputBytes, outputRecords: t.outputRecords,
|
|
906
|
+
peakExecutionMemoryMax: t.peakExecMem, durationMs: null,
|
|
907
|
+
};
|
|
908
|
+
}
|
|
909
|
+
|
|
910
|
+
// A task attempt the stage's figures drop because another attempt of the same task won (a retry
|
|
911
|
+
// after a failure, a speculative twin): its CPU, run time and I/O were still spent, so the
|
|
912
|
+
// metrics block counts them, but the task itself is already counted once.
|
|
913
|
+
function discardedAttemptTotals(t ) {
|
|
914
|
+
return { ...taskAttemptTotals(t), taskCount: 0, failedTasks: 0 };
|
|
915
|
+
}
|
|
916
|
+
|
|
917
|
+
// The replaced record's finalized attempt added to the attempts it had already folded. An attempt resubmitted before its StageCompleted was never finalized, so its
|
|
918
|
+
// tasks are not counted.
|
|
919
|
+
function foldEarlierAttempts(replaced ) {
|
|
920
|
+
if (!replaced) return null;
|
|
921
|
+
if (replaced.taskAttempts !== null) return replaced.earlierAttempts;
|
|
922
|
+
const attempt = {
|
|
923
|
+
...Object.fromEntries(SUMMED_ATTEMPT_FIELDS.map((field) => [field, replaced[field]])) ,
|
|
924
|
+
outputRecords: replaced.outputRecords,
|
|
925
|
+
peakExecutionMemoryMax: replaced.peakExecutionMemoryMax ?? 0,
|
|
926
|
+
durationMs: replaced.submittedAt > 0 && replaced.completedAt >= replaced.submittedAt
|
|
927
|
+
? replaced.completedAt - replaced.submittedAt : null,
|
|
928
|
+
};
|
|
929
|
+
return mergeAttemptTotals(attempt, replaced.earlierAttempts);
|
|
930
|
+
}
|
|
931
|
+
|
|
856
932
|
export function submitStage(event , state ) {
|
|
857
933
|
state.evidenceInputs.stageSubmissions++;
|
|
858
934
|
const info = event['Stage Info'];
|
|
859
935
|
const id = info['Stage ID'];
|
|
936
|
+
const replaced = state.stages.get(id);
|
|
860
937
|
state.stages.set(id, {
|
|
861
938
|
id, name: info['Stage Name'] ?? '', details: info['Details'] ?? '',
|
|
862
939
|
submittedAt: info['Submission Time'] ?? 0, completedAt: 0,
|
|
@@ -864,13 +941,18 @@ export function submitStage(event , st
|
|
|
864
941
|
shuffleReadBytes: 0, shuffleWriteBytes: 0, fetchWaitTime: 0,
|
|
865
942
|
memoryBytesSpilled: 0, diskBytesSpilled: 0,
|
|
866
943
|
jvmGCTime: 0, executorRunTime: 0, executorCpuTime: 0,
|
|
867
|
-
inputBytes: 0, outputBytes: 0,
|
|
944
|
+
inputBytes: 0, outputBytes: 0, outputRecords: null,
|
|
868
945
|
sqlExecutionId: state.stageToSqlExec.get(id) ?? null,
|
|
869
946
|
parentIds: info['Parent IDs'] ?? [],
|
|
870
947
|
hostStats: new Map(),
|
|
871
948
|
speculativeTasks: 0,
|
|
872
949
|
failureReasons: new Map(),
|
|
873
950
|
stageFailureReason: null,
|
|
951
|
+
stageAttemptId: info['Stage Attempt ID'] ?? 0,
|
|
952
|
+
stageAttempts: (replaced?.stageAttempts ?? 0) + 1,
|
|
953
|
+
failedStageAttempts: replaced?.failedStageAttempts ?? 0,
|
|
954
|
+
earlierAttempts: foldEarlierAttempts(replaced),
|
|
955
|
+
lateAttemptWork: replaced?.lateAttemptWork ?? null,
|
|
874
956
|
taskAttempts: new Map(),
|
|
875
957
|
failureDetails: new Map(),
|
|
876
958
|
retryTaskSamples: [],
|
|
@@ -1164,6 +1246,8 @@ export function processEvent(event , state ) {
|
|
|
1164
1246
|
// every stage-duration figure becomes the epoch timestamp itself (a "47-year" stage).
|
|
1165
1247
|
if (!stage.submittedAt && info['Submission Time'] != null) stage.submittedAt = info['Submission Time'];
|
|
1166
1248
|
stage.stageFailureReason = info['Failure Reason'] ?? null;
|
|
1249
|
+
// A duplicate StageCompleted of an already-finalized attempt is not another failed attempt.
|
|
1250
|
+
if (stage.stageFailureReason != null && stage.taskAttempts !== null) stage.failedStageAttempts++;
|
|
1167
1251
|
// finalizeStage keeps its `stage` parameter typed as a loose Record (see that module); bridge
|
|
1168
1252
|
// StageRecord's more precise shape across that boundary with an explicit cast.
|
|
1169
1253
|
return finalizeStage(
|
|
@@ -1363,6 +1447,18 @@ export function dispatchLine(
|
|
|
1363
1447
|
const BLOCK_UPDATED_PREFIX = '{"Event":"SparkListenerBlockUpdated",';
|
|
1364
1448
|
const RDD_BLOCK_ID_FRAGMENT = '"Block ID":"rdd_';
|
|
1365
1449
|
|
|
1450
|
+
const SQL_START_EVENT = 'org.apache.spark.sql.execution.ui.SparkListenerSQLExecutionStart';
|
|
1451
|
+
const SQL_START_ID = /"executionId":(\d+)/;
|
|
1452
|
+
|
|
1453
|
+
// Records the execution id of a skipped SQL start line, when the line's head still names it: a
|
|
1454
|
+
// line cut off mid-plan fails JSON.parse but keeps its leading fields.
|
|
1455
|
+
function noteUnreadableSqlStart(line , state ) {
|
|
1456
|
+
const head = line.slice(0, 300);
|
|
1457
|
+
if (!head.includes(`${SQL_START_EVENT}"`)) return;
|
|
1458
|
+
const id = SQL_START_ID.exec(head);
|
|
1459
|
+
if (id) state.unreadableSqlStarts.add(Number(id[1]));
|
|
1460
|
+
}
|
|
1461
|
+
|
|
1366
1462
|
function parseAndDispatch(line , state , emit ) {
|
|
1367
1463
|
if (line.startsWith(BLOCK_UPDATED_PREFIX) && !line.includes(RDD_BLOCK_ID_FRAGMENT)) return;
|
|
1368
1464
|
let parsed ;
|
|
@@ -1370,6 +1466,7 @@ function parseAndDispatch(line , state , emit
|
|
|
1370
1466
|
parsed = parseTaskEnd(line) ?? JSON.parse(stripPlanDescription(line));
|
|
1371
1467
|
} catch {
|
|
1372
1468
|
state.skippedLines++;
|
|
1469
|
+
noteUnreadableSqlStart(line, state);
|
|
1373
1470
|
return;
|
|
1374
1471
|
}
|
|
1375
1472
|
// A real Spark event log carries many event types this tool never modeled (BlockManagerAdded,
|
|
@@ -1382,6 +1479,7 @@ function parseAndDispatch(line , state , emit
|
|
|
1382
1479
|
const result = SparkEventSchema.safeParse(parsed);
|
|
1383
1480
|
if (!result.success) {
|
|
1384
1481
|
state.skippedLines++;
|
|
1482
|
+
if (eventType === SQL_START_EVENT) noteUnreadableSqlStart(line, state);
|
|
1385
1483
|
return;
|
|
1386
1484
|
}
|
|
1387
1485
|
if (result.data.Event === 'org.apache.spark.sql.execution.ui.SparkListenerSQLExecutionEnd') {
|
|
@@ -1428,14 +1526,29 @@ export function collectLateSpeculationWaste(
|
|
|
1428
1526
|
return out;
|
|
1429
1527
|
}
|
|
1430
1528
|
|
|
1529
|
+
// Late work of every stage a failed attempt's late TaskEnd added to, re-posted once before `done`:
|
|
1530
|
+
// the stage message posted at completion predates it.
|
|
1531
|
+
export function collectLateAttemptWork(state ) {
|
|
1532
|
+
const out = new Map ();
|
|
1533
|
+
for (const [id, stage] of state.stages) {
|
|
1534
|
+
if (stage.lateAttemptWork != null) out.set(id, stage.lateAttemptWork);
|
|
1535
|
+
}
|
|
1536
|
+
return out;
|
|
1537
|
+
}
|
|
1538
|
+
|
|
1431
1539
|
export function emitParseCompletion(state , emit , linesProcessed ) {
|
|
1432
1540
|
// Executions that never ended keep their latest AQE update, as they did before it was deferred.
|
|
1433
1541
|
for (const executionId of [...state.pendingAdaptiveUpdates.keys()]) flushAdaptiveUpdate(executionId, state, emit);
|
|
1434
1542
|
emit({ type: 'progress', pct: 1, linesProcessed });
|
|
1543
|
+
emit({ type: 'stageLateAttemptWork', data: collectLateAttemptWork(state) });
|
|
1435
1544
|
emit({ type: 'runAggregates', data: computeRunAggregates(state.taskStore) });
|
|
1436
1545
|
emit({ type: 'stageSpeculationWaste', data: collectLateSpeculationWaste(state) });
|
|
1437
1546
|
emit({ type: 'stageExecutorMetrics', data: collectStageExecutorMetrics(state) });
|
|
1438
1547
|
emit(appMessage(state));
|
|
1439
|
-
emit({
|
|
1548
|
+
emit({
|
|
1549
|
+
type: 'done',
|
|
1550
|
+
skippedLines: state.skippedLines,
|
|
1551
|
+
...(state.unreadableSqlStarts.size > 0 ? { unreadableSqlExecutions: [...state.unreadableSqlStarts].sort((a, b) => a - b) } : {}),
|
|
1552
|
+
});
|
|
1440
1553
|
state.accumState.clear();
|
|
1441
1554
|
}
|
|
@@ -201,6 +201,7 @@ export const StageSubmittedEventSchema = z.object({
|
|
|
201
201
|
Event: z.literal('SparkListenerStageSubmitted'),
|
|
202
202
|
'Stage Info': z.object({
|
|
203
203
|
'Stage ID': z.number(),
|
|
204
|
+
'Stage Attempt ID': z.number().optional(),
|
|
204
205
|
'Stage Name': z.string().optional(),
|
|
205
206
|
Details: z.string().optional(),
|
|
206
207
|
'Submission Time': z.number().optional(),
|
|
@@ -311,6 +312,7 @@ export const TaskEndEventSchema = z.object({
|
|
|
311
312
|
}).optional(),
|
|
312
313
|
'Output Metrics': z.object({
|
|
313
314
|
'Bytes Written': z.number().optional(),
|
|
315
|
+
'Records Written': z.number().optional(),
|
|
314
316
|
}).optional(),
|
|
315
317
|
}).optional(),
|
|
316
318
|
});
|
|
@@ -8,24 +8,25 @@ import {
|
|
|
8
8
|
} from './threshold-overrides.js';
|
|
9
9
|
import { getThresholdSummary } from './threshold-summary.js';
|
|
10
10
|
import {
|
|
11
|
-
typeTag, formatBytes, formatCores, formatDuration,
|
|
11
|
+
typeTag, formatBytes, formatCores, formatDuration, formatWallClockRange, IMPACT_BAND_ORDER,
|
|
12
12
|
} from './format-utils.js';
|
|
13
13
|
import { findingName, titleCase } from './finding-names.js';
|
|
14
14
|
import { redactReport, redactRunModel } from './redact.js';
|
|
15
15
|
import { formatTaskFailureHeadline, } from './task-failure.js';
|
|
16
16
|
import { findingActionLabel } from './finding-action-label.js';
|
|
17
17
|
import { matchesFindingFilterCriteria, singleStageId } from './finding-filter-predicate.js';
|
|
18
|
-
import { buildRecommendationRollup, isEligible, isRealFinding, rankFindings, } from './recommendation-rollup.js';
|
|
18
|
+
import { buildRecommendationRollup, isEligible, isRealFinding, rankFindings, resourceGroupTotal, } from './recommendation-rollup.js';
|
|
19
19
|
import { checkCoverage, isCleanRun } from './check-coverage.js';
|
|
20
20
|
import { buildRunVerdict, stepCopyRecommendation, stepCopyText, } from './run-verdict.js';
|
|
21
21
|
import {
|
|
22
22
|
estimateProvenance, impactEstimateFigure, impactFigure, rawWasteMeaning, savingsMeaning,
|
|
23
23
|
} from './impact-format.js';
|
|
24
24
|
import { computeRunShape, } from './run-shape.js';
|
|
25
|
+
import { extractWriteTargets, } from './write-targets.js';
|
|
25
26
|
import { detectorInfoByType } from './detector-docs.js';
|
|
26
27
|
|
|
27
28
|
|
|
28
|
-
|
|
29
|
+
|
|
29
30
|
|
|
30
31
|
|
|
31
32
|
export const EVIDENCE_SCHEMA_VERSION = 5;
|
|
@@ -43,9 +44,11 @@ export const EVIDENCE_SCHEMA_VERSION = 5;
|
|
|
43
44
|
|
|
44
45
|
|
|
45
46
|
|
|
46
|
-
|
|
47
|
-
|
|
48
|
-
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
|
|
49
52
|
|
|
50
53
|
|
|
51
54
|
|
|
@@ -165,6 +168,8 @@ export const EVIDENCE_SCHEMA_VERSION = 5;
|
|
|
165
168
|
|
|
166
169
|
|
|
167
170
|
|
|
171
|
+
|
|
172
|
+
|
|
168
173
|
|
|
169
174
|
|
|
170
175
|
|
|
@@ -244,6 +249,7 @@ function findingRow(f ) {
|
|
|
244
249
|
recommendation: f.recommendation ?? null,
|
|
245
250
|
detectorVersion: f.detectorVersion ?? 1,
|
|
246
251
|
evidence: projectEvidence(f),
|
|
252
|
+
remediation: f.remediation ?? [],
|
|
247
253
|
actionLabel: findingActionLabel(f),
|
|
248
254
|
} ;
|
|
249
255
|
// Threshold/confidence provenance, only when the detector emitted it.
|
|
@@ -311,15 +317,14 @@ function buildRecommendations(
|
|
|
311
317
|
};
|
|
312
318
|
}
|
|
313
319
|
if (group.kind === 'resource') {
|
|
314
|
-
const
|
|
315
|
-
const shown = readsAsZero(text) ? null : text;
|
|
320
|
+
const shown = resourceGroupTotal(group);
|
|
316
321
|
return {
|
|
317
322
|
...base,
|
|
318
323
|
kind: 'resource',
|
|
319
324
|
unit: group.unit,
|
|
320
325
|
total: group.total,
|
|
321
326
|
impact: shown,
|
|
322
|
-
impactMeaning: shown ? rawWasteMeaning(
|
|
327
|
+
impactMeaning: shown ? rawWasteMeaning(representative.impactEstimate?.rawWaste) : null,
|
|
323
328
|
};
|
|
324
329
|
}
|
|
325
330
|
return {
|
|
@@ -495,6 +500,9 @@ function buildJson(
|
|
|
495
500
|
// Order follows DETECTORS (stable) => byte-stable serialization.
|
|
496
501
|
detectors: tunedDetectorCatalog(thresholds),
|
|
497
502
|
findings: rows,
|
|
503
|
+
writeTargets: extractWriteTargets(sql ?? new Map(), {
|
|
504
|
+
skippedLines: appModel.skippedLines, unreadableSqlExecutions: appModel.unreadableSqlExecutions,
|
|
505
|
+
}),
|
|
498
506
|
recommendations,
|
|
499
507
|
cleanChecks,
|
|
500
508
|
notRunChecks,
|
|
@@ -581,7 +589,7 @@ function renderVerdict(verdict ) {
|
|
|
581
589
|
lines.push(`${i + 1}. [${step.tag}] ${step.text}`);
|
|
582
590
|
if (step.relatedTypes.length > 0) {
|
|
583
591
|
const related = step.relatedTypes.map(findingName).join(', ');
|
|
584
|
-
lines.push(` - Also flagged here
|
|
592
|
+
lines.push(` - Also flagged here, likely the same cause: ${related}.`);
|
|
585
593
|
}
|
|
586
594
|
});
|
|
587
595
|
if (verdict.remainingPlaces > 0) {
|
|
@@ -1,5 +1,6 @@
|
|
|
1
1
|
import { presentationOf } from './finding-presentation.js';
|
|
2
|
-
|
|
2
|
+
import { pathBasename } from './format-utils.js';
|
|
3
|
+
|
|
3
4
|
|
|
4
5
|
/** A generic, type-level recommendation sentence for a finding: the shape of the fix, with no
|
|
5
6
|
* instance data (numbers, stage ids, host names, file counts, config values), from its type's
|
|
@@ -12,3 +13,21 @@ import { presentationOf } from './finding-presentation.js';
|
|
|
12
13
|
export function coreFindingGenericRecommendation(finding ) {
|
|
13
14
|
return presentationOf(finding.type)?.genericRecommendation(finding);
|
|
14
15
|
}
|
|
16
|
+
|
|
17
|
+
/** What a duplicatePlanSubtree finding measured, for the detector's recommendation and the
|
|
18
|
+
* Redundant Plan Subtree row, which shows it beside the fix its card states once. */
|
|
19
|
+
export function duplicateSubtreeDetail(f ) {
|
|
20
|
+
const touching = f.sampleRelation ? ` (touching ${f.sampleRelation})` : '';
|
|
21
|
+
return `A ${f.subtreeSize}-node subtree rooted at ${pathBasename(f.rootName)} repeats ${f.value}x in this plan${touching}`;
|
|
22
|
+
}
|
|
23
|
+
|
|
24
|
+
export const DUPLICATE_SUBTREE_DIFFERING_NOTE = 'Their filters, columns or scanned tables differ, so the repeats may compute different data.';
|
|
25
|
+
|
|
26
|
+
/** Reader-facing names for slowHost's per-executor dimensions, for the detector's recommendation
|
|
27
|
+
* and the Slow Executor Host row. */
|
|
28
|
+
export const SLOW_HOST_DIMENSION_LABEL = {
|
|
29
|
+
taskTime: 'task time',
|
|
30
|
+
inputBytes: 'input read',
|
|
31
|
+
shuffleBytes: 'shuffle read and write',
|
|
32
|
+
storageMemory: 'storage memory',
|
|
33
|
+
};
|
|
@@ -25,3 +25,10 @@ export function recommendationText(finding ) {
|
|
|
25
25
|
if (text) return text;
|
|
26
26
|
return findingName(finding.type);
|
|
27
27
|
}
|
|
28
|
+
|
|
29
|
+
/** A recommendation's two halves: detectors write "<measurement>: <fix>", split at the last ": "
|
|
30
|
+
* since a measurement can quote an error with colons. Text with no split has no `measured`. */
|
|
31
|
+
export function recommendationParts(text ) {
|
|
32
|
+
const at = text.lastIndexOf(': ');
|
|
33
|
+
return at > 0 ? { measured: text.slice(0, at), fix: text.slice(at + 2) } : { measured: null, fix: text };
|
|
34
|
+
}
|