sparkforensics-mcp 0.4.0 → 0.5.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (64) hide show
  1. package/package.json +1 -1
  2. package/vendor-core/allocation.js +106 -0
  3. package/vendor-core/analyzer.js +13 -13
  4. package/vendor-core/cli/budgets.js +23 -9
  5. package/vendor-core/cli/collect-run.js +11 -4
  6. package/vendor-core/cli/regression-budgets.js +83 -0
  7. package/vendor-core/comparison-verdict.js +22 -22
  8. package/vendor-core/core-source-hash.txt +1 -1
  9. package/vendor-core/detectors.js +202 -68
  10. package/vendor-core/docs-content/detection/cache.md +3 -2
  11. package/vendor-core/docs-content/detection/cfg.md +9 -8
  12. package/vendor-core/docs-content/detection/chrn.md +1 -2
  13. package/vendor-core/docs-content/detection/cold.md +4 -2
  14. package/vendor-core/docs-content/detection/fail.md +3 -2
  15. package/vendor-core/docs-content/detection/gc.md +3 -2
  16. package/vendor-core/docs-content/detection/host.md +2 -1
  17. package/vendor-core/docs-content/detection/local.md +1 -1
  18. package/vendor-core/docs-content/detection/mem.md +5 -2
  19. package/vendor-core/docs-content/detection/plan.md +2 -1
  20. package/vendor-core/docs-content/detection/sfail.md +2 -1
  21. package/vendor-core/docs-content/detection/shape.md +5 -4
  22. package/vendor-core/docs-content/detection/skew.md +3 -1
  23. package/vendor-core/docs-content/detection/slow.md +2 -2
  24. package/vendor-core/docs-content/detection/spec.md +2 -3
  25. package/vendor-core/docs-content/detection/spill.md +1 -1
  26. package/vendor-core/docs-site-config.js +1 -1
  27. package/vendor-core/effective-conf.js +107 -0
  28. package/vendor-core/efficiency-model.js +8 -6
  29. package/vendor-core/event-handlers.js +160 -47
  30. package/vendor-core/event-schemas.js +2 -0
  31. package/vendor-core/evidence-report.js +18 -10
  32. package/vendor-core/finding-generic-recommendation.js +20 -1
  33. package/vendor-core/finding-names.js +7 -0
  34. package/vendor-core/finding-presentation.js +61 -26
  35. package/vendor-core/finding-tag-help.js +1 -1
  36. package/vendor-core/finding-types.js +12 -0
  37. package/vendor-core/format-utils.js +4 -3
  38. package/vendor-core/impact-estimator.js +20 -2
  39. package/vendor-core/impact-format.js +14 -13
  40. package/vendor-core/impact-model.js +27 -5
  41. package/vendor-core/ingest.js +4 -2
  42. package/vendor-core/list-runs.js +5 -2
  43. package/vendor-core/mcp-tools.js +1 -1
  44. package/vendor-core/model-assembler.js +23 -1
  45. package/vendor-core/parser-worker.js +1 -1
  46. package/vendor-core/proxy.js +3 -1
  47. package/vendor-core/python-stage.js +25 -0
  48. package/vendor-core/recommendation-rollup.js +16 -9
  49. package/vendor-core/redact.js +51 -10
  50. package/vendor-core/remediation.js +20 -0
  51. package/vendor-core/run-comparison.js +43 -24
  52. package/vendor-core/run-interpretation.js +2 -1
  53. package/vendor-core/run-metrics.js +198 -0
  54. package/vendor-core/run-totals.js +24 -0
  55. package/vendor-core/run-verdict.js +3 -4
  56. package/vendor-core/scorecard-estimates.js +1 -0
  57. package/vendor-core/session-snapshot.js +7 -0
  58. package/vendor-core/shs-schemas.js +2 -2
  59. package/vendor-core/spark-memory.js +17 -0
  60. package/vendor-core/stage-plan-nodes.js +18 -0
  61. package/vendor-core/stage-quantiles.js +4 -0
  62. package/vendor-core/types.js +49 -1
  63. package/vendor-core/wasted-core-hours.js +10 -7
  64. package/vendor-core/write-targets.js +312 -0
@@ -22,8 +22,9 @@ import { assertNever } from './assert-never.js';
22
22
  import { finalizeStage } from './stage-quantiles.js';
23
23
  import { MAX_FAILURE_DETAILS_PER_STAGE, extractTaskFailureDetail, taskFailureKey, } from './task-failure.js';
24
24
  import { computeRunAggregates } from './run-aggregates.js';
25
+ import { parseSparkMemoryMB } from './spark-memory.js';
25
26
 
26
-
27
+
27
28
 
28
29
 
29
30
  // Internal parser-state shapes: the real runtime objects the handlers build and mutate, not the
@@ -129,6 +130,8 @@ const MAX_TASK_SAMPLES = 20;
129
130
 
130
131
 
131
132
 
133
+
134
+
132
135
 
133
136
 
134
137
 
@@ -149,12 +152,29 @@ const MAX_TASK_SAMPLES = 20;
149
152
 
150
153
 
151
154
 
155
+
156
+
152
157
 
153
158
 
154
159
 
155
160
 
156
161
 
157
162
 
163
+
164
+
165
+
166
+
167
+
168
+
169
+
170
+
171
+
172
+
173
+
174
+
175
+
176
+
177
+
158
178
 
159
179
 
160
180
 
@@ -198,6 +218,9 @@ const MAX_TASK_SAMPLES = 20;
198
218
 
199
219
 
200
220
 
221
+
222
+
223
+
201
224
 
202
225
 
203
226
 
@@ -391,6 +414,7 @@ export function createState() {
391
414
  jobs: new Map(),
392
415
  executors: { added: [], removed: [] },
393
416
  skippedLines: 0,
417
+ unreadableSqlStarts: new Set(),
394
418
  accumState: new Map(),
395
419
  rddInfo: new Map(),
396
420
  rddBlocks: new Map(),
@@ -432,23 +456,7 @@ export function normalizeSparkProperties(
432
456
  return map;
433
457
  }
434
458
 
435
- // Parse a Spark memory-size string to MiB. Spark's JVM-memory configs use bytesConf(ByteUnit.MiB),
436
- // so a bare number means MiB. A k/m/g/t suffix sets the unit (trailing "b" redundant); a lone "b"
437
- // ("10b") means bytes.
438
- export function parseSparkMemoryMB(value ) {
439
- if (value == null) return null;
440
- const m = String(value).trim().toLowerCase().match(/^([\d.]+)\s*([kmgt]?)(b?)$/);
441
- if (!m) return null;
442
- const n = parseFloat(m[1]);
443
- if (!Number.isFinite(n)) return null;
444
- switch (m[2]) {
445
- case 'k': return Math.round(n / 1024);
446
- case 'g': return Math.round(n * 1024);
447
- case 't': return Math.round(n * 1024 * 1024);
448
- case 'm': return Math.round(n);
449
- default: return m[3] === 'b' ? Math.round(n / (1024 * 1024)) : Math.round(n);
450
- }
451
- }
459
+ export { parseSparkMemoryMB };
452
460
 
453
461
  // Derive an allocated-resource summary from the Spark config map. Absent keys degrade to null,
454
462
  // not guessed defaults.
@@ -533,29 +541,11 @@ function internTaskFailure(stage , endReason
533
541
  return detail;
534
542
  }
535
543
 
536
- export function accumulateTask(event , state ) {
537
- const stageId = event['Stage ID'];
538
- const stage = state.stages.get(stageId);
539
- if (!stage) return null;
540
- // Late TaskEnd for a stage whose StageCompleted already freed taskAttempts (finalizeStage): its
541
- // stats are already baked into the finalized stage, don't re-add. The one exception is a losing
542
- // speculative attempt, whose wasted time the finalized stage never saw.
543
- if (stage.taskAttempts === null) {
544
- accountLateSpeculativeLoser(event, stage);
545
- return null;
546
- }
547
-
548
- state.evidenceInputs.taskRecords++;
549
-
550
- const accumulables = event['Task Info']?.Accumulables ?? [];
551
- for (const acc of accumulables) {
552
- if (!state.taskAccumStages.has(acc.ID)) state.taskAccumStages.set(acc.ID, new Set());
553
- state.taskAccumStages.get(acc.ID) .add(stageId);
554
- }
544
+ // 'Task Info' and its Failed/Killed/Speculative fields are optional in the schema; Partial<>
545
+ // lets the {} fallback type-check while reads below default via ??/||.
546
+
555
547
 
556
- // 'Task Info' and its Failed/Killed/Speculative fields are optional in the schema; Partial<>
557
- // lets the {} fallback type-check while reads below default via ??/||.
558
-
548
+ function taskRecordOf(event , failure ) {
559
549
  const info = event['Task Info'] ?? {};
560
550
  const m = event['Task Metrics'] ?? {};
561
551
  const sr = m['Shuffle Read Metrics'] ?? {};
@@ -566,14 +556,14 @@ export function accumulateTask(event , state
566
556
  const duration = (info['Finish Time'] ?? 0) - (info['Launch Time'] ?? 0);
567
557
  const failed = !!(info['Failed'] || info['Killed']);
568
558
 
569
- const record = {
559
+ return {
570
560
  duration, failed,
571
561
  taskId: info['Task ID'] ?? null,
572
562
  attemptNumber: info['Attempt Number'] ?? 0,
573
563
  launchTime: info['Launch Time'] ?? 0,
574
564
  finishTime: info['Finish Time'] ?? 0,
575
565
  reason: event['Task End Reason']?.['Reason'] ?? null,
576
- failure: failed ? internTaskFailure(stage, event['Task End Reason']) : null,
566
+ failure,
577
567
  speculative: info['Speculative'] === true,
578
568
  host: info['Host'] ?? '',
579
569
  executorId: info['Executor ID'] ?? '',
@@ -589,7 +579,37 @@ export function accumulateTask(event , state
589
579
  executorCpuTime: m['Executor CPU Time'] ?? 0,
590
580
  inputBytes: inp['Bytes Read'] ?? 0,
591
581
  outputBytes: out['Bytes Written'] ?? 0,
582
+ outputRecords: out['Records Written'] ?? null,
592
583
  };
584
+ }
585
+
586
+ export function accumulateTask(event , state ) {
587
+ const stageId = event['Stage ID'];
588
+ const stage = state.stages.get(stageId);
589
+ if (!stage) return null;
590
+ // Late TaskEnd for a stage whose StageCompleted already freed taskAttempts (finalizeStage): the
591
+ // finalized stage's figures stay as posted. A losing speculative attempt adds the wasted time the
592
+ // finalized stage never saw; any other task of a failed or earlier attempt is work only the
593
+ // metrics block reads, from lateAttemptWork.
594
+ if (stage.taskAttempts === null) {
595
+ const earlierAttempt = (event['Stage Attempt ID'] ?? 0) !== stage.stageAttemptId;
596
+ if (!accountLateSpeculativeLoser(event, stage) && (stage.stageFailureReason != null || earlierAttempt)) {
597
+ stage.lateAttemptWork = mergeAttemptTotals(taskAttemptTotals(taskRecordOf(event, null)), stage.lateAttemptWork);
598
+ }
599
+ return null;
600
+ }
601
+
602
+ state.evidenceInputs.taskRecords++;
603
+
604
+ const accumulables = event['Task Info']?.Accumulables ?? [];
605
+ for (const acc of accumulables) {
606
+ if (!state.taskAccumStages.has(acc.ID)) state.taskAccumStages.set(acc.ID, new Set());
607
+ state.taskAccumStages.get(acc.ID) .add(stageId);
608
+ }
609
+
610
+ const info = event['Task Info'] ?? {};
611
+ const failed = !!(info['Failed'] || info['Killed']);
612
+ const record = taskRecordOf(event, failed ? internTaskFailure(stage, event['Task End Reason']) : null);
593
613
 
594
614
  // Dedupe only when Index is present (always true for real logs). Without it every event is a
595
615
  // distinct task, preserving behavior for fixtures that omit Index.
@@ -614,9 +634,11 @@ export function accumulateTask(event , state
614
634
  stage.retryTaskSamples.push(taskRecordToSample(existing));
615
635
  }
616
636
  }
637
+ stage.lateAttemptWork = mergeAttemptTotals(discardedAttemptTotals(existing), stage.lateAttemptWork);
617
638
  stage.taskAttempts.set(key, record);
618
639
  if (record.speculative) stage.speculativeWinners.add(key);
619
640
  } else {
641
+ stage.lateAttemptWork = mergeAttemptTotals(discardedAttemptTotals(record), stage.lateAttemptWork);
620
642
  // Non-winning duplicate (both failed, or a race where a winner is
621
643
  // already recorded): its time is waste, its metrics are discarded.
622
644
  if (existing.speculative || record.speculative) {
@@ -639,14 +661,16 @@ export function accumulateTask(event , state
639
661
  // its time as speculation waste, pairing it the same way accumulateTask does: the late attempt is
640
662
  // the speculative copy itself, or the original that a speculative winner beat. Every other stat
641
663
  // of a late attempt stays excluded, as the finalized stage already posted them.
642
- function accountLateSpeculativeLoser(event , stage ) {
664
+ function accountLateSpeculativeLoser(event , stage ) {
643
665
  const info = event['Task Info'];
644
- if (info?.['Index'] == null) return;
666
+ if (info?.['Index'] == null) return false;
645
667
  const key = `${event['Stage Attempt ID'] ?? 0}:${info['Index']}`;
646
- if (info['Speculative'] !== true && !stage.speculativeWinners.has(key)) return;
668
+ if (info['Speculative'] !== true && !stage.speculativeWinners.has(key)) return false;
647
669
  stage.speculationWasteMs += (info['Finish Time'] ?? 0) - (info['Launch Time'] ?? 0);
648
670
  stage.speculationWastedAttempts++;
649
671
  stage.lateSpeculationWaste = true;
672
+ stage.lateAttemptWork = mergeAttemptTotals(discardedAttemptTotals(taskRecordOf(event, null)), stage.lateAttemptWork);
673
+ return true;
650
674
  }
651
675
 
652
676
  export function resolvePlanTree(
@@ -853,10 +877,63 @@ export function endJob(event , state
853
877
  return { type: 'job', data: { ...job } };
854
878
  }
855
879
 
880
+ const SUMMED_ATTEMPT_FIELDS = [
881
+ 'taskCount', 'failedTasks', 'wastedAttempts', 'executorRunTime', 'executorCpuTime', 'jvmGCTime',
882
+ 'memoryBytesSpilled', 'diskBytesSpilled', 'shuffleReadBytes', 'shuffleWriteBytes', 'inputBytes', 'outputBytes',
883
+ ] ;
884
+
885
+ const addNullable = (a , b ) => (a == null ? b : b == null ? a : a + b);
886
+
887
+ function mergeAttemptTotals(a , b ) {
888
+ if (b == null) return a;
889
+ const totals = {
890
+ outputRecords: addNullable(a.outputRecords, b.outputRecords),
891
+ peakExecutionMemoryMax: Math.max(a.peakExecutionMemoryMax, b.peakExecutionMemoryMax),
892
+ durationMs: addNullable(a.durationMs, b.durationMs),
893
+ } ;
894
+ for (const field of SUMMED_ATTEMPT_FIELDS) totals[field] = a[field] + b[field];
895
+ return totals;
896
+ }
897
+
898
+ // One task's work; a late task adds no stage duration.
899
+ function taskAttemptTotals(t ) {
900
+ return {
901
+ taskCount: 1, failedTasks: t.failed ? 1 : 0, wastedAttempts: 0,
902
+ executorRunTime: t.executorRunTime, executorCpuTime: t.executorCpuTime, jvmGCTime: t.gcTime,
903
+ memoryBytesSpilled: t.memSpilled, diskBytesSpilled: t.diskSpilled,
904
+ shuffleReadBytes: t.shuffleRead, shuffleWriteBytes: t.shuffleWrite,
905
+ inputBytes: t.inputBytes, outputBytes: t.outputBytes, outputRecords: t.outputRecords,
906
+ peakExecutionMemoryMax: t.peakExecMem, durationMs: null,
907
+ };
908
+ }
909
+
910
+ // A task attempt the stage's figures drop because another attempt of the same task won (a retry
911
+ // after a failure, a speculative twin): its CPU, run time and I/O were still spent, so the
912
+ // metrics block counts them, but the task itself is already counted once.
913
+ function discardedAttemptTotals(t ) {
914
+ return { ...taskAttemptTotals(t), taskCount: 0, failedTasks: 0 };
915
+ }
916
+
917
+ // The replaced record's finalized attempt added to the attempts it had already folded. An attempt resubmitted before its StageCompleted was never finalized, so its
918
+ // tasks are not counted.
919
+ function foldEarlierAttempts(replaced ) {
920
+ if (!replaced) return null;
921
+ if (replaced.taskAttempts !== null) return replaced.earlierAttempts;
922
+ const attempt = {
923
+ ...Object.fromEntries(SUMMED_ATTEMPT_FIELDS.map((field) => [field, replaced[field]])) ,
924
+ outputRecords: replaced.outputRecords,
925
+ peakExecutionMemoryMax: replaced.peakExecutionMemoryMax ?? 0,
926
+ durationMs: replaced.submittedAt > 0 && replaced.completedAt >= replaced.submittedAt
927
+ ? replaced.completedAt - replaced.submittedAt : null,
928
+ };
929
+ return mergeAttemptTotals(attempt, replaced.earlierAttempts);
930
+ }
931
+
856
932
  export function submitStage(event , state ) {
857
933
  state.evidenceInputs.stageSubmissions++;
858
934
  const info = event['Stage Info'];
859
935
  const id = info['Stage ID'];
936
+ const replaced = state.stages.get(id);
860
937
  state.stages.set(id, {
861
938
  id, name: info['Stage Name'] ?? '', details: info['Details'] ?? '',
862
939
  submittedAt: info['Submission Time'] ?? 0, completedAt: 0,
@@ -864,13 +941,18 @@ export function submitStage(event , st
864
941
  shuffleReadBytes: 0, shuffleWriteBytes: 0, fetchWaitTime: 0,
865
942
  memoryBytesSpilled: 0, diskBytesSpilled: 0,
866
943
  jvmGCTime: 0, executorRunTime: 0, executorCpuTime: 0,
867
- inputBytes: 0, outputBytes: 0,
944
+ inputBytes: 0, outputBytes: 0, outputRecords: null,
868
945
  sqlExecutionId: state.stageToSqlExec.get(id) ?? null,
869
946
  parentIds: info['Parent IDs'] ?? [],
870
947
  hostStats: new Map(),
871
948
  speculativeTasks: 0,
872
949
  failureReasons: new Map(),
873
950
  stageFailureReason: null,
951
+ stageAttemptId: info['Stage Attempt ID'] ?? 0,
952
+ stageAttempts: (replaced?.stageAttempts ?? 0) + 1,
953
+ failedStageAttempts: replaced?.failedStageAttempts ?? 0,
954
+ earlierAttempts: foldEarlierAttempts(replaced),
955
+ lateAttemptWork: replaced?.lateAttemptWork ?? null,
874
956
  taskAttempts: new Map(),
875
957
  failureDetails: new Map(),
876
958
  retryTaskSamples: [],
@@ -1164,6 +1246,8 @@ export function processEvent(event , state ) {
1164
1246
  // every stage-duration figure becomes the epoch timestamp itself (a "47-year" stage).
1165
1247
  if (!stage.submittedAt && info['Submission Time'] != null) stage.submittedAt = info['Submission Time'];
1166
1248
  stage.stageFailureReason = info['Failure Reason'] ?? null;
1249
+ // A duplicate StageCompleted of an already-finalized attempt is not another failed attempt.
1250
+ if (stage.stageFailureReason != null && stage.taskAttempts !== null) stage.failedStageAttempts++;
1167
1251
  // finalizeStage keeps its `stage` parameter typed as a loose Record (see that module); bridge
1168
1252
  // StageRecord's more precise shape across that boundary with an explicit cast.
1169
1253
  return finalizeStage(
@@ -1363,6 +1447,18 @@ export function dispatchLine(
1363
1447
  const BLOCK_UPDATED_PREFIX = '{"Event":"SparkListenerBlockUpdated",';
1364
1448
  const RDD_BLOCK_ID_FRAGMENT = '"Block ID":"rdd_';
1365
1449
 
1450
+ const SQL_START_EVENT = 'org.apache.spark.sql.execution.ui.SparkListenerSQLExecutionStart';
1451
+ const SQL_START_ID = /"executionId":(\d+)/;
1452
+
1453
+ // Records the execution id of a skipped SQL start line, when the line's head still names it: a
1454
+ // line cut off mid-plan fails JSON.parse but keeps its leading fields.
1455
+ function noteUnreadableSqlStart(line , state ) {
1456
+ const head = line.slice(0, 300);
1457
+ if (!head.includes(`${SQL_START_EVENT}"`)) return;
1458
+ const id = SQL_START_ID.exec(head);
1459
+ if (id) state.unreadableSqlStarts.add(Number(id[1]));
1460
+ }
1461
+
1366
1462
  function parseAndDispatch(line , state , emit ) {
1367
1463
  if (line.startsWith(BLOCK_UPDATED_PREFIX) && !line.includes(RDD_BLOCK_ID_FRAGMENT)) return;
1368
1464
  let parsed ;
@@ -1370,6 +1466,7 @@ function parseAndDispatch(line , state , emit
1370
1466
  parsed = parseTaskEnd(line) ?? JSON.parse(stripPlanDescription(line));
1371
1467
  } catch {
1372
1468
  state.skippedLines++;
1469
+ noteUnreadableSqlStart(line, state);
1373
1470
  return;
1374
1471
  }
1375
1472
  // A real Spark event log carries many event types this tool never modeled (BlockManagerAdded,
@@ -1382,6 +1479,7 @@ function parseAndDispatch(line , state , emit
1382
1479
  const result = SparkEventSchema.safeParse(parsed);
1383
1480
  if (!result.success) {
1384
1481
  state.skippedLines++;
1482
+ if (eventType === SQL_START_EVENT) noteUnreadableSqlStart(line, state);
1385
1483
  return;
1386
1484
  }
1387
1485
  if (result.data.Event === 'org.apache.spark.sql.execution.ui.SparkListenerSQLExecutionEnd') {
@@ -1428,14 +1526,29 @@ export function collectLateSpeculationWaste(
1428
1526
  return out;
1429
1527
  }
1430
1528
 
1529
+ // Late work of every stage a failed attempt's late TaskEnd added to, re-posted once before `done`:
1530
+ // the stage message posted at completion predates it.
1531
+ export function collectLateAttemptWork(state ) {
1532
+ const out = new Map ();
1533
+ for (const [id, stage] of state.stages) {
1534
+ if (stage.lateAttemptWork != null) out.set(id, stage.lateAttemptWork);
1535
+ }
1536
+ return out;
1537
+ }
1538
+
1431
1539
  export function emitParseCompletion(state , emit , linesProcessed ) {
1432
1540
  // Executions that never ended keep their latest AQE update, as they did before it was deferred.
1433
1541
  for (const executionId of [...state.pendingAdaptiveUpdates.keys()]) flushAdaptiveUpdate(executionId, state, emit);
1434
1542
  emit({ type: 'progress', pct: 1, linesProcessed });
1543
+ emit({ type: 'stageLateAttemptWork', data: collectLateAttemptWork(state) });
1435
1544
  emit({ type: 'runAggregates', data: computeRunAggregates(state.taskStore) });
1436
1545
  emit({ type: 'stageSpeculationWaste', data: collectLateSpeculationWaste(state) });
1437
1546
  emit({ type: 'stageExecutorMetrics', data: collectStageExecutorMetrics(state) });
1438
1547
  emit(appMessage(state));
1439
- emit({ type: 'done', skippedLines: state.skippedLines });
1548
+ emit({
1549
+ type: 'done',
1550
+ skippedLines: state.skippedLines,
1551
+ ...(state.unreadableSqlStarts.size > 0 ? { unreadableSqlExecutions: [...state.unreadableSqlStarts].sort((a, b) => a - b) } : {}),
1552
+ });
1440
1553
  state.accumState.clear();
1441
1554
  }
@@ -201,6 +201,7 @@ export const StageSubmittedEventSchema = z.object({
201
201
  Event: z.literal('SparkListenerStageSubmitted'),
202
202
  'Stage Info': z.object({
203
203
  'Stage ID': z.number(),
204
+ 'Stage Attempt ID': z.number().optional(),
204
205
  'Stage Name': z.string().optional(),
205
206
  Details: z.string().optional(),
206
207
  'Submission Time': z.number().optional(),
@@ -311,6 +312,7 @@ export const TaskEndEventSchema = z.object({
311
312
  }).optional(),
312
313
  'Output Metrics': z.object({
313
314
  'Bytes Written': z.number().optional(),
315
+ 'Records Written': z.number().optional(),
314
316
  }).optional(),
315
317
  }).optional(),
316
318
  });
@@ -8,24 +8,25 @@ import {
8
8
  } from './threshold-overrides.js';
9
9
  import { getThresholdSummary } from './threshold-summary.js';
10
10
  import {
11
- typeTag, formatBytes, formatCores, formatDuration, formatRawWaste, formatWallClockRange, IMPACT_BAND_ORDER, readsAsZero,
11
+ typeTag, formatBytes, formatCores, formatDuration, formatWallClockRange, IMPACT_BAND_ORDER,
12
12
  } from './format-utils.js';
13
13
  import { findingName, titleCase } from './finding-names.js';
14
14
  import { redactReport, redactRunModel } from './redact.js';
15
15
  import { formatTaskFailureHeadline, } from './task-failure.js';
16
16
  import { findingActionLabel } from './finding-action-label.js';
17
17
  import { matchesFindingFilterCriteria, singleStageId } from './finding-filter-predicate.js';
18
- import { buildRecommendationRollup, isEligible, isRealFinding, rankFindings, } from './recommendation-rollup.js';
18
+ import { buildRecommendationRollup, isEligible, isRealFinding, rankFindings, resourceGroupTotal, } from './recommendation-rollup.js';
19
19
  import { checkCoverage, isCleanRun } from './check-coverage.js';
20
20
  import { buildRunVerdict, stepCopyRecommendation, stepCopyText, } from './run-verdict.js';
21
21
  import {
22
22
  estimateProvenance, impactEstimateFigure, impactFigure, rawWasteMeaning, savingsMeaning,
23
23
  } from './impact-format.js';
24
24
  import { computeRunShape, } from './run-shape.js';
25
+ import { extractWriteTargets, } from './write-targets.js';
25
26
  import { detectorInfoByType } from './detector-docs.js';
26
27
 
27
28
 
28
-
29
+
29
30
 
30
31
 
31
32
  export const EVIDENCE_SCHEMA_VERSION = 5;
@@ -43,9 +44,11 @@ export const EVIDENCE_SCHEMA_VERSION = 5;
43
44
 
44
45
 
45
46
 
46
-
47
-
48
-
47
+
48
+
49
+
50
+
51
+
49
52
 
50
53
 
51
54
 
@@ -165,6 +168,8 @@ export const EVIDENCE_SCHEMA_VERSION = 5;
165
168
 
166
169
 
167
170
 
171
+
172
+
168
173
 
169
174
 
170
175
 
@@ -244,6 +249,7 @@ function findingRow(f ) {
244
249
  recommendation: f.recommendation ?? null,
245
250
  detectorVersion: f.detectorVersion ?? 1,
246
251
  evidence: projectEvidence(f),
252
+ remediation: f.remediation ?? [],
247
253
  actionLabel: findingActionLabel(f),
248
254
  } ;
249
255
  // Threshold/confidence provenance, only when the detector emitted it.
@@ -311,15 +317,14 @@ function buildRecommendations(
311
317
  };
312
318
  }
313
319
  if (group.kind === 'resource') {
314
- const text = formatRawWaste({ value: group.total, unit: group.unit });
315
- const shown = readsAsZero(text) ? null : text;
320
+ const shown = resourceGroupTotal(group);
316
321
  return {
317
322
  ...base,
318
323
  kind: 'resource',
319
324
  unit: group.unit,
320
325
  total: group.total,
321
326
  impact: shown,
322
- impactMeaning: shown ? rawWasteMeaning(group.unit) : null,
327
+ impactMeaning: shown ? rawWasteMeaning(representative.impactEstimate?.rawWaste) : null,
323
328
  };
324
329
  }
325
330
  return {
@@ -495,6 +500,9 @@ function buildJson(
495
500
  // Order follows DETECTORS (stable) => byte-stable serialization.
496
501
  detectors: tunedDetectorCatalog(thresholds),
497
502
  findings: rows,
503
+ writeTargets: extractWriteTargets(sql ?? new Map(), {
504
+ skippedLines: appModel.skippedLines, unreadableSqlExecutions: appModel.unreadableSqlExecutions,
505
+ }),
498
506
  recommendations,
499
507
  cleanChecks,
500
508
  notRunChecks,
@@ -581,7 +589,7 @@ function renderVerdict(verdict ) {
581
589
  lines.push(`${i + 1}. [${step.tag}] ${step.text}`);
582
590
  if (step.relatedTypes.length > 0) {
583
591
  const related = step.relatedTypes.map(findingName).join(', ');
584
- lines.push(` - Also flagged here: ${related}. These often share this cause, so the same fix may clear them too.`);
592
+ lines.push(` - Also flagged here, likely the same cause: ${related}.`);
585
593
  }
586
594
  });
587
595
  if (verdict.remainingPlaces > 0) {
@@ -1,5 +1,6 @@
1
1
  import { presentationOf } from './finding-presentation.js';
2
-
2
+ import { pathBasename } from './format-utils.js';
3
+
3
4
 
4
5
  /** A generic, type-level recommendation sentence for a finding: the shape of the fix, with no
5
6
  * instance data (numbers, stage ids, host names, file counts, config values), from its type's
@@ -12,3 +13,21 @@ import { presentationOf } from './finding-presentation.js';
12
13
  export function coreFindingGenericRecommendation(finding ) {
13
14
  return presentationOf(finding.type)?.genericRecommendation(finding);
14
15
  }
16
+
17
+ /** What a duplicatePlanSubtree finding measured, for the detector's recommendation and the
18
+ * Redundant Plan Subtree row, which shows it beside the fix its card states once. */
19
+ export function duplicateSubtreeDetail(f ) {
20
+ const touching = f.sampleRelation ? ` (touching ${f.sampleRelation})` : '';
21
+ return `A ${f.subtreeSize}-node subtree rooted at ${pathBasename(f.rootName)} repeats ${f.value}x in this plan${touching}`;
22
+ }
23
+
24
+ export const DUPLICATE_SUBTREE_DIFFERING_NOTE = 'Their filters, columns or scanned tables differ, so the repeats may compute different data.';
25
+
26
+ /** Reader-facing names for slowHost's per-executor dimensions, for the detector's recommendation
27
+ * and the Slow Executor Host row. */
28
+ export const SLOW_HOST_DIMENSION_LABEL = {
29
+ taskTime: 'task time',
30
+ inputBytes: 'input read',
31
+ shuffleBytes: 'shuffle read and write',
32
+ storageMemory: 'storage memory',
33
+ };
@@ -25,3 +25,10 @@ export function recommendationText(finding ) {
25
25
  if (text) return text;
26
26
  return findingName(finding.type);
27
27
  }
28
+
29
+ /** A recommendation's two halves: detectors write "<measurement>: <fix>", split at the last ": "
30
+ * since a measurement can quote an error with colons. Text with no split has no `measured`. */
31
+ export function recommendationParts(text ) {
32
+ const at = text.lastIndexOf(': ');
33
+ return at > 0 ? { measured: text.slice(0, at), fix: text.slice(at + 2) } : { measured: null, fix: text };
34
+ }