sparkforensics-mcp 0.3.0 → 0.5.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (92) hide show
  1. package/bin/sparkforensics-mcp.mjs +41 -10
  2. package/package.json +1 -1
  3. package/vendor-core/allocation.js +106 -0
  4. package/vendor-core/analyzer.js +168 -60
  5. package/vendor-core/check-coverage.js +88 -0
  6. package/vendor-core/cli/budgets.js +54 -27
  7. package/vendor-core/cli/collect-run.js +84 -32
  8. package/vendor-core/cli/regression-budgets.js +83 -0
  9. package/vendor-core/cli/threshold-config.js +28 -0
  10. package/vendor-core/comparison-verdict.js +177 -0
  11. package/vendor-core/core-source-hash.txt +1 -0
  12. package/vendor-core/core-usage-locality.js +56 -2
  13. package/vendor-core/detector-docs.js +58 -0
  14. package/vendor-core/detectors.js +1094 -500
  15. package/vendor-core/docs-config.js +0 -36
  16. package/vendor-core/docs-content/chapters/nav-index.json +31 -0
  17. package/vendor-core/docs-content/detection/cache.md +3 -2
  18. package/vendor-core/docs-content/detection/cfg.md +9 -8
  19. package/vendor-core/docs-content/detection/chrn.md +1 -2
  20. package/vendor-core/docs-content/detection/cold.md +4 -2
  21. package/vendor-core/docs-content/detection/cstor.md +9 -0
  22. package/vendor-core/docs-content/detection/fail.md +3 -2
  23. package/vendor-core/docs-content/detection/gc.md +3 -2
  24. package/vendor-core/docs-content/detection/host.md +2 -1
  25. package/vendor-core/docs-content/detection/local.md +1 -1
  26. package/vendor-core/docs-content/detection/mem.md +5 -2
  27. package/vendor-core/docs-content/detection/plan.md +2 -1
  28. package/vendor-core/docs-content/detection/sfail.md +2 -1
  29. package/vendor-core/docs-content/detection/shape.md +5 -4
  30. package/vendor-core/docs-content/detection/skew.md +3 -1
  31. package/vendor-core/docs-content/detection/slow.md +2 -2
  32. package/vendor-core/docs-content/detection/spec.md +2 -3
  33. package/vendor-core/docs-content/detection/spill.md +1 -1
  34. package/vendor-core/docs-site-config.js +3 -0
  35. package/vendor-core/effective-conf.js +107 -0
  36. package/vendor-core/efficiency-model.js +8 -6
  37. package/vendor-core/event-handlers.js +321 -44
  38. package/vendor-core/event-schemas.js +23 -0
  39. package/vendor-core/evidence-report.js +432 -115
  40. package/vendor-core/export-data.js +79 -6
  41. package/vendor-core/finding-action-label.js +9 -88
  42. package/vendor-core/finding-filter-predicate.js +9 -0
  43. package/vendor-core/finding-generic-recommendation.js +26 -105
  44. package/vendor-core/finding-names.js +28 -45
  45. package/vendor-core/finding-presentation.js +368 -0
  46. package/vendor-core/finding-tag-help.js +110 -0
  47. package/vendor-core/finding-types.js +373 -0
  48. package/vendor-core/findings-of-type.js +11 -0
  49. package/vendor-core/format-utils.js +96 -30
  50. package/vendor-core/html-export.js +51 -0
  51. package/vendor-core/impact-band.js +21 -8
  52. package/vendor-core/impact-estimator.js +25 -520
  53. package/vendor-core/impact-format.js +115 -0
  54. package/vendor-core/impact-model.js +197 -0
  55. package/vendor-core/ingest.js +6 -2
  56. package/vendor-core/intervals.js +13 -0
  57. package/vendor-core/list-runs.js +7 -5
  58. package/vendor-core/load-vendored.js +70 -5
  59. package/vendor-core/mcp-server-factory.js +14 -10
  60. package/vendor-core/mcp-tools.js +105 -45
  61. package/vendor-core/model-assembler.js +35 -1
  62. package/vendor-core/occupancy.js +1 -1
  63. package/vendor-core/parser-worker.js +2 -2
  64. package/vendor-core/plan-graph-model.js +3 -2
  65. package/vendor-core/plan-node-detail.js +1 -1
  66. package/vendor-core/proxy.js +3 -1
  67. package/vendor-core/python-stage.js +25 -0
  68. package/vendor-core/recommendation-rollup.js +70 -3
  69. package/vendor-core/redact.js +96 -37
  70. package/vendor-core/remediation.js +20 -0
  71. package/vendor-core/run-comparison.js +73 -29
  72. package/vendor-core/run-interpretation.js +291 -0
  73. package/vendor-core/run-metrics.js +198 -0
  74. package/vendor-core/run-outcome.js +74 -0
  75. package/vendor-core/run-payload.js +17 -0
  76. package/vendor-core/run-shape.js +40 -0
  77. package/vendor-core/run-totals.js +24 -0
  78. package/vendor-core/run-verdict.js +352 -0
  79. package/vendor-core/scaling-sim.js +4 -5
  80. package/vendor-core/scorecard-estimates.js +63 -0
  81. package/vendor-core/session-snapshot.js +7 -0
  82. package/vendor-core/shs-schemas.js +2 -2
  83. package/vendor-core/spark-memory.js +17 -0
  84. package/vendor-core/sql-stages.js +11 -0
  85. package/vendor-core/stage-plan-nodes.js +18 -0
  86. package/vendor-core/stage-quantiles.js +6 -0
  87. package/vendor-core/threshold-overrides.js +160 -0
  88. package/vendor-core/threshold-summary.js +11 -33
  89. package/vendor-core/types.js +54 -42
  90. package/vendor-core/wall-clock.js +1 -12
  91. package/vendor-core/wasted-core-hours.js +12 -9
  92. package/vendor-core/write-targets.js +312 -0
@@ -1,528 +1,33 @@
1
-
2
- import { nsToMs } from './format-utils.js';
3
- import {
4
- computeOccupancy, estimateSingleStage, estimateMultiStage, tailRecoveryMs, tailRemovedWorkMs, stragglerFixLongestTaskMs,
5
-
6
- } from './occupancy.js';
1
+
2
+ import { ENTRY_BY_TYPE } from './detectors.js';
3
+ import { coreTimeFor, } from './impact-model.js';
7
4
 
8
- // Assumed shuffle-network throughput per executor link, ~1 Gbps. Starting assumption, unvalidated.
9
- const SHUFFLE_THROUGHPUT_BPS = 125_000_000;
10
- // Assumed disk I/O throughput per executor for spilled data, ~200 MB/s (conservative HDD/SSD blend).
11
- const SPILL_IO_THROUGHPUT_BPS = 200_000_000;
5
+
12
6
 
13
- // A stage's own per-task overhead: task wall time (launch to finish, summed per executor by
14
- // finalizeStage) minus executorRunTime, i.e. deserialization, result serialization and the
15
- // launch round trip that coalescing tasks removes. `concurrency` is the stage's achieved task
16
- // concurrency (task time / stage duration). Null when the stage has no such data or no overhead.
17
- function measuredTaskOverhead(stage ) {
18
- const executorStats = Array.isArray(stage.executorStats)
19
- ? (stage.executorStats ) : [];
20
- const taskTimeMs = executorStats.reduce((sum, e) => sum + (e.totalDuration ?? 0), 0);
21
- const taskCount = stage.taskCount ?? 0;
22
- const durationMs = (stage.completedAt ?? 0) - (stage.submittedAt ?? 0);
23
- const overheadMs = taskTimeMs - (stage.executorRunTime ?? 0);
24
- if (taskTimeMs <= 0 || taskCount <= 0 || durationMs <= 0 || overheadMs <= 0) return null;
25
- return { perTaskMs: overheadMs / taskCount, concurrency: taskTimeMs / durationMs };
26
- }
27
-
28
- // Both constants above are single-device figures (one NIC, one local disk). A stage's shuffle
29
- // reads and spills are spread over every executor that ran its tasks, each moving its own share
30
- // in parallel, so the stage's aggregate bandwidth scales with that executor count. Dividing a
31
- // stage-wide byte total by one device's bandwidth modeled the whole cluster as one link: on 14
32
- // real logs that claimed up to 1,870s of shuffle time on stages whose tasks measured ~0s of
33
- // shuffle fetch wait (222 of 284 shuffle findings). No executor data falls back to one device.
34
- function stageIoParallelism(stage ) {
35
- const executors = Array.isArray(stage.executorStats) ? stage.executorStats.length : 0;
36
- return Math.max(1, executors);
37
- }
38
-
39
- // Wall-clock the stage's tasks spent blocked fetching shuffle blocks: fetchWaitTime is a
40
- // cross-task sum like executorRunTime, so dividing by the stage's average concurrency converts it
41
- // (the gc estimate's conversion). Null when the stage has no run time or duration to convert with.
42
- // On 14 real logs the link model claimed 2780s over 284 shuffle findings, 256s once capped at
43
- // this, with about zero fetch wait on 5 of the 10 non-info ones: reads that overlapped compute
44
- // stalled nothing.
45
- function fetchWaitWallClockMs(stage ) {
46
- const fetchWaitMs = stage.fetchWaitTime;
47
- const runTimeMs = stage.executorRunTime ?? 0;
48
- const durationMs = (stage.completedAt ?? 0) - (stage.submittedAt ?? 0);
49
- if (typeof fetchWaitMs !== 'number' || runTimeMs <= 0 || durationMs <= 0) return null;
50
- return fetchWaitMs / (runTimeMs / durationMs);
51
- }
52
-
53
- // Wall-clock a stage's wasted (retried) attempts cost it. Each one delayed only its own task, and
54
- // attempts of different tasks ran side by side: one lost executor fails every task it was running
55
- // at once (4 wasted attempts of 36.6s each, all first attempts, on a real stage that ran 41 tasks
56
- // at once, claimed as 146.6s). So the stage lost at most its longest retry chain, the most attempts
57
- // one task wasted (its highest attempt number + 1) times the mean wasted attempt, or their summed
58
- // time spread over its slots, whichever is larger. Without a sample of every wasted attempt
59
- // (retryTaskSamples is capped), the chain isn't known: the summed time.
60
- function retryWallClockMs(stage ) {
61
- const totalMs = (stage.retryWasteMs ) ?? 0;
62
- const attempts = (stage.wastedAttempts ) ?? 0;
63
- const samples = Array.isArray(stage.retryTaskSamples)
64
- ? (stage.retryTaskSamples ) : [];
65
- if (totalMs <= 0 || attempts <= 0 || samples.length < attempts) return totalMs;
66
- const chain = samples.reduce((longest, s) => Math.max(longest, (s.attemptNumber ?? 0) + 1), 1);
67
- const slots = Math.max(1, stage.peakConcurrentTasks ?? 1);
68
- return Math.min(totalMs, Math.max((chain * totalMs) / attempts, totalMs / slots));
69
- }
70
- // Spark's classic recommended shuffle partition size.
71
- const IDEAL_BYTES_PER_PARTITION_TASK = 128 * 1024 * 1024;
72
- // Assumed per-task scheduling/launch overhead: the fallback when a stage lacks the per-executor
73
- // task-time sums measuredTaskOverhead needs.
74
- const TASK_SCHEDULING_OVERHEAD_MS = 50;
75
- // Assumed per-file open latency (small-file overhead).
76
- const FILE_OPEN_OVERHEAD_MS = 10;
77
- // Assumed broadcast-transfer bandwidth, shared with overBroadcast/underBroadcast.
78
- const BROADCAST_BANDWIDTH_BPS = 125_000_000;
79
- // Assumed per-non-local-task network-fetch penalty, reported as extra core-time.
80
- const NETWORK_FETCH_PENALTY_MS = 20;
81
- // Assumed executor JVM+container startup overhead.
82
- const EXECUTOR_STARTUP_OVERHEAD_MS = 15000;
83
- // Assumed re-read throughput, shared by cachingOpportunity and cacheUtilization.
84
- const RE_READ_THROUGHPUT_BPS = 125_000_000;
85
-
86
- // Below this share of executorRunTime spent on CPU, a stage's tasks were idle, waiting on something
87
- // outside Spark: on 14 real logs (2026-09-23) every non-Python stage under 1% was a JDBC read, a
88
- // file listing or a Delta log read, while file writes, which more partitions do parallelize,
89
- // start at 2%.
90
- const IDLE_CPU_SHARE_MAX = 0.01;
91
-
92
- // True when the stage's tasks spent under IDLE_CPU_SHARE_MAX of their run time on CPU. False when
93
- // the share can't be trusted: no CPU time recorded (older Spark logs omit the metric), or Python
94
- // code run through PythonRDD, whose worker-process CPU executorCpuTime (the JVM task thread's)
95
- // never counts (such stages read 0.1% on the same logs while computing).
96
- function tasksMostlyIdle(stage ) {
97
- const runMs = stage.executorRunTime ?? 0;
98
- const cpuMs = nsToMs(stage.executorCpuTime ?? 0);
99
- if (runMs <= 0 || cpuMs <= 0) return false;
100
- if (/PythonRDD/.test(stage.name ?? '') || /org\.apache\.spark\.api\.python\./.test(stage.details ?? '')) return false;
101
- return cpuMs / runMs < IDLE_CPU_SHARE_MAX;
102
- }
103
-
104
- // skew and straggler claim time off the stage's longest task itself, so the occupancy clip must
105
- // not floor them at that same task (see estimateSingleStage). detectors.ts's clippedWasteMs gates
106
- // both detectors on the same option so the firing floor and the displayed estimate agree.
107
- const TAIL_CLAIM = { shortensLongestTask: true };
108
-
109
- // No quantifiable magnitude -> 'informational'; a rawWaste figure with no stage window ->
110
- // 'resourceOnly'. Never a fake {low:0, high:0}: wallClock is null in both cases.
111
- function costOnly(estimateMethod , rawWaste ) {
112
- return rawWaste
113
- ? { basis: 'resourceOnly', wallClock: null, estimateMethod, rawWaste }
114
- : { basis: 'informational', wallClock: null, estimateMethod };
115
- }
116
-
117
- function singleStageImpact(
118
- wasteMs ,
119
- stageId ,
120
- stages ,
121
- occupancy ,
122
- estimateMethod ,
123
- rawWaste ,
124
- opts ,
125
- ) {
126
- const est = estimateSingleStage(wasteMs, stageId, stages , occupancy, opts);
127
- if (est) return { basis: est.basis, wallClock: est.wallClock, estimateMethod, rawWaste };
128
- return costOnly(estimateMethod, rawWaste); // stage excluded from the sweep (duration <= 0)
129
- }
130
-
131
- function stageMappableWasteOrCostOnly(
132
- wasteMs ,
133
- stageIds ,
134
- stages ,
135
- occupancy ,
136
- ) {
137
- const rawWaste = wasteMs > 0 ? { value: wasteMs, unit: 'ms' } : undefined;
138
- if (!stageIds || stageIds.length === 0) {
139
- return costOnly('modeled', rawWaste);
7
+ /** Attaches each finding's `impactEstimate` from its entry's estimate(), in place. A null estimate
8
+ * leaves the finding uncovered (no impactEstimate). `ctx` is analyze()'s one occupancy sweep. */
9
+ export function estimateImpact(findings , ctx ) {
10
+ for (const f of findings) {
11
+ const estimate = ENTRY_BY_TYPE.get(f.type)?.estimate(f, ctx);
12
+ if (!estimate) continue;
13
+ f.impactEstimate = { ...estimate, coreTimeMs: coreTimeFor(f, estimate) };
140
14
  }
141
- // One waste event spread over a span of stages, not N independent wastes: apportion evenly so
142
- // estimateMultiStage's union cap doesn't absorb the same amount claimed once per stage.
143
- const perStageWasteMs = wasteMs / stageIds.length;
144
- const wasteMsByStage = new Map(stageIds.map((id) => [id, perStageWasteMs]));
145
- const est = estimateMultiStage(stageIds, wasteMsByStage, stages , occupancy);
146
- if (!est) return costOnly('modeled', rawWaste); // every stage excluded from the sweep
147
- return { basis: est.basis, wallClock: est.wallClock, estimateMethod: 'modeled', rawWaste };
15
+ countTailCoreTimeOnce(findings);
16
+ return findings;
148
17
  }
149
18
 
150
- /** Per-finding-type dispatch. A type with no case stays uncovered (no impactEstimate attached);
151
- * docs/architecture.md's Impact estimation table is the authoritative completeness check, not this switch. */
152
- function computeEstimateForFinding(
153
- finding ,
154
- stages ,
155
- occupancy ,
156
- totalCores ,
157
- ) {
158
- switch (finding.type) {
159
- case 'retryWaste': {
160
- // The waste figure lives on the Stage, not the Finding: the detector only re-publishes it as metric/value.
161
- if (finding.stageId == null) return null;
162
- const stage = stages.get(finding.stageId);
163
- if (!stage) return null;
164
- const wasteMs = (stage.retryWasteMs ) ?? 0;
165
- const wallClockMs = retryWallClockMs(stage);
166
- return singleStageImpact(wallClockMs, finding.stageId, stages, occupancy,
167
- wallClockMs === wasteMs ? 'measured' : 'modeled', { value: wasteMs, unit: 'ms' });
168
- }
169
- case 'speculationWaste': {
170
- if (finding.stageId == null) return null;
171
- const stage = stages.get(finding.stageId);
172
- if (!stage) return null;
173
- const wasteMs = (stage.speculationWasteMs ) ?? 0;
174
- return singleStageImpact(wasteMs, finding.stageId, stages, occupancy, 'measured', { value: wasteMs, unit: 'ms' });
175
- }
176
- case 'coldStart': {
177
- // The detector reports the gap as `metric: 'startupGapSeconds', value: <seconds>`.
178
- if (typeof finding.value !== 'number') return null;
179
- const wasteMs = finding.value * 1000;
180
- // Time before any task starts can never overlap any stage; a genuine unclipped point estimate,
181
- // not tied to any stage's gate (coldStart is app-scoped, stageId: null).
182
- return { basis: 'serial', wallClock: { low: wasteMs, high: wasteMs }, estimateMethod: 'measured' };
183
- }
184
- case 'gc': {
185
- // The low-GC branch is an over-provisioning signal whose fix (less executor memory) raises
186
- // GC rather than recovering it: the stage's GC time is no saving there, so no waste model.
187
- if (finding.direction === 'low') return costOnly('none');
188
- if (finding.stageId == null) return null;
189
- const stage = stages.get(finding.stageId);
190
- if (!stage) return null;
191
- const stageDurationMs = (stage.completedAt ?? 0) - (stage.submittedAt ?? 0);
192
- const executorRunTime = stage.executorRunTime ?? 0;
193
- const jvmGCTime = stage.jvmGCTime ?? 0;
194
- // The raw cross-task core-time sum, before any conversion: the one figure here
195
- // that is straight from the log rather than modeled.
196
- const rawWaste = { value: jvmGCTime, unit: 'coreMs' } ;
197
- if (executorRunTime <= 0 || stageDurationMs <= 0) {
198
- return costOnly('modeled', rawWaste);
199
- }
200
- const avgConcurrency = executorRunTime / stageDurationMs;
201
- // jvmGCTime is a cross-task core-time sum (same shape as executorRunTime); dividing by the
202
- // stage's average concurrency converts it to an approximate wall-clock figure. Modeled, not exact.
203
- const wasteMs = jvmGCTime / avgConcurrency;
204
- return singleStageImpact(wasteMs, finding.stageId, stages, occupancy, 'modeled', rawWaste);
205
- }
206
- case 'skew': {
207
- if (finding.stageId == null) return null;
208
- const stage = stages.get(finding.stageId);
209
- if (!stage) return null;
210
- const p50 = stage.taskDurationP50 ?? 0;
211
- // computeSkewRatio's own metric labels (src/detectors.ts): 'P95/median' or 'max/median'.
212
- const usesP95Branch = finding.metric === 'P95/median';
213
- const singleDelta = Math.max(0, usesP95Branch ? (stage.taskDurationP95 ?? 0) - p50 : (stage.taskDurationMax ?? 0) - p50);
214
- const wasteMs = tailRecoveryMs(stage, singleDelta);
215
- // Fixing the skew still waits on the longest task it leaves, as for straggler.
216
- return singleStageImpact(wasteMs, finding.stageId, stages, occupancy, 'measured', { value: wasteMs, unit: 'ms' },
217
- { ...TAIL_CLAIM, removedCoreWorkMs: tailRemovedWorkMs(stage, singleDelta), longestTaskAfterFixMs: stragglerFixLongestTaskMs(stage) });
218
- }
219
- case 'straggler':
220
- case 'stageShape': {
221
- // Shared case for two finding types. straggler (no `rule` field) falls through to the max-P50
222
- // computation below; every stageShape rule returns early (not `break`, which would fall off
223
- // the switch and return undefined instead of null since the switch is the function's last statement).
224
- if (finding.type === 'stageShape') {
225
- if (finding.rule === 'lowParallelism') {
226
- const stage = stages.get(finding.stageId );
227
- if (!stage) return null;
228
- const stageDurationMs = (stage.completedAt ?? 0) - (stage.submittedAt ?? 0);
229
- const idleCoreMs =
230
- Math.max(0, ((finding.totalCores ) ?? 0) - (stage.taskCount ?? 0)) * stageDurationMs;
231
- // Real per-stage data (cores, task count, duration), no assumed constant.
232
- return costOnly('measured', { value: idleCoreMs, unit: 'coreMs' });
233
- }
234
- if (finding.rule === 'dataExplosion') {
235
- const stage = stages.get(finding.stageId );
236
- if (!stage) return null;
237
- const excessBytes = Math.max(0, (stage.outputBytes ?? 0) - (stage.inputBytes ?? 0));
238
- // Measured input/output byte counts, no assumed constant.
239
- return costOnly('measured', { value: excessBytes, unit: 'bytes' });
240
- }
241
- if (finding.rule === 'taskStageSkew') {
242
- const stage = stages.get(finding.stageId );
243
- if (!stage) return null;
244
- const totalCores = (finding.totalCores ) ?? 0;
245
- const taskCount = stage.taskCount ?? 0;
246
- // Cores idle during the straggler's tail, at achieved concurrency (not full cluster
247
- // capacity, which is lowParallelism's territory): this rule's trigger forces the
248
- // occupancy-clipped estimate to zero on every firing, so it's resourceOnly, not a wall-clock claim.
249
- const idleCoreMs =
250
- Math.max(0, Math.min(totalCores, taskCount) - 1) *
251
- Math.max(0, (stage.taskDurationMax ?? 0) - (stage.taskDurationP50 ?? 0));
252
- return costOnly('measured', { value: idleCoreMs, unit: 'coreMs' });
253
- }
254
- return null;
255
- }
256
- if (finding.stageId == null) return null;
257
- const stage = stages.get(finding.stageId);
258
- if (!stage) return null;
259
- const longestTaskAfterFixMs = stragglerFixLongestTaskMs(stage);
260
- const singleDelta = Math.max(0, (stage.taskDurationMax ?? 0) - longestTaskAfterFixMs);
261
- const wasteMs = tailRecoveryMs(stage, singleDelta);
262
- return singleStageImpact(wasteMs, finding.stageId, stages, occupancy, 'measured', { value: wasteMs, unit: 'ms' },
263
- { ...TAIL_CLAIM, removedCoreWorkMs: tailRemovedWorkMs(stage, singleDelta), longestTaskAfterFixMs });
264
- }
265
- case 'slowHost': {
266
- // Three duration-based shapes, each carrying its absolute-ms figure under a different field
267
- // (`value` is always a ratio/share, never ms): the per-host mean branch (discriminated by
268
- // `metric`), the duration-share branch (`variant`), and the multiDim taskTime dimension. Every
269
- // byte-based multiDim dimension has no absolute figure today, so it stays informational.
270
- const absoluteMs =
271
- finding.metric === 'hostMeanRatio' || finding.variant === 'durationShare'
272
- ? (finding.hostMeanMs )
273
- : finding.variant === 'multiDim' && finding.dimension === 'taskTime'
274
- ? (finding.execMaxValue )
275
- : null;
276
- if (absoluteMs == null) {
277
- return costOnly('none'); // byte-based multiDim dims: no absolute figure today, no model applied
278
- }
279
- if (finding.stageId == null) return null;
280
- const stage = stages.get(finding.stageId);
281
- if (!stage) return null;
282
- const wasteMs = Math.max(0, absoluteMs - (stage.taskDurationP50 ?? 0));
283
- return singleStageImpact(wasteMs, finding.stageId, stages, occupancy, 'measured', { value: wasteMs, unit: 'ms' });
284
- }
285
- case 'duplicatePlanSubtree': {
286
- const stageIds = finding.stageIds ;
287
- if (!stageIds || stageIds.length === 0) return null;
288
- // The detector reports subtreeOccurrences >= 2. Only repeats past the first are redundant:
289
- // computing the subtree once is real work, so waste is (occurrences-1)/occurrences of the stages' time.
290
- const occurrences = typeof finding.value === 'number' ? finding.value : 0;
291
- // Defensive only: minOccurrences guarantees occurrences >= 2 on real data; a malformed-value fallback.
292
- if (occurrences < 2) return costOnly('none');
293
- // Same-shaped repeats whose details differ compute different data: nothing is known to be
294
- // recomputed, so there is no time to claim.
295
- if (finding.occurrencesIdentical === false) return costOnly('none');
296
- // Each stage contributes the share of its operators inside the repeated subtree: a stage it
297
- // shares with other operators (the consuming join, the join's other side) isn't all its
298
- // time, and claiming whole stages let sibling groups claim the same stage twice. Findings
299
- // built without the field (hand-made fixtures) count every linked stage whole.
300
- const shares = (finding.stageShares ?? null) ;
301
- const redundantFraction = (occurrences - 1) / occurrences;
302
- const wasteMsByStage = new Map ();
303
- for (const id of stageIds) {
304
- const s = stages.get(id);
305
- const share = shares ? (shares[id] ?? 0) : 1;
306
- if (s && share > 0) {
307
- // Time with tasks running, not submit-to-complete: a stage left waiting for cores
308
- // (2491s open, 60s of tasks on a real log) isn't recomputing anything while it waits.
309
- const activeMs = s.taskActiveMs ?? Math.max(0, (s.completedAt ?? 0) - (s.submittedAt ?? 0));
310
- wasteMsByStage.set(id, activeMs * share * redundantFraction);
311
- }
312
- }
313
- // No operator of the subtree ran in a known stage: no time to attribute.
314
- if (wasteMsByStage.size === 0) return costOnly('none');
315
- const totalWasteMs = [...wasteMsByStage.values()].reduce((sum, ms) => sum + ms, 0);
316
- const rawWaste = { value: totalWasteMs, unit: 'ms' };
317
- const est = estimateMultiStage([...wasteMsByStage.keys()], wasteMsByStage, stages , occupancy);
318
- if (!est) return costOnly('measured', rawWaste);
319
- return { basis: est.basis, wallClock: est.wallClock, estimateMethod: 'measured', rawWaste };
320
- }
321
- case 'shuffle': {
322
- if (finding.stageId == null) return null;
323
- const stage = stages.get(finding.stageId);
324
- if (!stage) return null;
325
- const shuffleReadBytes = stage.shuffleReadBytes ?? 0;
326
- const modeledMs = (shuffleReadBytes / (SHUFFLE_THROUGHPUT_BPS * stageIoParallelism(stage))) * 1000;
327
- // The link model can't see whether the reads stalled the tasks: capped at the fetch wait the
328
- // tasks measured, the claim never exceeds what the stage spent blocked on the network.
329
- const measuredMs = fetchWaitWallClockMs(stage);
330
- const wasteMs = measuredMs == null ? modeledMs : Math.min(modeledMs, measuredMs);
331
- // rawWaste: the measured byte volume behind the modeled figure.
332
- return singleStageImpact(wasteMs, finding.stageId, stages, occupancy,
333
- measuredMs != null && measuredMs < modeledMs ? 'measured' : 'modeled', { value: shuffleReadBytes, unit: 'bytes' });
334
- }
335
- case 'spill': {
336
- if (finding.stageId == null) return null;
337
- const stage = stages.get(finding.stageId);
338
- if (!stage) return null;
339
- const diskBytesSpilled = stage.diskBytesSpilled ?? 0;
340
- const wasteMs = (diskBytesSpilled / (SPILL_IO_THROUGHPUT_BPS * stageIoParallelism(stage))) * 1000;
341
- // Surfaces the number the formula uses: the displayed metric is memoryBytesSpilled, but disk spill costs the I/O time.
342
- return singleStageImpact(wasteMs, finding.stageId, stages, occupancy, 'modeled', { value: diskBytesSpilled, unit: 'bytes' });
343
- }
344
- case 'stageSlowness': {
345
- if (finding.stageId == null) return null;
346
- const stage = stages.get(finding.stageId);
347
- if (!stage) return null;
348
- // The recommended fix is more partitions, which only helps a stage that ran fewer tasks than
349
- // the cluster has cores: the time its tasks were running could then spread over up to
350
- // totalCores (lowShuffleParallelism's shape). Time the stage sat open with no task running
351
- // is queueing no partition count recovers. Splitting partitions splits the longest task
352
- // too, hence TAIL_CLAIM's post-fix floor. Unknown cluster size: no defensible figure.
353
- if (totalCores <= 0) return costOnly('modeled');
354
- // A stage that read no input and no shuffle, its tasks idle waiting on an external system,
355
- // gains nothing from more partitions (a 1-task JDBC count stage open 27 minutes on 5s of CPU
356
- // was claimed 99% recoverable): claim 0. Stages that read bytes keep their claim.
357
- const readBytes = (stage.inputBytes ?? 0) + (stage.shuffleReadBytes ?? 0);
358
- const activeMs = typeof stage.taskActiveMs === 'number'
359
- ? stage.taskActiveMs
360
- : Math.max(0, (stage.completedAt ?? 0) - (stage.submittedAt ?? 0));
361
- const taskCount = stage.taskCount ?? 0;
362
- const wasteMs = readBytes <= 0 && tasksMostlyIdle(stage) ? 0 : activeMs * Math.max(0, 1 - taskCount / totalCores);
363
- return singleStageImpact(wasteMs, finding.stageId, stages, occupancy, 'modeled', { value: wasteMs, unit: 'ms' }, TAIL_CLAIM);
364
- }
365
- case 'partitionSizing': {
366
- if (finding.stageId == null) return null;
367
- const stage = stages.get(finding.stageId);
368
- if (!stage) return null;
369
- let wasteMs = 0;
370
- if (finding.rule === 'maxPartitionTooBig') {
371
- wasteMs = ((stage.shuffleReadMax ?? 0) / SHUFFLE_THROUGHPUT_BPS) * 1000;
372
- } else if (finding.rule === 'shufflePartitionSkew') {
373
- const delta = Math.max(0, (stage.shuffleReadMax ?? 0) - (stage.shuffleReadP50 ?? 0));
374
- wasteMs = (delta / SHUFFLE_THROUGHPUT_BPS) * 1000;
375
- } else if (finding.rule === 'lowShuffleParallelism') {
376
- const targetTaskCount = Math.ceil((stage.shuffleReadBytes ?? 0) / IDEAL_BYTES_PER_PARTITION_TASK);
377
- const taskCount = stage.taskCount ?? 0;
378
- if (targetTaskCount > taskCount && taskCount > 0) {
379
- const stageDurationMs = Math.max(0, (stage.completedAt ?? 0) - (stage.submittedAt ?? 0));
380
- // Too few shuffle partitions means each task processes more than the ideal bytes,
381
- // serializing work more partitions would run concurrently: the waste is that serialized
382
- // work, not the scheduling cost of tasks you'd add (adding tasks incurs overhead, recovers
383
- // nothing). Model the achievable duration at target parallelism by scaling down proportionally.
384
- wasteMs = stageDurationMs * (1 - taskCount / targetTaskCount);
385
- }
386
- } else {
387
- return null;
388
- }
389
- return singleStageImpact(wasteMs, finding.stageId, stages, occupancy, 'modeled', { value: wasteMs, unit: 'ms' });
390
- }
391
- case 'tinyTask': {
392
- if (finding.stageId == null) return null;
393
- const stage = stages.get(finding.stageId);
394
- if (!stage) return null;
395
- const taskCount = stage.taskCount ?? 0;
396
- const excessTaskCount = Math.max(0, taskCount - Math.round(taskCount / 10));
397
- const measured = measuredTaskOverhead(stage);
398
- if (measured) {
399
- // Coalescing to a tenth of the tasks removes the excess tasks' per-task overhead: core
400
- // time spent in parallel, so wall-clock at the stage's achieved concurrency (floored at
401
- // 1: a mostly-idle stage can't save more wall-clock than the task time it removes).
402
- const wasteMs = (excessTaskCount * measured.perTaskMs) / Math.max(1, measured.concurrency);
403
- return singleStageImpact(wasteMs, finding.stageId, stages, occupancy, 'measured', { value: wasteMs, unit: 'ms' });
404
- }
405
- const wasteMs = excessTaskCount * TASK_SCHEDULING_OVERHEAD_MS;
406
- return singleStageImpact(wasteMs, finding.stageId, stages, occupancy, 'modeled', { value: wasteMs, unit: 'ms' });
407
- }
408
- case 'smallFiles': {
409
- const fileMs = ((finding.fileCount ) ?? 0) * FILE_OPEN_OVERHEAD_MS;
410
- // A read's files are opened by the scan's tasks, in parallel: spread the per-file cost over
411
- // the most tasks its stages ran at once (91344 files x 10ms is 913s, claimed against a 117s
412
- // stage that ran 314 tasks at once). A write keeps the serial sum: the job commit moves each
413
- // output file on the driver, one after another.
414
- const stageIds = finding.stageIds ;
415
- let slots = 1;
416
- if (finding.direction === 'read') {
417
- for (const id of stageIds ?? []) slots = Math.max(slots, stages.get(id)?.peakConcurrentTasks ?? 1);
418
- }
419
- return stageMappableWasteOrCostOnly(fileMs / slots, stageIds, stages, occupancy);
420
- }
421
- case 'overBroadcast': {
422
- // metric: 'broadcastBytes', value: <bytes>.
423
- const wasteMs = (((finding.value ) ?? 0) / BROADCAST_BANDWIDTH_BPS) * 1000;
424
- return stageMappableWasteOrCostOnly(wasteMs, finding.stageIds , stages, occupancy);
425
- }
426
- case 'underBroadcast': {
427
- // metric: 'smallerSideBytes', value: <bytes of the smaller join side>.
428
- const wasteMs = (((finding.value ) ?? 0) / BROADCAST_BANDWIDTH_BPS) * 1000;
429
- return stageMappableWasteOrCostOnly(wasteMs, finding.stageIds , stages, occupancy);
430
- }
431
- case 'memoryUtilization': {
432
- // The wasteModel variant reports metric: 'wastedMBSeconds', value: <MB-seconds>.
433
- if (finding.variant === 'wasteModel' && typeof finding.value === 'number') {
434
- return costOnly('measured', { value: finding.value, unit: 'mbSeconds' });
435
- }
436
- if (finding.variant === 'idleCores') {
437
- // Idle core-time priced as memory held but unused: the same MB-seconds unit as wasteModel, so comparable.
438
- const idleRateFraction = finding.idleRateFraction ;
439
- const allocatedMB = finding.allocatedMB ;
440
- const peakExecutors = finding.peakExecutors ;
441
- const appDurationMs = finding.appDurationMs ;
442
- if (idleRateFraction != null && allocatedMB != null && peakExecutors != null && appDurationMs != null) {
443
- const wastedMBSeconds = idleRateFraction * allocatedMB * peakExecutors * (appDurationMs / 1000);
444
- return costOnly('modeled', { value: wastedMBSeconds, unit: 'mbSeconds' });
445
- }
446
- return costOnly('modeled');
447
- }
448
- // Only the over-provisioned band is a waste; the near-capacity band is an OOM-risk signal with
449
- // no magnitude, and the dataUnavailable shape has no inputs: both stay informational.
450
- if (finding.variant === 'memoryBand' && finding.rule === 'heapOverProvisioned') {
451
- const allocatedBytes = finding.allocatedBytes ;
452
- const heap = finding.heap ;
453
- const appDurationMs = finding.appDurationMs ;
454
- if (allocatedBytes != null && heap != null && appDurationMs != null) {
455
- const unusedMB = (allocatedBytes - heap) / (1024 * 1024);
456
- const wastedMBSeconds = unusedMB * (appDurationMs / 1000);
457
- return costOnly('modeled', { value: wastedMBSeconds, unit: 'mbSeconds' });
458
- }
459
- }
460
- return costOnly('modeled');
461
- }
462
- case 'utilization': {
463
- const fraction = finding.utilizationFraction ;
464
- const appDurationMs = finding.appDurationMs ;
465
- const totalCores = finding.totalCores ;
466
- if (fraction == null || appDurationMs == null || totalCores == null) {
467
- return costOnly('measured');
468
- }
469
- const idleCoreHours = (1 - fraction) * appDurationMs * totalCores / 3.6e6;
470
- return costOnly('measured', { value: idleCoreHours, unit: 'coreHours' });
471
- }
472
- case 'coreLocality': {
473
- const nonLocal = (finding.nonLocalTaskCount ) ?? 0;
474
- const coreMs = nonLocal * NETWORK_FETCH_PENALTY_MS;
475
- return costOnly('modeled', { value: coreMs, unit: 'coreMs' });
476
- }
477
- case 'autoscalingChurn': {
478
- const shortLived = (finding.shortLivedExecutorCount ) ?? 0;
479
- const executorHours = (shortLived * EXECUTOR_STARTUP_OVERHEAD_MS) / 3.6e6;
480
- return costOnly('modeled', { value: executorHours, unit: 'coreHours' });
481
- }
482
- case 'configAudit': {
483
- return costOnly('none'); // purely informational: no waste model applied
484
- }
485
- case 'jobFailureRate': {
486
- const failedJobs = (finding.failedJobs ) ?? 0;
487
- const avgJobDurationMs = (finding.avgJobDurationMs ) ?? 0;
488
- const coreHoursIsh = (failedJobs * avgJobDurationMs) / 3.6e6;
489
- return costOnly('modeled', { value: coreHoursIsh, unit: 'coreHours' });
490
- }
491
- case 'cachingOpportunity': {
492
- const totalReadBytes = (finding.totalReadBytes ) ?? 0;
493
- const wasteMs = (totalReadBytes / RE_READ_THROUGHPUT_BPS) * 1000;
494
- return costOnly('modeled', { value: wasteMs, unit: 'ms' });
495
- }
496
- case 'cacheUtilization': {
497
- const memorySize = (finding.memorySize ) ?? 0;
498
- const diskSize = (finding.diskSize ) ?? 0;
499
- const numCachedPartitions = (finding.numCachedPartitions ) ?? 0;
500
- const numPartitions = (finding.numPartitions ) ?? 0;
501
- const numUncachedPartitions = Math.max(0, numPartitions - numCachedPartitions);
502
- const cachedBytes = memorySize + diskSize;
503
- // Extrapolate never-cached partitions' size from the CACHED partitions' average (uncached/
504
- // cached, not uncached/total: numCachedPartitions produced cachedBytes). diskSize is added
505
- // once more: those bytes are cached but on disk, so re-reading them still costs I/O like an uncached partition.
506
- const uncachedBytes = numCachedPartitions > 0 ? (cachedBytes / numCachedPartitions) * numUncachedPartitions : 0;
507
- const uncachedOrSpilledBytes = uncachedBytes + diskSize;
508
- const wasteMs = (uncachedOrSpilledBytes / RE_READ_THROUGHPUT_BPS) * 1000;
509
- return costOnly('modeled', { value: wasteMs, unit: 'ms' });
510
- }
511
- case 'stageFailed':
512
- case 'failures':
513
- case 'incompleteRun': {
514
- return costOnly('none'); // purely informational: no waste model applied
515
- }
516
- default:
517
- return null;
518
- }
519
- }
19
+ // skew and straggler claim the same slow tail of a stage, so each reports its removed task time:
20
+ // summed over a stage's findings that would count the tail twice. skew keeps the figure; a
21
+ // straggler on the same stage carries null, since its tail is already counted there.
22
+ const TAIL_CORE_TIME_ORDER = ['skew', 'straggler'];
520
23
 
521
- export function estimateImpact(findings , stages , totalCores = 0) {
522
- const occupancy = computeOccupancy(stages , totalCores);
523
- for (const f of findings) {
524
- const estimate = computeEstimateForFinding(f, stages, occupancy, totalCores);
525
- if (estimate) f.impactEstimate = estimate;
24
+ function countTailCoreTimeOnce(findings ) {
25
+ const counted = new Set ();
26
+ const tails = findings
27
+ .filter((f) => f.stageId != null && f.impactEstimate?.coreTimeMs != null && TAIL_CORE_TIME_ORDER.includes(f.type))
28
+ .sort((a, b) => TAIL_CORE_TIME_ORDER.indexOf(a.type) - TAIL_CORE_TIME_ORDER.indexOf(b.type));
29
+ for (const f of tails) {
30
+ if (counted.has(f.stageId )) f.impactEstimate .coreTimeMs = null;
31
+ else counted.add(f.stageId );
526
32
  }
527
- return findings;
528
33
  }