sparkforensics-mcp 0.2.2 → 0.2.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/package.json +1 -1
- package/vendor-core/cli/collect-run.js +3 -2
- package/vendor-core/cli/native-zstd.js +351 -0
- package/vendor-core/detectors.js +243 -53
- package/vendor-core/docs-config.js +34 -8
- package/vendor-core/docs-content/chapters/03-memory-model.md +39 -0
- package/vendor-core/docs-content/chapters/11-cluster-config.md +40 -0
- package/vendor-core/docs-content/detection/gc.md +2 -0
- package/vendor-core/docs-content/detection/host.md +2 -1
- package/vendor-core/docs-content/detection/plan.md +3 -1
- package/vendor-core/docs-content/detection/shape.md +2 -1
- package/vendor-core/docs-content/detection/shfl.md +2 -1
- package/vendor-core/docs-content/detection/spill.md +1 -1
- package/vendor-core/docs-content/detection/strag.md +2 -1
- package/vendor-core/docs-content/detection/tiny.md +2 -1
- package/vendor-core/docs-content/tuning/failures.md +1 -1
- package/vendor-core/docs-content/tuning/gc.md +11 -4
- package/vendor-core/docs-content/tuning/shuffle.md +25 -5
- package/vendor-core/docs-content/tuning/skew.md +14 -6
- package/vendor-core/docs-content/tuning/small-files.md +12 -7
- package/vendor-core/docs-content/tuning/straggler.md +34 -0
- package/vendor-core/docs-content/tuning/tiny-tasks.md +1 -1
- package/vendor-core/docs-content/tuning/utilization.md +57 -7
- package/vendor-core/docs-content/upstream.json +4 -0
- package/vendor-core/event-handlers.js +321 -69
- package/vendor-core/event-schemas.js +8 -6
- package/vendor-core/evidence-report.js +3 -1
- package/vendor-core/impact-estimator.js +150 -34
- package/vendor-core/mcp-tools.js +20 -7
- package/vendor-core/occupancy.js +70 -2
- package/vendor-core/parser-worker.js +28 -12
- package/vendor-core/plan-summary.js +4 -0
- package/vendor-core/run-comparison.js +1 -15
- package/vendor-core/shs-fetch.js +18 -7
- package/vendor-core/shs-load.js +2 -1
- package/vendor-core/stage-quantiles.js +59 -3
- package/vendor-core/string-hash.js +15 -0
- package/vendor-core/types.js +9 -1
- package/vendor-core/vendor/fzstd.js +94 -18
|
@@ -1,14 +1,75 @@
|
|
|
1
1
|
|
|
2
|
-
import {
|
|
3
|
-
|
|
2
|
+
import {
|
|
3
|
+
computeOccupancy, estimateSingleStage, estimateMultiStage, tailRecoveryMs, tailRemovedWorkMs, stragglerFixLongestTaskMs,
|
|
4
|
+
|
|
5
|
+
} from './occupancy.js';
|
|
4
6
|
|
|
5
|
-
// Assumed shuffle-network throughput, ~1 Gbps. Starting assumption, unvalidated.
|
|
7
|
+
// Assumed shuffle-network throughput per executor link, ~1 Gbps. Starting assumption, unvalidated.
|
|
6
8
|
const SHUFFLE_THROUGHPUT_BPS = 125_000_000;
|
|
7
|
-
// Assumed disk I/O throughput for spilled data, ~200 MB/s (conservative HDD/SSD blend).
|
|
9
|
+
// Assumed disk I/O throughput per executor for spilled data, ~200 MB/s (conservative HDD/SSD blend).
|
|
8
10
|
const SPILL_IO_THROUGHPUT_BPS = 200_000_000;
|
|
11
|
+
|
|
12
|
+
// A stage's own per-task overhead: task wall time (launch to finish, summed per executor by
|
|
13
|
+
// finalizeStage) minus executorRunTime, i.e. deserialization, result serialization and the
|
|
14
|
+
// launch round trip that coalescing tasks removes. `concurrency` is the stage's achieved task
|
|
15
|
+
// concurrency (task time / stage duration). Null when the stage has no such data or no overhead.
|
|
16
|
+
function measuredTaskOverhead(stage ) {
|
|
17
|
+
const executorStats = Array.isArray(stage.executorStats)
|
|
18
|
+
? (stage.executorStats ) : [];
|
|
19
|
+
const taskTimeMs = executorStats.reduce((sum, e) => sum + (e.totalDuration ?? 0), 0);
|
|
20
|
+
const taskCount = stage.taskCount ?? 0;
|
|
21
|
+
const durationMs = (stage.completedAt ?? 0) - (stage.submittedAt ?? 0);
|
|
22
|
+
const overheadMs = taskTimeMs - (stage.executorRunTime ?? 0);
|
|
23
|
+
if (taskTimeMs <= 0 || taskCount <= 0 || durationMs <= 0 || overheadMs <= 0) return null;
|
|
24
|
+
return { perTaskMs: overheadMs / taskCount, concurrency: taskTimeMs / durationMs };
|
|
25
|
+
}
|
|
26
|
+
|
|
27
|
+
// Both constants above are single-device figures (one NIC, one local disk). A stage's shuffle
|
|
28
|
+
// reads and spills are spread over every executor that ran its tasks, each moving its own share
|
|
29
|
+
// in parallel, so the stage's aggregate bandwidth scales with that executor count. Dividing a
|
|
30
|
+
// stage-wide byte total by one device's bandwidth modeled the whole cluster as one link: on 14
|
|
31
|
+
// real logs that claimed up to 1,870s of shuffle time on stages whose tasks measured ~0s of
|
|
32
|
+
// shuffle fetch wait (222 of 284 shuffle findings). No executor data falls back to one device.
|
|
33
|
+
function stageIoParallelism(stage ) {
|
|
34
|
+
const executors = Array.isArray(stage.executorStats) ? stage.executorStats.length : 0;
|
|
35
|
+
return Math.max(1, executors);
|
|
36
|
+
}
|
|
37
|
+
|
|
38
|
+
// Wall-clock the stage's tasks spent blocked fetching shuffle blocks: fetchWaitTime is a
|
|
39
|
+
// cross-task sum like executorRunTime, so dividing by the stage's average concurrency converts it
|
|
40
|
+
// (the gc estimate's conversion). Null when the stage has no run time or duration to convert with.
|
|
41
|
+
// On 14 real logs the link model claimed 2780s over 284 shuffle findings, 256s once capped at
|
|
42
|
+
// this, with about zero fetch wait on 5 of the 10 non-info ones: reads that overlapped compute
|
|
43
|
+
// stalled nothing.
|
|
44
|
+
function fetchWaitWallClockMs(stage ) {
|
|
45
|
+
const fetchWaitMs = stage.fetchWaitTime;
|
|
46
|
+
const runTimeMs = stage.executorRunTime ?? 0;
|
|
47
|
+
const durationMs = (stage.completedAt ?? 0) - (stage.submittedAt ?? 0);
|
|
48
|
+
if (typeof fetchWaitMs !== 'number' || runTimeMs <= 0 || durationMs <= 0) return null;
|
|
49
|
+
return fetchWaitMs / (runTimeMs / durationMs);
|
|
50
|
+
}
|
|
51
|
+
|
|
52
|
+
// Wall-clock a stage's wasted (retried) attempts cost it. Each one delayed only its own task, and
|
|
53
|
+
// attempts of different tasks ran side by side: one lost executor fails every task it was running
|
|
54
|
+
// at once (4 wasted attempts of 36.6s each, all first attempts, on a real stage that ran 41 tasks
|
|
55
|
+
// at once, claimed as 146.6s). So the stage lost at most its longest retry chain, the most attempts
|
|
56
|
+
// one task wasted (its highest attempt number + 1) times the mean wasted attempt, or their summed
|
|
57
|
+
// time spread over its slots, whichever is larger. Without a sample of every wasted attempt
|
|
58
|
+
// (retryTaskSamples is capped), the chain isn't known: the summed time.
|
|
59
|
+
function retryWallClockMs(stage ) {
|
|
60
|
+
const totalMs = (stage.retryWasteMs ) ?? 0;
|
|
61
|
+
const attempts = (stage.wastedAttempts ) ?? 0;
|
|
62
|
+
const samples = Array.isArray(stage.retryTaskSamples)
|
|
63
|
+
? (stage.retryTaskSamples ) : [];
|
|
64
|
+
if (totalMs <= 0 || attempts <= 0 || samples.length < attempts) return totalMs;
|
|
65
|
+
const chain = samples.reduce((longest, s) => Math.max(longest, (s.attemptNumber ?? 0) + 1), 1);
|
|
66
|
+
const slots = Math.max(1, stage.peakConcurrentTasks ?? 1);
|
|
67
|
+
return Math.min(totalMs, Math.max((chain * totalMs) / attempts, totalMs / slots));
|
|
68
|
+
}
|
|
9
69
|
// Spark's classic recommended shuffle partition size.
|
|
10
70
|
const IDEAL_BYTES_PER_PARTITION_TASK = 128 * 1024 * 1024;
|
|
11
|
-
// Assumed per-task scheduling/launch overhead
|
|
71
|
+
// Assumed per-task scheduling/launch overhead: the fallback when a stage lacks the per-executor
|
|
72
|
+
// task-time sums measuredTaskOverhead needs.
|
|
12
73
|
const TASK_SCHEDULING_OVERHEAD_MS = 50;
|
|
13
74
|
// Assumed per-file open latency (small-file overhead).
|
|
14
75
|
const FILE_OPEN_OVERHEAD_MS = 10;
|
|
@@ -21,15 +82,10 @@ const EXECUTOR_STARTUP_OVERHEAD_MS = 15000;
|
|
|
21
82
|
// Assumed re-read throughput, shared by cachingOpportunity and cacheUtilization.
|
|
22
83
|
const RE_READ_THROUGHPUT_BPS = 125_000_000;
|
|
23
84
|
|
|
24
|
-
//
|
|
25
|
-
//
|
|
26
|
-
|
|
27
|
-
|
|
28
|
-
if (typeof infoMin !== 'number') {
|
|
29
|
-
throw new Error("impact-estimator: stageSlowness detector's 'infoMin' threshold not found in detectorCatalog()");
|
|
30
|
-
}
|
|
31
|
-
return infoMin;
|
|
32
|
-
})();
|
|
85
|
+
// skew and straggler claim time off the stage's longest task itself, so the occupancy clip must
|
|
86
|
+
// not floor them at that same task (see estimateSingleStage). detectors.ts's clippedWasteMs gates
|
|
87
|
+
// both detectors on the same option so the firing floor and the displayed estimate agree.
|
|
88
|
+
const TAIL_CLAIM = { shortensLongestTask: true };
|
|
33
89
|
|
|
34
90
|
// No quantifiable magnitude -> 'informational'; a rawWaste figure with no stage window ->
|
|
35
91
|
// 'resourceOnly'. Never a fake {low:0, high:0}: wallClock is null in both cases.
|
|
@@ -46,8 +102,9 @@ function singleStageImpact(
|
|
|
46
102
|
occupancy ,
|
|
47
103
|
estimateMethod ,
|
|
48
104
|
rawWaste ,
|
|
105
|
+
opts ,
|
|
49
106
|
) {
|
|
50
|
-
const est = estimateSingleStage(wasteMs, stageId, stages , occupancy);
|
|
107
|
+
const est = estimateSingleStage(wasteMs, stageId, stages , occupancy, opts);
|
|
51
108
|
if (est) return { basis: est.basis, wallClock: est.wallClock, estimateMethod, rawWaste };
|
|
52
109
|
return costOnly(estimateMethod, rawWaste); // stage excluded from the sweep (duration <= 0)
|
|
53
110
|
}
|
|
@@ -77,6 +134,7 @@ function computeEstimateForFinding(
|
|
|
77
134
|
finding ,
|
|
78
135
|
stages ,
|
|
79
136
|
occupancy ,
|
|
137
|
+
totalCores ,
|
|
80
138
|
) {
|
|
81
139
|
switch (finding.type) {
|
|
82
140
|
case 'retryWaste': {
|
|
@@ -85,7 +143,9 @@ function computeEstimateForFinding(
|
|
|
85
143
|
const stage = stages.get(finding.stageId);
|
|
86
144
|
if (!stage) return null;
|
|
87
145
|
const wasteMs = (stage.retryWasteMs ) ?? 0;
|
|
88
|
-
|
|
146
|
+
const wallClockMs = retryWallClockMs(stage);
|
|
147
|
+
return singleStageImpact(wallClockMs, finding.stageId, stages, occupancy,
|
|
148
|
+
wallClockMs === wasteMs ? 'measured' : 'modeled', { value: wasteMs, unit: 'ms' });
|
|
89
149
|
}
|
|
90
150
|
case 'speculationWaste': {
|
|
91
151
|
if (finding.stageId == null) return null;
|
|
@@ -103,6 +163,9 @@ function computeEstimateForFinding(
|
|
|
103
163
|
return { basis: 'serial', wallClock: { low: wasteMs, high: wasteMs }, estimateMethod: 'measured' };
|
|
104
164
|
}
|
|
105
165
|
case 'gc': {
|
|
166
|
+
// The low-GC branch is an over-provisioning signal whose fix (less executor memory) raises
|
|
167
|
+
// GC rather than recovering it: the stage's GC time is no saving there, so no waste model.
|
|
168
|
+
if (finding.direction === 'low') return costOnly('none');
|
|
106
169
|
if (finding.stageId == null) return null;
|
|
107
170
|
const stage = stages.get(finding.stageId);
|
|
108
171
|
if (!stage) return null;
|
|
@@ -128,8 +191,11 @@ function computeEstimateForFinding(
|
|
|
128
191
|
const p50 = stage.taskDurationP50 ?? 0;
|
|
129
192
|
// computeSkewRatio's own metric labels (src/detectors.ts): 'P95/median' or 'max/median'.
|
|
130
193
|
const usesP95Branch = finding.metric === 'P95/median';
|
|
131
|
-
const
|
|
132
|
-
|
|
194
|
+
const singleDelta = Math.max(0, usesP95Branch ? (stage.taskDurationP95 ?? 0) - p50 : (stage.taskDurationMax ?? 0) - p50);
|
|
195
|
+
const wasteMs = tailRecoveryMs(stage, singleDelta);
|
|
196
|
+
// Fixing the skew still waits on the longest task it leaves, as for straggler.
|
|
197
|
+
return singleStageImpact(wasteMs, finding.stageId, stages, occupancy, 'measured', { value: wasteMs, unit: 'ms' },
|
|
198
|
+
{ ...TAIL_CLAIM, removedCoreWorkMs: tailRemovedWorkMs(stage, singleDelta), longestTaskAfterFixMs: stragglerFixLongestTaskMs(stage) });
|
|
133
199
|
}
|
|
134
200
|
case 'straggler':
|
|
135
201
|
case 'stageShape': {
|
|
@@ -171,8 +237,11 @@ function computeEstimateForFinding(
|
|
|
171
237
|
if (finding.stageId == null) return null;
|
|
172
238
|
const stage = stages.get(finding.stageId);
|
|
173
239
|
if (!stage) return null;
|
|
174
|
-
const
|
|
175
|
-
|
|
240
|
+
const longestTaskAfterFixMs = stragglerFixLongestTaskMs(stage);
|
|
241
|
+
const singleDelta = Math.max(0, (stage.taskDurationMax ?? 0) - longestTaskAfterFixMs);
|
|
242
|
+
const wasteMs = tailRecoveryMs(stage, singleDelta);
|
|
243
|
+
return singleStageImpact(wasteMs, finding.stageId, stages, occupancy, 'measured', { value: wasteMs, unit: 'ms' },
|
|
244
|
+
{ ...TAIL_CLAIM, removedCoreWorkMs: tailRemovedWorkMs(stage, singleDelta), longestTaskAfterFixMs });
|
|
176
245
|
}
|
|
177
246
|
case 'slowHost': {
|
|
178
247
|
// Three duration-based shapes, each carrying its absolute-ms figure under a different field
|
|
@@ -202,18 +271,31 @@ function computeEstimateForFinding(
|
|
|
202
271
|
const occurrences = typeof finding.value === 'number' ? finding.value : 0;
|
|
203
272
|
// Defensive only: minOccurrences guarantees occurrences >= 2 on real data; a malformed-value fallback.
|
|
204
273
|
if (occurrences < 2) return costOnly('none');
|
|
274
|
+
// Same-shaped repeats whose details differ compute different data: nothing is known to be
|
|
275
|
+
// recomputed, so there is no time to claim.
|
|
276
|
+
if (finding.occurrencesIdentical === false) return costOnly('none');
|
|
277
|
+
// Each stage contributes the share of its operators inside the repeated subtree: a stage it
|
|
278
|
+
// shares with other operators (the consuming join, the join's other side) isn't all its
|
|
279
|
+
// time, and claiming whole stages let sibling groups claim the same stage twice. Findings
|
|
280
|
+
// built without the field (hand-made fixtures) count every linked stage whole.
|
|
281
|
+
const shares = (finding.stageShares ?? null) ;
|
|
205
282
|
const redundantFraction = (occurrences - 1) / occurrences;
|
|
206
283
|
const wasteMsByStage = new Map ();
|
|
207
284
|
for (const id of stageIds) {
|
|
208
285
|
const s = stages.get(id);
|
|
209
|
-
|
|
210
|
-
|
|
211
|
-
|
|
286
|
+
const share = shares ? (shares[id] ?? 0) : 1;
|
|
287
|
+
if (s && share > 0) {
|
|
288
|
+
// Time with tasks running, not submit-to-complete: a stage left waiting for cores
|
|
289
|
+
// (2491s open, 60s of tasks on a real log) isn't recomputing anything while it waits.
|
|
290
|
+
const activeMs = s.taskActiveMs ?? Math.max(0, (s.completedAt ?? 0) - (s.submittedAt ?? 0));
|
|
291
|
+
wasteMsByStage.set(id, activeMs * share * redundantFraction);
|
|
212
292
|
}
|
|
213
293
|
}
|
|
294
|
+
// No operator of the subtree ran in a known stage: no time to attribute.
|
|
295
|
+
if (wasteMsByStage.size === 0) return costOnly('none');
|
|
214
296
|
const totalWasteMs = [...wasteMsByStage.values()].reduce((sum, ms) => sum + ms, 0);
|
|
215
297
|
const rawWaste = { value: totalWasteMs, unit: 'ms' };
|
|
216
|
-
const est = estimateMultiStage(
|
|
298
|
+
const est = estimateMultiStage([...wasteMsByStage.keys()], wasteMsByStage, stages , occupancy);
|
|
217
299
|
if (!est) return costOnly('measured', rawWaste);
|
|
218
300
|
return { basis: est.basis, wallClock: est.wallClock, estimateMethod: 'measured', rawWaste };
|
|
219
301
|
}
|
|
@@ -222,16 +304,21 @@ function computeEstimateForFinding(
|
|
|
222
304
|
const stage = stages.get(finding.stageId);
|
|
223
305
|
if (!stage) return null;
|
|
224
306
|
const shuffleReadBytes = stage.shuffleReadBytes ?? 0;
|
|
225
|
-
const
|
|
226
|
-
// The
|
|
227
|
-
|
|
307
|
+
const modeledMs = (shuffleReadBytes / (SHUFFLE_THROUGHPUT_BPS * stageIoParallelism(stage))) * 1000;
|
|
308
|
+
// The link model can't see whether the reads stalled the tasks: capped at the fetch wait the
|
|
309
|
+
// tasks measured, the claim never exceeds what the stage spent blocked on the network.
|
|
310
|
+
const measuredMs = fetchWaitWallClockMs(stage);
|
|
311
|
+
const wasteMs = measuredMs == null ? modeledMs : Math.min(modeledMs, measuredMs);
|
|
312
|
+
// rawWaste: the measured byte volume behind the modeled figure.
|
|
313
|
+
return singleStageImpact(wasteMs, finding.stageId, stages, occupancy,
|
|
314
|
+
measuredMs != null && measuredMs < modeledMs ? 'measured' : 'modeled', { value: shuffleReadBytes, unit: 'bytes' });
|
|
228
315
|
}
|
|
229
316
|
case 'spill': {
|
|
230
317
|
if (finding.stageId == null) return null;
|
|
231
318
|
const stage = stages.get(finding.stageId);
|
|
232
319
|
if (!stage) return null;
|
|
233
320
|
const diskBytesSpilled = stage.diskBytesSpilled ?? 0;
|
|
234
|
-
const wasteMs = (diskBytesSpilled / SPILL_IO_THROUGHPUT_BPS) * 1000;
|
|
321
|
+
const wasteMs = (diskBytesSpilled / (SPILL_IO_THROUGHPUT_BPS * stageIoParallelism(stage))) * 1000;
|
|
235
322
|
// Surfaces the number the formula uses: the displayed metric is memoryBytesSpilled, but disk spill costs the I/O time.
|
|
236
323
|
return singleStageImpact(wasteMs, finding.stageId, stages, occupancy, 'modeled', { value: diskBytesSpilled, unit: 'bytes' });
|
|
237
324
|
}
|
|
@@ -239,9 +326,21 @@ function computeEstimateForFinding(
|
|
|
239
326
|
if (finding.stageId == null) return null;
|
|
240
327
|
const stage = stages.get(finding.stageId);
|
|
241
328
|
if (!stage) return null;
|
|
242
|
-
|
|
243
|
-
|
|
244
|
-
|
|
329
|
+
// The recommended fix is more partitions, which only helps a stage that ran fewer tasks than
|
|
330
|
+
// the cluster has cores: the time its tasks were running could then spread over up to
|
|
331
|
+
// totalCores (lowShuffleParallelism's shape). Time the stage sat open with no task running
|
|
332
|
+
// is queueing no partition count recovers. Splitting partitions splits the longest task
|
|
333
|
+
// too, hence TAIL_CLAIM's post-fix floor. Unknown cluster size: no defensible figure.
|
|
334
|
+
if (totalCores <= 0) return costOnly('modeled');
|
|
335
|
+
// A stage that read no input and no shuffle has no data for more partitions to split (a
|
|
336
|
+
// 1-task count stage open 27 minutes on 5s of CPU was claimed 99% recoverable): claim 0.
|
|
337
|
+
const readBytes = (stage.inputBytes ?? 0) + (stage.shuffleReadBytes ?? 0);
|
|
338
|
+
const activeMs = typeof stage.taskActiveMs === 'number'
|
|
339
|
+
? stage.taskActiveMs
|
|
340
|
+
: Math.max(0, (stage.completedAt ?? 0) - (stage.submittedAt ?? 0));
|
|
341
|
+
const taskCount = stage.taskCount ?? 0;
|
|
342
|
+
const wasteMs = readBytes > 0 ? activeMs * Math.max(0, 1 - taskCount / totalCores) : 0;
|
|
343
|
+
return singleStageImpact(wasteMs, finding.stageId, stages, occupancy, 'modeled', { value: wasteMs, unit: 'ms' }, TAIL_CLAIM);
|
|
245
344
|
}
|
|
246
345
|
case 'partitionSizing': {
|
|
247
346
|
if (finding.stageId == null) return null;
|
|
@@ -275,12 +374,29 @@ function computeEstimateForFinding(
|
|
|
275
374
|
if (!stage) return null;
|
|
276
375
|
const taskCount = stage.taskCount ?? 0;
|
|
277
376
|
const excessTaskCount = Math.max(0, taskCount - Math.round(taskCount / 10));
|
|
377
|
+
const measured = measuredTaskOverhead(stage);
|
|
378
|
+
if (measured) {
|
|
379
|
+
// Coalescing to a tenth of the tasks removes the excess tasks' per-task overhead: core
|
|
380
|
+
// time spent in parallel, so wall-clock at the stage's achieved concurrency (floored at
|
|
381
|
+
// 1: a mostly-idle stage can't save more wall-clock than the task time it removes).
|
|
382
|
+
const wasteMs = (excessTaskCount * measured.perTaskMs) / Math.max(1, measured.concurrency);
|
|
383
|
+
return singleStageImpact(wasteMs, finding.stageId, stages, occupancy, 'measured', { value: wasteMs, unit: 'ms' });
|
|
384
|
+
}
|
|
278
385
|
const wasteMs = excessTaskCount * TASK_SCHEDULING_OVERHEAD_MS;
|
|
279
386
|
return singleStageImpact(wasteMs, finding.stageId, stages, occupancy, 'modeled', { value: wasteMs, unit: 'ms' });
|
|
280
387
|
}
|
|
281
388
|
case 'smallFiles': {
|
|
282
|
-
const
|
|
283
|
-
|
|
389
|
+
const fileMs = ((finding.fileCount ) ?? 0) * FILE_OPEN_OVERHEAD_MS;
|
|
390
|
+
// A read's files are opened by the scan's tasks, in parallel: spread the per-file cost over
|
|
391
|
+
// the most tasks its stages ran at once (91344 files x 10ms is 913s, claimed against a 117s
|
|
392
|
+
// stage that ran 314 tasks at once). A write keeps the serial sum: the job commit moves each
|
|
393
|
+
// output file on the driver, one after another.
|
|
394
|
+
const stageIds = finding.stageIds ;
|
|
395
|
+
let slots = 1;
|
|
396
|
+
if (finding.direction === 'read') {
|
|
397
|
+
for (const id of stageIds ?? []) slots = Math.max(slots, stages.get(id)?.peakConcurrentTasks ?? 1);
|
|
398
|
+
}
|
|
399
|
+
return stageMappableWasteOrCostOnly(fileMs / slots, stageIds, stages, occupancy);
|
|
284
400
|
}
|
|
285
401
|
case 'overBroadcast': {
|
|
286
402
|
// metric: 'broadcastBytes', value: <bytes>.
|
|
@@ -385,7 +501,7 @@ function computeEstimateForFinding(
|
|
|
385
501
|
export function estimateImpact(findings , stages , totalCores = 0) {
|
|
386
502
|
const occupancy = computeOccupancy(stages , totalCores);
|
|
387
503
|
for (const f of findings) {
|
|
388
|
-
const estimate = computeEstimateForFinding(f, stages, occupancy);
|
|
504
|
+
const estimate = computeEstimateForFinding(f, stages, occupancy, totalCores);
|
|
389
505
|
if (estimate) f.impactEstimate = estimate;
|
|
390
506
|
}
|
|
391
507
|
return findings;
|
package/vendor-core/mcp-tools.js
CHANGED
|
@@ -228,6 +228,11 @@ export function getFindingDocumentation(type ) {
|
|
|
228
228
|
if (tuningPath && anchor && existsSync(tuningPath)) {
|
|
229
229
|
const tuningContent = readFileSync(tuningPath, 'utf8');
|
|
230
230
|
tuningDoc = { anchor, title: extractDocTitle(tuningContent), content: tuningContent };
|
|
231
|
+
} else if (anchor && !slug) {
|
|
232
|
+
// A section hosted on a chapter page (e.g. autoscaling churn on cluster-config): return the
|
|
233
|
+
// owning chapter, the same page the web view's docs link opens.
|
|
234
|
+
const entry = findNavEntry(pageForAnchor(anchor.replace(/^#/, '')));
|
|
235
|
+
if (entry) tuningDoc = { anchor, title: entry.title, content: readNavEntryContent(entry) };
|
|
231
236
|
}
|
|
232
237
|
|
|
233
238
|
return { type, name: titleCase(label), detectionDoc, tuningDoc };
|
|
@@ -235,21 +240,29 @@ export function getFindingDocumentation(type ) {
|
|
|
235
240
|
|
|
236
241
|
const CHAPTERS_NAV_FILE = join(DOCS_CONTENT_DIR, 'chapters', 'nav-index.json');
|
|
237
242
|
|
|
243
|
+
|
|
244
|
+
|
|
245
|
+
function findNavEntry(page ) {
|
|
246
|
+
const nav = JSON.parse(readFileSync(CHAPTERS_NAV_FILE, 'utf8')) ;
|
|
247
|
+
return nav.find((e) => e.anchor === page);
|
|
248
|
+
}
|
|
249
|
+
|
|
250
|
+
function readNavEntryContent(entry ) {
|
|
251
|
+
const dir = entry.store === 'tuning' ? 'tuning' : 'chapters';
|
|
252
|
+
return readFileSync(join(DOCS_CONTENT_DIR, dir, `${entry.slug}.md`), 'utf8');
|
|
253
|
+
}
|
|
254
|
+
|
|
238
255
|
|
|
239
256
|
|
|
240
257
|
/** Full tuning-reference markdown for one doc anchor, run-independent. Resolves the anchor to its
|
|
241
258
|
* owning page via pageForAnchor (so '#metric-task-duration' returns the 'metrics' page), looks it up
|
|
242
|
-
* in the
|
|
259
|
+
* in the generated nav-index, and reads the markdown from the same docs-content store the website
|
|
243
260
|
* renders from. The general-chapter counterpart to getFindingDocumentation (keyed by finding type). */
|
|
244
261
|
export function getReferenceDoc(anchor ) {
|
|
245
262
|
const page = pageForAnchor(String(anchor).replace(/^#/, ''));
|
|
246
|
-
const
|
|
247
|
-
;
|
|
248
|
-
const entry = nav.find((e) => e.anchor === page);
|
|
263
|
+
const entry = findNavEntry(page);
|
|
249
264
|
if (!entry) throw mcpError('invalid-anchor', `Unknown reference anchor: ${anchor}`);
|
|
250
|
-
|
|
251
|
-
const content = readFileSync(join(DOCS_CONTENT_DIR, dir, `${entry.slug}.md`), 'utf8');
|
|
252
|
-
return { anchor: page, title: entry.title, content };
|
|
265
|
+
return { anchor: page, title: entry.title, content: readNavEntryContent(entry) };
|
|
253
266
|
}
|
|
254
267
|
|
|
255
268
|
export function getFindingEvidence(
|
package/vendor-core/occupancy.js
CHANGED
|
@@ -117,6 +117,9 @@ export function clipToCeiling(wasteMsClaimed , stage , cei
|
|
|
117
117
|
|
|
118
118
|
|
|
119
119
|
|
|
120
|
+
|
|
121
|
+
|
|
122
|
+
|
|
120
123
|
|
|
121
124
|
|
|
122
125
|
export function computeOccupancy(
|
|
@@ -128,7 +131,11 @@ export function computeOccupancy(
|
|
|
128
131
|
for (const s of stages.values()) {
|
|
129
132
|
const g = gate.get(s.id);
|
|
130
133
|
if (g === undefined) continue; // excluded from the sweep (duration <= 0)
|
|
131
|
-
info.set(s.id, {
|
|
134
|
+
info.set(s.id, {
|
|
135
|
+
gate: g,
|
|
136
|
+
ceiling: computeCeiling(s, totalCores),
|
|
137
|
+
coreWorkFloor: totalCores > 0 ? (s.executorRunTime ?? 0) / totalCores : 0,
|
|
138
|
+
});
|
|
132
139
|
}
|
|
133
140
|
return info;
|
|
134
141
|
}
|
|
@@ -140,6 +147,59 @@ const SERIAL_GATE_THRESHOLD = 0.999;
|
|
|
140
147
|
|
|
141
148
|
|
|
142
149
|
|
|
150
|
+
|
|
151
|
+
|
|
152
|
+
|
|
153
|
+
|
|
154
|
+
|
|
155
|
+
|
|
156
|
+
|
|
157
|
+
|
|
158
|
+
|
|
159
|
+
// Wall-clock a skew/straggler fix recovers, given the slowest task's excess over the median. A
|
|
160
|
+
// lone straggler costs that excess; a tail of many slow tasks (a bimodal stage: 26% of 1400
|
|
161
|
+
// tasks over 4x P50 on a real log) costs its summed excess (stragglerExcessMs) spread over the
|
|
162
|
+
// slots the stage had (peakConcurrentTasks), far more than one task's. The task-level replay in
|
|
163
|
+
// dev/eval-tail-replay.mjs recovers about the larger of the two. Average concurrency would be the
|
|
164
|
+
// wrong divisor: a tail-dominated stage runs few tasks for most of its span (5.7 average vs 14
|
|
165
|
+
// peak on one real stage), which doubled the claim.
|
|
166
|
+
export function tailRecoveryMs(stage , singleTaskExcessMs ) {
|
|
167
|
+
const excessMs = stage.stragglerExcessMs ?? 0;
|
|
168
|
+
const slots = stage.peakConcurrentTasks ?? 0;
|
|
169
|
+
if (excessMs <= 0 || slots <= 0) return singleTaskExcessMs;
|
|
170
|
+
return Math.max(singleTaskExcessMs, excessMs / slots);
|
|
171
|
+
}
|
|
172
|
+
|
|
173
|
+
// Task time a skew/straggler fix removes, by the same measure: the tasks over 4x P50 capped at the
|
|
174
|
+
// median (stragglerExcessMs), or the slowest task's excess when that alone is larger.
|
|
175
|
+
export function tailRemovedWorkMs(stage , singleTaskExcessMs ) {
|
|
176
|
+
return Math.max(singleTaskExcessMs, stage.stragglerExcessMs ?? 0);
|
|
177
|
+
}
|
|
178
|
+
|
|
179
|
+
// Longest task a straggler fix leaves: every task over 4x P50 comes down to the median, so the
|
|
180
|
+
// longest one at or under that (finalizeStage's longestNonStragglerMs). Without stragglers (a
|
|
181
|
+
// speculation-driven finding) or the field (older snapshots), the median.
|
|
182
|
+
export function stragglerFixLongestTaskMs(stage ) {
|
|
183
|
+
const p50 = stage.taskDurationP50 ?? 0;
|
|
184
|
+
if (!((stage.stragglerCount ?? 0) > 0) || stage.longestNonStragglerMs == null) return p50;
|
|
185
|
+
return Math.max(p50, stage.longestNonStragglerMs);
|
|
186
|
+
}
|
|
187
|
+
|
|
188
|
+
|
|
189
|
+
|
|
190
|
+
|
|
191
|
+
|
|
192
|
+
|
|
193
|
+
|
|
194
|
+
|
|
195
|
+
|
|
196
|
+
|
|
197
|
+
|
|
198
|
+
|
|
199
|
+
|
|
200
|
+
|
|
201
|
+
|
|
202
|
+
|
|
143
203
|
/**
|
|
144
204
|
* Per-finding estimate for a single-stage waste claim. Returns null when the
|
|
145
205
|
* stage was excluded from the sweep (duration <= 0): callers must fall back
|
|
@@ -151,11 +211,19 @@ export function estimateSingleStage(
|
|
|
151
211
|
stageId ,
|
|
152
212
|
stages ,
|
|
153
213
|
info ,
|
|
214
|
+
{ shortensLongestTask = false, removedCoreWorkMs = 0, longestTaskAfterFixMs = 0 } = {},
|
|
154
215
|
) {
|
|
155
216
|
const stage = stages.get(stageId);
|
|
156
217
|
const stageInfo = info.get(stageId);
|
|
157
218
|
if (!stage || !stageInfo) return null;
|
|
158
|
-
const
|
|
219
|
+
const coreWork = stage.executorRunTime ?? 0;
|
|
220
|
+
const coreWorkFloor = removedCoreWorkMs > 0 && coreWork > 0
|
|
221
|
+
? stageInfo.coreWorkFloor * Math.max(0, 1 - removedCoreWorkMs / coreWork)
|
|
222
|
+
: stageInfo.coreWorkFloor;
|
|
223
|
+
const ceiling = shortensLongestTask
|
|
224
|
+
? Math.max((stage.taskDurationMax ?? 0) - wasteMsClaimed, longestTaskAfterFixMs, coreWorkFloor)
|
|
225
|
+
: stageInfo.ceiling;
|
|
226
|
+
const clipped = clipToCeiling(wasteMsClaimed, stage, ceiling);
|
|
159
227
|
if (stageInfo.gate >= SERIAL_GATE_THRESHOLD) {
|
|
160
228
|
return { basis: 'serial', wallClock: { low: clipped, high: clipped } };
|
|
161
229
|
}
|
|
@@ -2,7 +2,7 @@ import { Gunzip } from './vendor/fflate.js';
|
|
|
2
2
|
import { createLz4BlockDecoder } from './lz4-block.js';
|
|
3
3
|
import { Decompress as ZstdDecompress } from './vendor/fzstd.js';
|
|
4
4
|
import { createSnappyBlockDecoder } from './snappy-block.js';
|
|
5
|
-
import { createState, dispatchLine, buildChunkDecoder, emitParseCompletion,
|
|
5
|
+
import { createState, dispatchLine, buildChunkDecoder, emitParseCompletion, } from './event-handlers.js';
|
|
6
6
|
import { TASK_FIELD_NAMES } from './stage-quantiles.js';
|
|
7
7
|
import { runParseFromUrl, sniffCodec } from './shs-fetch.js';
|
|
8
8
|
|
|
@@ -34,7 +34,9 @@ const MIN_PROGRESS_STEPS = 100;
|
|
|
34
34
|
const PROGRESS_EMIT_LINES = 300;
|
|
35
35
|
|
|
36
36
|
|
|
37
|
-
|
|
37
|
+
// zstdDecoder replaces the vendored fzstd for zstd input; the Node CLI/MCP path passes
|
|
38
|
+
// cli/native-zstd.ts's native-zlib decoder, which a browser bundle can't import.
|
|
39
|
+
|
|
38
40
|
|
|
39
41
|
// Minimal shape streamFile/runParse/runParseFiles read off `file` (name, size,
|
|
40
42
|
// slice(start,end).arrayBuffer()): narrower than the full DOM `File`. A real
|
|
@@ -59,6 +61,10 @@ const PROGRESS_EMIT_LINES = 300;
|
|
|
59
61
|
// without touching the vendored files (mirrors shs-fetch.ts's shim).
|
|
60
62
|
|
|
61
63
|
|
|
64
|
+
// A Node decoder may decompress off the main thread: streamFile awaits each push.
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
|
|
62
68
|
|
|
63
69
|
// Stream one File's (possibly compressed) bytes through the codec dispatch,
|
|
64
70
|
// in `chunkSize` slices, invoking `onChunk` with each decompressed buffer as
|
|
@@ -69,6 +75,7 @@ export async function streamFile(
|
|
|
69
75
|
file ,
|
|
70
76
|
onChunk ,
|
|
71
77
|
chunkSize ,
|
|
78
|
+
zstdDecoder ,
|
|
72
79
|
) {
|
|
73
80
|
const header = new Uint8Array(await file.slice(0, Math.min(8, file.size)).arrayBuffer());
|
|
74
81
|
const codec = sniffCodec(header);
|
|
@@ -81,7 +88,10 @@ export async function streamFile(
|
|
|
81
88
|
let currentPct = 0;
|
|
82
89
|
const gunzip = codec === 'gz' ? new (Gunzip )((inflated) => onChunk(inflated, currentPct)) : null;
|
|
83
90
|
const lz4 = codec === 'lz4' ? createLz4BlockDecoder((inflated) => onChunk(inflated, currentPct)) : null;
|
|
84
|
-
const
|
|
91
|
+
const onZstdChunk = (inflated ) => onChunk(inflated, currentPct);
|
|
92
|
+
const zstd = codec !== 'zstd' ? null
|
|
93
|
+
: zstdDecoder ? zstdDecoder(onZstdChunk)
|
|
94
|
+
: new (ZstdDecompress )(onZstdChunk);
|
|
85
95
|
const snappy = codec === 'snappy' ? createSnappyBlockDecoder((inflated) => onChunk(inflated, currentPct)) : null;
|
|
86
96
|
|
|
87
97
|
// A fixed read size gives too few progress checkpoints on smaller files
|
|
@@ -99,7 +109,7 @@ export async function streamFile(
|
|
|
99
109
|
currentPct = start / file.size;
|
|
100
110
|
if (gunzip) gunzip.push(slice, final);
|
|
101
111
|
else if (lz4) lz4.push(slice);
|
|
102
|
-
else if (zstd) zstd.push(slice, final);
|
|
112
|
+
else if (zstd) await zstd.push(slice, final);
|
|
103
113
|
else if (snappy) snappy.push(slice);
|
|
104
114
|
else onChunk(slice, currentPct);
|
|
105
115
|
}
|
|
@@ -115,7 +125,7 @@ export async function streamFile(
|
|
|
115
125
|
export async function runParse(
|
|
116
126
|
file ,
|
|
117
127
|
state ,
|
|
118
|
-
{ emit = (msg ) => self.postMessage(msg), chunkSize = CHUNK_SIZE } = {},
|
|
128
|
+
{ emit = (msg ) => self.postMessage(msg), chunkSize = CHUNK_SIZE, zstdDecoder } = {},
|
|
119
129
|
) {
|
|
120
130
|
if (file.size === 0) {
|
|
121
131
|
emit({ type: 'error', message: 'File is empty.' });
|
|
@@ -123,10 +133,13 @@ export async function runParse(
|
|
|
123
133
|
}
|
|
124
134
|
|
|
125
135
|
const decoder = buildChunkDecoder();
|
|
136
|
+
const joined = [];
|
|
126
137
|
let linesProcessed = 0;
|
|
127
138
|
const feed = (bytes , pct ) => {
|
|
128
|
-
|
|
129
|
-
|
|
139
|
+
joined.length = 0;
|
|
140
|
+
const lines = decoder.decode(bytes, joined);
|
|
141
|
+
for (let i = 0, j = 0; i < lines.length; i++) {
|
|
142
|
+
dispatchLine(lines[i], state, emit, joined[j]?.index === i ? joined[j++] : undefined);
|
|
130
143
|
linesProcessed++;
|
|
131
144
|
if (linesProcessed % PROGRESS_EMIT_LINES === 0) {
|
|
132
145
|
emit({ type: 'progress', pct: pct ?? null, linesProcessed });
|
|
@@ -135,7 +148,7 @@ export async function runParse(
|
|
|
135
148
|
};
|
|
136
149
|
|
|
137
150
|
try {
|
|
138
|
-
await streamFile(file, feed, chunkSize);
|
|
151
|
+
await streamFile(file, feed, chunkSize, zstdDecoder);
|
|
139
152
|
} catch (e) {
|
|
140
153
|
const message = e instanceof Error ? e.message : String(e);
|
|
141
154
|
emit({ type: 'error', message: `Could not decompress "${file.name}": ${message}` });
|
|
@@ -163,7 +176,7 @@ export async function runParse(
|
|
|
163
176
|
export async function runParseFiles(
|
|
164
177
|
files ,
|
|
165
178
|
state ,
|
|
166
|
-
{ emit = (msg ) => self.postMessage(msg), chunkSize = CHUNK_SIZE } = {},
|
|
179
|
+
{ emit = (msg ) => self.postMessage(msg), chunkSize = CHUNK_SIZE, zstdDecoder } = {},
|
|
167
180
|
) {
|
|
168
181
|
if (files.length === 0) {
|
|
169
182
|
emit({ type: 'error', message: 'Rolling event-log directory contained no event files.' });
|
|
@@ -171,13 +184,16 @@ export async function runParseFiles(
|
|
|
171
184
|
}
|
|
172
185
|
|
|
173
186
|
const decoder = buildChunkDecoder();
|
|
187
|
+
const joined = [];
|
|
174
188
|
let linesProcessed = 0;
|
|
175
189
|
const totalSize = files.reduce((sum, f) => sum + f.size, 0);
|
|
176
190
|
let bytesBeforeCurrentFile = 0;
|
|
177
191
|
let currentFileSize = 0;
|
|
178
192
|
const feed = (bytes , pct ) => {
|
|
179
|
-
|
|
180
|
-
|
|
193
|
+
joined.length = 0;
|
|
194
|
+
const lines = decoder.decode(bytes, joined);
|
|
195
|
+
for (let i = 0, j = 0; i < lines.length; i++) {
|
|
196
|
+
dispatchLine(lines[i], state, emit, joined[j]?.index === i ? joined[j++] : undefined);
|
|
181
197
|
linesProcessed++;
|
|
182
198
|
if (linesProcessed % PROGRESS_EMIT_LINES === 0) {
|
|
183
199
|
const overallPct = totalSize > 0 ? (bytesBeforeCurrentFile + (pct ?? 0) * currentFileSize) / totalSize : null;
|
|
@@ -189,7 +205,7 @@ export async function runParseFiles(
|
|
|
189
205
|
for (const file of files) {
|
|
190
206
|
currentFileSize = file.size;
|
|
191
207
|
try {
|
|
192
|
-
await streamFile(file, feed, chunkSize);
|
|
208
|
+
await streamFile(file, feed, chunkSize, zstdDecoder);
|
|
193
209
|
} catch (e) {
|
|
194
210
|
const message = e instanceof Error ? e.message : String(e);
|
|
195
211
|
emit({ type: 'error', message: `Could not decompress "${file.name}": ${message}` });
|
|
@@ -55,6 +55,10 @@ export function scanRelationId(name , detail ) {
|
|
|
55
55
|
// Delta tables). All catalog tables surface as `spark_catalog.<db>.<table>`.
|
|
56
56
|
let format = null, table = null;
|
|
57
57
|
const nameM = name.match(/^Scan\s+(parquet|orc|csv|json)\s+(spark_catalog\.\S+)$/i);
|
|
58
|
+
// Fast reject: every branch below that returns a key needs this name match, a FileScan
|
|
59
|
+
// detail or a JDBCRelation detail. Most plan nodes (Project, Filter, joins) have none, and
|
|
60
|
+
// their long details otherwise pay every regex below.
|
|
61
|
+
if (!nameM && !/FileScan/i.test(detail) && !detail.includes('JDBCRelation')) return null;
|
|
58
62
|
if (nameM) { format = nameM[1].toLowerCase(); table = nameM[2]; }
|
|
59
63
|
else {
|
|
60
64
|
const detM = detail.match(/FileScan\s+(parquet|orc|csv|json)\s+(spark_catalog\.[^[\s]+)\[/i);
|
|
@@ -1,5 +1,6 @@
|
|
|
1
1
|
import { computeWallClock } from './wall-clock.js';
|
|
2
2
|
import { normalizeDetail } from './detectors.js';
|
|
3
|
+
import { cyrb53 } from './string-hash.js';
|
|
3
4
|
import { captureSnapshot } from './session-snapshot.js';
|
|
4
5
|
|
|
5
6
|
|
|
@@ -35,21 +36,6 @@ export function normalizeStageName(name ) {
|
|
|
35
36
|
.trim();
|
|
36
37
|
}
|
|
37
38
|
|
|
38
|
-
// cyrb53: fast, non-cryptographic, deterministic 53-bit string hash. Not a
|
|
39
|
-
// security boundary; 53 bits keeps collision risk negligible at the per-node,
|
|
40
|
-
// per-stage call volume planTreeIdentity/sqlNodeIdentity put through it.
|
|
41
|
-
function cyrb53(str , seed = 0) {
|
|
42
|
-
let h1 = 0xdeadbeef ^ seed, h2 = 0x41c6ce57 ^ seed;
|
|
43
|
-
for (let i = 0; i < str.length; i++) {
|
|
44
|
-
const ch = str.charCodeAt(i);
|
|
45
|
-
h1 = Math.imul(h1 ^ ch, 2654435761);
|
|
46
|
-
h2 = Math.imul(h2 ^ ch, 1597334677);
|
|
47
|
-
}
|
|
48
|
-
h1 = Math.imul(h1 ^ (h1 >>> 16), 2246822507) ^ Math.imul(h2 ^ (h2 >>> 13), 3266489909);
|
|
49
|
-
h2 = Math.imul(h2 ^ (h2 >>> 16), 2246822507) ^ Math.imul(h1 ^ (h1 >>> 13), 3266489909);
|
|
50
|
-
return (4294967296 * (2097151 & h2) + (h1 >>> 0)).toString(16);
|
|
51
|
-
}
|
|
52
|
-
|
|
53
39
|
// Bottom-up, order-independent structural identity of a resolved plan tree:
|
|
54
40
|
// each node folds its normalized name/detail with its children's digests
|
|
55
41
|
// (children sorted, so AQE picking a different broadcast side still matches),
|