sparkforensics-mcp 0.2.2 → 0.2.4
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/bin/sparkforensics-mcp.mjs +11 -5
- package/package.json +3 -3
- package/vendor-core/cli/collect-run.js +3 -2
- package/vendor-core/cli/native-zstd.js +351 -0
- package/vendor-core/detectors.js +243 -53
- package/vendor-core/docs-config.js +34 -8
- package/vendor-core/docs-content/chapters/03-memory-model.md +39 -0
- package/vendor-core/docs-content/chapters/11-cluster-config.md +40 -0
- package/vendor-core/docs-content/detection/gc.md +2 -0
- package/vendor-core/docs-content/detection/host.md +2 -1
- package/vendor-core/docs-content/detection/plan.md +3 -1
- package/vendor-core/docs-content/detection/shape.md +2 -1
- package/vendor-core/docs-content/detection/shfl.md +2 -1
- package/vendor-core/docs-content/detection/spill.md +1 -1
- package/vendor-core/docs-content/detection/strag.md +2 -1
- package/vendor-core/docs-content/detection/tiny.md +2 -1
- package/vendor-core/docs-content/tuning/failures.md +1 -1
- package/vendor-core/docs-content/tuning/gc.md +11 -4
- package/vendor-core/docs-content/tuning/shuffle.md +25 -5
- package/vendor-core/docs-content/tuning/skew.md +14 -6
- package/vendor-core/docs-content/tuning/small-files.md +12 -7
- package/vendor-core/docs-content/tuning/straggler.md +34 -0
- package/vendor-core/docs-content/tuning/tiny-tasks.md +1 -1
- package/vendor-core/docs-content/tuning/utilization.md +57 -7
- package/vendor-core/docs-content/upstream.json +4 -0
- package/vendor-core/event-handlers.js +321 -69
- package/vendor-core/event-schemas.js +8 -6
- package/vendor-core/evidence-report.js +3 -1
- package/vendor-core/impact-estimator.js +170 -34
- package/vendor-core/mcp-tools.js +20 -7
- package/vendor-core/occupancy.js +71 -2
- package/vendor-core/parser-worker.js +56 -24
- package/vendor-core/plan-summary.js +5 -1
- package/vendor-core/run-comparison.js +1 -15
- package/vendor-core/shs-fetch.js +18 -7
- package/vendor-core/shs-load.js +2 -1
- package/vendor-core/stage-quantiles.js +111 -3
- package/vendor-core/string-hash.js +15 -0
- package/vendor-core/types.js +14 -1
- package/vendor-core/vendor/fzstd.js +94 -18
- package/vendor-core/zstd-worker-client.js +180 -0
- package/vendor-core/zstd-worker.js +103 -0
|
@@ -1,14 +1,76 @@
|
|
|
1
1
|
|
|
2
|
-
import {
|
|
3
|
-
import {
|
|
2
|
+
import { nsToMs } from './format-utils.js';
|
|
3
|
+
import {
|
|
4
|
+
computeOccupancy, estimateSingleStage, estimateMultiStage, tailRecoveryMs, tailRemovedWorkMs, stragglerFixLongestTaskMs,
|
|
5
|
+
|
|
6
|
+
} from './occupancy.js';
|
|
4
7
|
|
|
5
|
-
// Assumed shuffle-network throughput, ~1 Gbps. Starting assumption, unvalidated.
|
|
8
|
+
// Assumed shuffle-network throughput per executor link, ~1 Gbps. Starting assumption, unvalidated.
|
|
6
9
|
const SHUFFLE_THROUGHPUT_BPS = 125_000_000;
|
|
7
|
-
// Assumed disk I/O throughput for spilled data, ~200 MB/s (conservative HDD/SSD blend).
|
|
10
|
+
// Assumed disk I/O throughput per executor for spilled data, ~200 MB/s (conservative HDD/SSD blend).
|
|
8
11
|
const SPILL_IO_THROUGHPUT_BPS = 200_000_000;
|
|
12
|
+
|
|
13
|
+
// A stage's own per-task overhead: task wall time (launch to finish, summed per executor by
|
|
14
|
+
// finalizeStage) minus executorRunTime, i.e. deserialization, result serialization and the
|
|
15
|
+
// launch round trip that coalescing tasks removes. `concurrency` is the stage's achieved task
|
|
16
|
+
// concurrency (task time / stage duration). Null when the stage has no such data or no overhead.
|
|
17
|
+
function measuredTaskOverhead(stage ) {
|
|
18
|
+
const executorStats = Array.isArray(stage.executorStats)
|
|
19
|
+
? (stage.executorStats ) : [];
|
|
20
|
+
const taskTimeMs = executorStats.reduce((sum, e) => sum + (e.totalDuration ?? 0), 0);
|
|
21
|
+
const taskCount = stage.taskCount ?? 0;
|
|
22
|
+
const durationMs = (stage.completedAt ?? 0) - (stage.submittedAt ?? 0);
|
|
23
|
+
const overheadMs = taskTimeMs - (stage.executorRunTime ?? 0);
|
|
24
|
+
if (taskTimeMs <= 0 || taskCount <= 0 || durationMs <= 0 || overheadMs <= 0) return null;
|
|
25
|
+
return { perTaskMs: overheadMs / taskCount, concurrency: taskTimeMs / durationMs };
|
|
26
|
+
}
|
|
27
|
+
|
|
28
|
+
// Both constants above are single-device figures (one NIC, one local disk). A stage's shuffle
|
|
29
|
+
// reads and spills are spread over every executor that ran its tasks, each moving its own share
|
|
30
|
+
// in parallel, so the stage's aggregate bandwidth scales with that executor count. Dividing a
|
|
31
|
+
// stage-wide byte total by one device's bandwidth modeled the whole cluster as one link: on 14
|
|
32
|
+
// real logs that claimed up to 1,870s of shuffle time on stages whose tasks measured ~0s of
|
|
33
|
+
// shuffle fetch wait (222 of 284 shuffle findings). No executor data falls back to one device.
|
|
34
|
+
function stageIoParallelism(stage ) {
|
|
35
|
+
const executors = Array.isArray(stage.executorStats) ? stage.executorStats.length : 0;
|
|
36
|
+
return Math.max(1, executors);
|
|
37
|
+
}
|
|
38
|
+
|
|
39
|
+
// Wall-clock the stage's tasks spent blocked fetching shuffle blocks: fetchWaitTime is a
|
|
40
|
+
// cross-task sum like executorRunTime, so dividing by the stage's average concurrency converts it
|
|
41
|
+
// (the gc estimate's conversion). Null when the stage has no run time or duration to convert with.
|
|
42
|
+
// On 14 real logs the link model claimed 2780s over 284 shuffle findings, 256s once capped at
|
|
43
|
+
// this, with about zero fetch wait on 5 of the 10 non-info ones: reads that overlapped compute
|
|
44
|
+
// stalled nothing.
|
|
45
|
+
function fetchWaitWallClockMs(stage ) {
|
|
46
|
+
const fetchWaitMs = stage.fetchWaitTime;
|
|
47
|
+
const runTimeMs = stage.executorRunTime ?? 0;
|
|
48
|
+
const durationMs = (stage.completedAt ?? 0) - (stage.submittedAt ?? 0);
|
|
49
|
+
if (typeof fetchWaitMs !== 'number' || runTimeMs <= 0 || durationMs <= 0) return null;
|
|
50
|
+
return fetchWaitMs / (runTimeMs / durationMs);
|
|
51
|
+
}
|
|
52
|
+
|
|
53
|
+
// Wall-clock a stage's wasted (retried) attempts cost it. Each one delayed only its own task, and
|
|
54
|
+
// attempts of different tasks ran side by side: one lost executor fails every task it was running
|
|
55
|
+
// at once (4 wasted attempts of 36.6s each, all first attempts, on a real stage that ran 41 tasks
|
|
56
|
+
// at once, claimed as 146.6s). So the stage lost at most its longest retry chain, the most attempts
|
|
57
|
+
// one task wasted (its highest attempt number + 1) times the mean wasted attempt, or their summed
|
|
58
|
+
// time spread over its slots, whichever is larger. Without a sample of every wasted attempt
|
|
59
|
+
// (retryTaskSamples is capped), the chain isn't known: the summed time.
|
|
60
|
+
function retryWallClockMs(stage ) {
|
|
61
|
+
const totalMs = (stage.retryWasteMs ) ?? 0;
|
|
62
|
+
const attempts = (stage.wastedAttempts ) ?? 0;
|
|
63
|
+
const samples = Array.isArray(stage.retryTaskSamples)
|
|
64
|
+
? (stage.retryTaskSamples ) : [];
|
|
65
|
+
if (totalMs <= 0 || attempts <= 0 || samples.length < attempts) return totalMs;
|
|
66
|
+
const chain = samples.reduce((longest, s) => Math.max(longest, (s.attemptNumber ?? 0) + 1), 1);
|
|
67
|
+
const slots = Math.max(1, stage.peakConcurrentTasks ?? 1);
|
|
68
|
+
return Math.min(totalMs, Math.max((chain * totalMs) / attempts, totalMs / slots));
|
|
69
|
+
}
|
|
9
70
|
// Spark's classic recommended shuffle partition size.
|
|
10
71
|
const IDEAL_BYTES_PER_PARTITION_TASK = 128 * 1024 * 1024;
|
|
11
|
-
// Assumed per-task scheduling/launch overhead
|
|
72
|
+
// Assumed per-task scheduling/launch overhead: the fallback when a stage lacks the per-executor
|
|
73
|
+
// task-time sums measuredTaskOverhead needs.
|
|
12
74
|
const TASK_SCHEDULING_OVERHEAD_MS = 50;
|
|
13
75
|
// Assumed per-file open latency (small-file overhead).
|
|
14
76
|
const FILE_OPEN_OVERHEAD_MS = 10;
|
|
@@ -21,15 +83,28 @@ const EXECUTOR_STARTUP_OVERHEAD_MS = 15000;
|
|
|
21
83
|
// Assumed re-read throughput, shared by cachingOpportunity and cacheUtilization.
|
|
22
84
|
const RE_READ_THROUGHPUT_BPS = 125_000_000;
|
|
23
85
|
|
|
24
|
-
//
|
|
25
|
-
//
|
|
26
|
-
|
|
27
|
-
|
|
28
|
-
|
|
29
|
-
|
|
30
|
-
|
|
31
|
-
|
|
32
|
-
|
|
86
|
+
// Below this share of executorRunTime spent on CPU, a stage's tasks were idle, waiting on something
|
|
87
|
+
// outside Spark: on 14 real logs (2026-09-23) every non-Python stage under 1% was a JDBC read, a
|
|
88
|
+
// file listing or a Delta log read, while file writes, which more partitions do parallelize,
|
|
89
|
+
// start at 2%.
|
|
90
|
+
const IDLE_CPU_SHARE_MAX = 0.01;
|
|
91
|
+
|
|
92
|
+
// True when the stage's tasks spent under IDLE_CPU_SHARE_MAX of their run time on CPU. False when
|
|
93
|
+
// the share can't be trusted: no CPU time recorded (older Spark logs omit the metric), or Python
|
|
94
|
+
// code run through PythonRDD, whose worker-process CPU executorCpuTime (the JVM task thread's)
|
|
95
|
+
// never counts (such stages read 0.1% on the same logs while computing).
|
|
96
|
+
function tasksMostlyIdle(stage ) {
|
|
97
|
+
const runMs = stage.executorRunTime ?? 0;
|
|
98
|
+
const cpuMs = nsToMs(stage.executorCpuTime ?? 0);
|
|
99
|
+
if (runMs <= 0 || cpuMs <= 0) return false;
|
|
100
|
+
if (/PythonRDD/.test(stage.name ?? '') || /org\.apache\.spark\.api\.python\./.test(stage.details ?? '')) return false;
|
|
101
|
+
return cpuMs / runMs < IDLE_CPU_SHARE_MAX;
|
|
102
|
+
}
|
|
103
|
+
|
|
104
|
+
// skew and straggler claim time off the stage's longest task itself, so the occupancy clip must
|
|
105
|
+
// not floor them at that same task (see estimateSingleStage). detectors.ts's clippedWasteMs gates
|
|
106
|
+
// both detectors on the same option so the firing floor and the displayed estimate agree.
|
|
107
|
+
const TAIL_CLAIM = { shortensLongestTask: true };
|
|
33
108
|
|
|
34
109
|
// No quantifiable magnitude -> 'informational'; a rawWaste figure with no stage window ->
|
|
35
110
|
// 'resourceOnly'. Never a fake {low:0, high:0}: wallClock is null in both cases.
|
|
@@ -46,8 +121,9 @@ function singleStageImpact(
|
|
|
46
121
|
occupancy ,
|
|
47
122
|
estimateMethod ,
|
|
48
123
|
rawWaste ,
|
|
124
|
+
opts ,
|
|
49
125
|
) {
|
|
50
|
-
const est = estimateSingleStage(wasteMs, stageId, stages , occupancy);
|
|
126
|
+
const est = estimateSingleStage(wasteMs, stageId, stages , occupancy, opts);
|
|
51
127
|
if (est) return { basis: est.basis, wallClock: est.wallClock, estimateMethod, rawWaste };
|
|
52
128
|
return costOnly(estimateMethod, rawWaste); // stage excluded from the sweep (duration <= 0)
|
|
53
129
|
}
|
|
@@ -77,6 +153,7 @@ function computeEstimateForFinding(
|
|
|
77
153
|
finding ,
|
|
78
154
|
stages ,
|
|
79
155
|
occupancy ,
|
|
156
|
+
totalCores ,
|
|
80
157
|
) {
|
|
81
158
|
switch (finding.type) {
|
|
82
159
|
case 'retryWaste': {
|
|
@@ -85,7 +162,9 @@ function computeEstimateForFinding(
|
|
|
85
162
|
const stage = stages.get(finding.stageId);
|
|
86
163
|
if (!stage) return null;
|
|
87
164
|
const wasteMs = (stage.retryWasteMs ) ?? 0;
|
|
88
|
-
|
|
165
|
+
const wallClockMs = retryWallClockMs(stage);
|
|
166
|
+
return singleStageImpact(wallClockMs, finding.stageId, stages, occupancy,
|
|
167
|
+
wallClockMs === wasteMs ? 'measured' : 'modeled', { value: wasteMs, unit: 'ms' });
|
|
89
168
|
}
|
|
90
169
|
case 'speculationWaste': {
|
|
91
170
|
if (finding.stageId == null) return null;
|
|
@@ -103,6 +182,9 @@ function computeEstimateForFinding(
|
|
|
103
182
|
return { basis: 'serial', wallClock: { low: wasteMs, high: wasteMs }, estimateMethod: 'measured' };
|
|
104
183
|
}
|
|
105
184
|
case 'gc': {
|
|
185
|
+
// The low-GC branch is an over-provisioning signal whose fix (less executor memory) raises
|
|
186
|
+
// GC rather than recovering it: the stage's GC time is no saving there, so no waste model.
|
|
187
|
+
if (finding.direction === 'low') return costOnly('none');
|
|
106
188
|
if (finding.stageId == null) return null;
|
|
107
189
|
const stage = stages.get(finding.stageId);
|
|
108
190
|
if (!stage) return null;
|
|
@@ -128,8 +210,11 @@ function computeEstimateForFinding(
|
|
|
128
210
|
const p50 = stage.taskDurationP50 ?? 0;
|
|
129
211
|
// computeSkewRatio's own metric labels (src/detectors.ts): 'P95/median' or 'max/median'.
|
|
130
212
|
const usesP95Branch = finding.metric === 'P95/median';
|
|
131
|
-
const
|
|
132
|
-
|
|
213
|
+
const singleDelta = Math.max(0, usesP95Branch ? (stage.taskDurationP95 ?? 0) - p50 : (stage.taskDurationMax ?? 0) - p50);
|
|
214
|
+
const wasteMs = tailRecoveryMs(stage, singleDelta);
|
|
215
|
+
// Fixing the skew still waits on the longest task it leaves, as for straggler.
|
|
216
|
+
return singleStageImpact(wasteMs, finding.stageId, stages, occupancy, 'measured', { value: wasteMs, unit: 'ms' },
|
|
217
|
+
{ ...TAIL_CLAIM, removedCoreWorkMs: tailRemovedWorkMs(stage, singleDelta), longestTaskAfterFixMs: stragglerFixLongestTaskMs(stage) });
|
|
133
218
|
}
|
|
134
219
|
case 'straggler':
|
|
135
220
|
case 'stageShape': {
|
|
@@ -171,8 +256,11 @@ function computeEstimateForFinding(
|
|
|
171
256
|
if (finding.stageId == null) return null;
|
|
172
257
|
const stage = stages.get(finding.stageId);
|
|
173
258
|
if (!stage) return null;
|
|
174
|
-
const
|
|
175
|
-
|
|
259
|
+
const longestTaskAfterFixMs = stragglerFixLongestTaskMs(stage);
|
|
260
|
+
const singleDelta = Math.max(0, (stage.taskDurationMax ?? 0) - longestTaskAfterFixMs);
|
|
261
|
+
const wasteMs = tailRecoveryMs(stage, singleDelta);
|
|
262
|
+
return singleStageImpact(wasteMs, finding.stageId, stages, occupancy, 'measured', { value: wasteMs, unit: 'ms' },
|
|
263
|
+
{ ...TAIL_CLAIM, removedCoreWorkMs: tailRemovedWorkMs(stage, singleDelta), longestTaskAfterFixMs });
|
|
176
264
|
}
|
|
177
265
|
case 'slowHost': {
|
|
178
266
|
// Three duration-based shapes, each carrying its absolute-ms figure under a different field
|
|
@@ -202,18 +290,31 @@ function computeEstimateForFinding(
|
|
|
202
290
|
const occurrences = typeof finding.value === 'number' ? finding.value : 0;
|
|
203
291
|
// Defensive only: minOccurrences guarantees occurrences >= 2 on real data; a malformed-value fallback.
|
|
204
292
|
if (occurrences < 2) return costOnly('none');
|
|
293
|
+
// Same-shaped repeats whose details differ compute different data: nothing is known to be
|
|
294
|
+
// recomputed, so there is no time to claim.
|
|
295
|
+
if (finding.occurrencesIdentical === false) return costOnly('none');
|
|
296
|
+
// Each stage contributes the share of its operators inside the repeated subtree: a stage it
|
|
297
|
+
// shares with other operators (the consuming join, the join's other side) isn't all its
|
|
298
|
+
// time, and claiming whole stages let sibling groups claim the same stage twice. Findings
|
|
299
|
+
// built without the field (hand-made fixtures) count every linked stage whole.
|
|
300
|
+
const shares = (finding.stageShares ?? null) ;
|
|
205
301
|
const redundantFraction = (occurrences - 1) / occurrences;
|
|
206
302
|
const wasteMsByStage = new Map ();
|
|
207
303
|
for (const id of stageIds) {
|
|
208
304
|
const s = stages.get(id);
|
|
209
|
-
|
|
210
|
-
|
|
211
|
-
|
|
305
|
+
const share = shares ? (shares[id] ?? 0) : 1;
|
|
306
|
+
if (s && share > 0) {
|
|
307
|
+
// Time with tasks running, not submit-to-complete: a stage left waiting for cores
|
|
308
|
+
// (2491s open, 60s of tasks on a real log) isn't recomputing anything while it waits.
|
|
309
|
+
const activeMs = s.taskActiveMs ?? Math.max(0, (s.completedAt ?? 0) - (s.submittedAt ?? 0));
|
|
310
|
+
wasteMsByStage.set(id, activeMs * share * redundantFraction);
|
|
212
311
|
}
|
|
213
312
|
}
|
|
313
|
+
// No operator of the subtree ran in a known stage: no time to attribute.
|
|
314
|
+
if (wasteMsByStage.size === 0) return costOnly('none');
|
|
214
315
|
const totalWasteMs = [...wasteMsByStage.values()].reduce((sum, ms) => sum + ms, 0);
|
|
215
316
|
const rawWaste = { value: totalWasteMs, unit: 'ms' };
|
|
216
|
-
const est = estimateMultiStage(
|
|
317
|
+
const est = estimateMultiStage([...wasteMsByStage.keys()], wasteMsByStage, stages , occupancy);
|
|
217
318
|
if (!est) return costOnly('measured', rawWaste);
|
|
218
319
|
return { basis: est.basis, wallClock: est.wallClock, estimateMethod: 'measured', rawWaste };
|
|
219
320
|
}
|
|
@@ -222,16 +323,21 @@ function computeEstimateForFinding(
|
|
|
222
323
|
const stage = stages.get(finding.stageId);
|
|
223
324
|
if (!stage) return null;
|
|
224
325
|
const shuffleReadBytes = stage.shuffleReadBytes ?? 0;
|
|
225
|
-
const
|
|
226
|
-
// The
|
|
227
|
-
|
|
326
|
+
const modeledMs = (shuffleReadBytes / (SHUFFLE_THROUGHPUT_BPS * stageIoParallelism(stage))) * 1000;
|
|
327
|
+
// The link model can't see whether the reads stalled the tasks: capped at the fetch wait the
|
|
328
|
+
// tasks measured, the claim never exceeds what the stage spent blocked on the network.
|
|
329
|
+
const measuredMs = fetchWaitWallClockMs(stage);
|
|
330
|
+
const wasteMs = measuredMs == null ? modeledMs : Math.min(modeledMs, measuredMs);
|
|
331
|
+
// rawWaste: the measured byte volume behind the modeled figure.
|
|
332
|
+
return singleStageImpact(wasteMs, finding.stageId, stages, occupancy,
|
|
333
|
+
measuredMs != null && measuredMs < modeledMs ? 'measured' : 'modeled', { value: shuffleReadBytes, unit: 'bytes' });
|
|
228
334
|
}
|
|
229
335
|
case 'spill': {
|
|
230
336
|
if (finding.stageId == null) return null;
|
|
231
337
|
const stage = stages.get(finding.stageId);
|
|
232
338
|
if (!stage) return null;
|
|
233
339
|
const diskBytesSpilled = stage.diskBytesSpilled ?? 0;
|
|
234
|
-
const wasteMs = (diskBytesSpilled / SPILL_IO_THROUGHPUT_BPS) * 1000;
|
|
340
|
+
const wasteMs = (diskBytesSpilled / (SPILL_IO_THROUGHPUT_BPS * stageIoParallelism(stage))) * 1000;
|
|
235
341
|
// Surfaces the number the formula uses: the displayed metric is memoryBytesSpilled, but disk spill costs the I/O time.
|
|
236
342
|
return singleStageImpact(wasteMs, finding.stageId, stages, occupancy, 'modeled', { value: diskBytesSpilled, unit: 'bytes' });
|
|
237
343
|
}
|
|
@@ -239,9 +345,22 @@ function computeEstimateForFinding(
|
|
|
239
345
|
if (finding.stageId == null) return null;
|
|
240
346
|
const stage = stages.get(finding.stageId);
|
|
241
347
|
if (!stage) return null;
|
|
242
|
-
|
|
243
|
-
|
|
244
|
-
|
|
348
|
+
// The recommended fix is more partitions, which only helps a stage that ran fewer tasks than
|
|
349
|
+
// the cluster has cores: the time its tasks were running could then spread over up to
|
|
350
|
+
// totalCores (lowShuffleParallelism's shape). Time the stage sat open with no task running
|
|
351
|
+
// is queueing no partition count recovers. Splitting partitions splits the longest task
|
|
352
|
+
// too, hence TAIL_CLAIM's post-fix floor. Unknown cluster size: no defensible figure.
|
|
353
|
+
if (totalCores <= 0) return costOnly('modeled');
|
|
354
|
+
// A stage that read no input and no shuffle, its tasks idle waiting on an external system,
|
|
355
|
+
// gains nothing from more partitions (a 1-task JDBC count stage open 27 minutes on 5s of CPU
|
|
356
|
+
// was claimed 99% recoverable): claim 0. Stages that read bytes keep their claim.
|
|
357
|
+
const readBytes = (stage.inputBytes ?? 0) + (stage.shuffleReadBytes ?? 0);
|
|
358
|
+
const activeMs = typeof stage.taskActiveMs === 'number'
|
|
359
|
+
? stage.taskActiveMs
|
|
360
|
+
: Math.max(0, (stage.completedAt ?? 0) - (stage.submittedAt ?? 0));
|
|
361
|
+
const taskCount = stage.taskCount ?? 0;
|
|
362
|
+
const wasteMs = readBytes <= 0 && tasksMostlyIdle(stage) ? 0 : activeMs * Math.max(0, 1 - taskCount / totalCores);
|
|
363
|
+
return singleStageImpact(wasteMs, finding.stageId, stages, occupancy, 'modeled', { value: wasteMs, unit: 'ms' }, TAIL_CLAIM);
|
|
245
364
|
}
|
|
246
365
|
case 'partitionSizing': {
|
|
247
366
|
if (finding.stageId == null) return null;
|
|
@@ -275,12 +394,29 @@ function computeEstimateForFinding(
|
|
|
275
394
|
if (!stage) return null;
|
|
276
395
|
const taskCount = stage.taskCount ?? 0;
|
|
277
396
|
const excessTaskCount = Math.max(0, taskCount - Math.round(taskCount / 10));
|
|
397
|
+
const measured = measuredTaskOverhead(stage);
|
|
398
|
+
if (measured) {
|
|
399
|
+
// Coalescing to a tenth of the tasks removes the excess tasks' per-task overhead: core
|
|
400
|
+
// time spent in parallel, so wall-clock at the stage's achieved concurrency (floored at
|
|
401
|
+
// 1: a mostly-idle stage can't save more wall-clock than the task time it removes).
|
|
402
|
+
const wasteMs = (excessTaskCount * measured.perTaskMs) / Math.max(1, measured.concurrency);
|
|
403
|
+
return singleStageImpact(wasteMs, finding.stageId, stages, occupancy, 'measured', { value: wasteMs, unit: 'ms' });
|
|
404
|
+
}
|
|
278
405
|
const wasteMs = excessTaskCount * TASK_SCHEDULING_OVERHEAD_MS;
|
|
279
406
|
return singleStageImpact(wasteMs, finding.stageId, stages, occupancy, 'modeled', { value: wasteMs, unit: 'ms' });
|
|
280
407
|
}
|
|
281
408
|
case 'smallFiles': {
|
|
282
|
-
const
|
|
283
|
-
|
|
409
|
+
const fileMs = ((finding.fileCount ) ?? 0) * FILE_OPEN_OVERHEAD_MS;
|
|
410
|
+
// A read's files are opened by the scan's tasks, in parallel: spread the per-file cost over
|
|
411
|
+
// the most tasks its stages ran at once (91344 files x 10ms is 913s, claimed against a 117s
|
|
412
|
+
// stage that ran 314 tasks at once). A write keeps the serial sum: the job commit moves each
|
|
413
|
+
// output file on the driver, one after another.
|
|
414
|
+
const stageIds = finding.stageIds ;
|
|
415
|
+
let slots = 1;
|
|
416
|
+
if (finding.direction === 'read') {
|
|
417
|
+
for (const id of stageIds ?? []) slots = Math.max(slots, stages.get(id)?.peakConcurrentTasks ?? 1);
|
|
418
|
+
}
|
|
419
|
+
return stageMappableWasteOrCostOnly(fileMs / slots, stageIds, stages, occupancy);
|
|
284
420
|
}
|
|
285
421
|
case 'overBroadcast': {
|
|
286
422
|
// metric: 'broadcastBytes', value: <bytes>.
|
|
@@ -385,7 +521,7 @@ function computeEstimateForFinding(
|
|
|
385
521
|
export function estimateImpact(findings , stages , totalCores = 0) {
|
|
386
522
|
const occupancy = computeOccupancy(stages , totalCores);
|
|
387
523
|
for (const f of findings) {
|
|
388
|
-
const estimate = computeEstimateForFinding(f, stages, occupancy);
|
|
524
|
+
const estimate = computeEstimateForFinding(f, stages, occupancy, totalCores);
|
|
389
525
|
if (estimate) f.impactEstimate = estimate;
|
|
390
526
|
}
|
|
391
527
|
return findings;
|
package/vendor-core/mcp-tools.js
CHANGED
|
@@ -228,6 +228,11 @@ export function getFindingDocumentation(type ) {
|
|
|
228
228
|
if (tuningPath && anchor && existsSync(tuningPath)) {
|
|
229
229
|
const tuningContent = readFileSync(tuningPath, 'utf8');
|
|
230
230
|
tuningDoc = { anchor, title: extractDocTitle(tuningContent), content: tuningContent };
|
|
231
|
+
} else if (anchor && !slug) {
|
|
232
|
+
// A section hosted on a chapter page (e.g. autoscaling churn on cluster-config): return the
|
|
233
|
+
// owning chapter, the same page the web view's docs link opens.
|
|
234
|
+
const entry = findNavEntry(pageForAnchor(anchor.replace(/^#/, '')));
|
|
235
|
+
if (entry) tuningDoc = { anchor, title: entry.title, content: readNavEntryContent(entry) };
|
|
231
236
|
}
|
|
232
237
|
|
|
233
238
|
return { type, name: titleCase(label), detectionDoc, tuningDoc };
|
|
@@ -235,21 +240,29 @@ export function getFindingDocumentation(type ) {
|
|
|
235
240
|
|
|
236
241
|
const CHAPTERS_NAV_FILE = join(DOCS_CONTENT_DIR, 'chapters', 'nav-index.json');
|
|
237
242
|
|
|
243
|
+
|
|
244
|
+
|
|
245
|
+
function findNavEntry(page ) {
|
|
246
|
+
const nav = JSON.parse(readFileSync(CHAPTERS_NAV_FILE, 'utf8')) ;
|
|
247
|
+
return nav.find((e) => e.anchor === page);
|
|
248
|
+
}
|
|
249
|
+
|
|
250
|
+
function readNavEntryContent(entry ) {
|
|
251
|
+
const dir = entry.store === 'tuning' ? 'tuning' : 'chapters';
|
|
252
|
+
return readFileSync(join(DOCS_CONTENT_DIR, dir, `${entry.slug}.md`), 'utf8');
|
|
253
|
+
}
|
|
254
|
+
|
|
238
255
|
|
|
239
256
|
|
|
240
257
|
/** Full tuning-reference markdown for one doc anchor, run-independent. Resolves the anchor to its
|
|
241
258
|
* owning page via pageForAnchor (so '#metric-task-duration' returns the 'metrics' page), looks it up
|
|
242
|
-
* in the
|
|
259
|
+
* in the generated nav-index, and reads the markdown from the same docs-content store the website
|
|
243
260
|
* renders from. The general-chapter counterpart to getFindingDocumentation (keyed by finding type). */
|
|
244
261
|
export function getReferenceDoc(anchor ) {
|
|
245
262
|
const page = pageForAnchor(String(anchor).replace(/^#/, ''));
|
|
246
|
-
const
|
|
247
|
-
;
|
|
248
|
-
const entry = nav.find((e) => e.anchor === page);
|
|
263
|
+
const entry = findNavEntry(page);
|
|
249
264
|
if (!entry) throw mcpError('invalid-anchor', `Unknown reference anchor: ${anchor}`);
|
|
250
|
-
|
|
251
|
-
const content = readFileSync(join(DOCS_CONTENT_DIR, dir, `${entry.slug}.md`), 'utf8');
|
|
252
|
-
return { anchor: page, title: entry.title, content };
|
|
265
|
+
return { anchor: page, title: entry.title, content: readNavEntryContent(entry) };
|
|
253
266
|
}
|
|
254
267
|
|
|
255
268
|
export function getFindingEvidence(
|
package/vendor-core/occupancy.js
CHANGED
|
@@ -117,6 +117,9 @@ export function clipToCeiling(wasteMsClaimed , stage , cei
|
|
|
117
117
|
|
|
118
118
|
|
|
119
119
|
|
|
120
|
+
|
|
121
|
+
|
|
122
|
+
|
|
120
123
|
|
|
121
124
|
|
|
122
125
|
export function computeOccupancy(
|
|
@@ -128,7 +131,11 @@ export function computeOccupancy(
|
|
|
128
131
|
for (const s of stages.values()) {
|
|
129
132
|
const g = gate.get(s.id);
|
|
130
133
|
if (g === undefined) continue; // excluded from the sweep (duration <= 0)
|
|
131
|
-
info.set(s.id, {
|
|
134
|
+
info.set(s.id, {
|
|
135
|
+
gate: g,
|
|
136
|
+
ceiling: computeCeiling(s, totalCores),
|
|
137
|
+
coreWorkFloor: totalCores > 0 ? (s.executorRunTime ?? 0) / totalCores : 0,
|
|
138
|
+
});
|
|
132
139
|
}
|
|
133
140
|
return info;
|
|
134
141
|
}
|
|
@@ -140,6 +147,60 @@ const SERIAL_GATE_THRESHOLD = 0.999;
|
|
|
140
147
|
|
|
141
148
|
|
|
142
149
|
|
|
150
|
+
|
|
151
|
+
|
|
152
|
+
|
|
153
|
+
|
|
154
|
+
|
|
155
|
+
|
|
156
|
+
|
|
157
|
+
|
|
158
|
+
|
|
159
|
+
|
|
160
|
+
// Wall-clock a skew/straggler fix recovers: finalizeStage's task-level replay
|
|
161
|
+
// (tailReplayRecoveryMs, computeTailReplayRecoveryMs) when the stage carries it. Without it (a
|
|
162
|
+
// stage built by hand, as in detector tests) an estimate from the slowest task's excess over the
|
|
163
|
+
// median: a lone straggler costs that excess; a tail of many slow tasks costs its summed excess
|
|
164
|
+
// (stragglerExcessMs) spread over the slots the stage had (peakConcurrentTasks), and the replay
|
|
165
|
+
// recovers about the larger of the two.
|
|
166
|
+
export function tailRecoveryMs(stage , singleTaskExcessMs ) {
|
|
167
|
+
if (stage.tailReplayRecoveryMs != null) return stage.tailReplayRecoveryMs;
|
|
168
|
+
const excessMs = stage.stragglerExcessMs ?? 0;
|
|
169
|
+
const slots = stage.peakConcurrentTasks ?? 0;
|
|
170
|
+
if (excessMs <= 0 || slots <= 0) return singleTaskExcessMs;
|
|
171
|
+
return Math.max(singleTaskExcessMs, excessMs / slots);
|
|
172
|
+
}
|
|
173
|
+
|
|
174
|
+
// Task time a skew/straggler fix removes, by the same measure: the tasks over 4x P50 capped at the
|
|
175
|
+
// median (stragglerExcessMs), or the slowest task's excess when that alone is larger.
|
|
176
|
+
export function tailRemovedWorkMs(stage , singleTaskExcessMs ) {
|
|
177
|
+
return Math.max(singleTaskExcessMs, stage.stragglerExcessMs ?? 0);
|
|
178
|
+
}
|
|
179
|
+
|
|
180
|
+
// Longest task a straggler fix leaves: every task over 4x P50 comes down to the median, so the
|
|
181
|
+
// longest one at or under that (finalizeStage's longestNonStragglerMs). Without stragglers (a
|
|
182
|
+
// speculation-driven finding) or the field (older snapshots), the median.
|
|
183
|
+
export function stragglerFixLongestTaskMs(stage ) {
|
|
184
|
+
const p50 = stage.taskDurationP50 ?? 0;
|
|
185
|
+
if (!((stage.stragglerCount ?? 0) > 0) || stage.longestNonStragglerMs == null) return p50;
|
|
186
|
+
return Math.max(p50, stage.longestNonStragglerMs);
|
|
187
|
+
}
|
|
188
|
+
|
|
189
|
+
|
|
190
|
+
|
|
191
|
+
|
|
192
|
+
|
|
193
|
+
|
|
194
|
+
|
|
195
|
+
|
|
196
|
+
|
|
197
|
+
|
|
198
|
+
|
|
199
|
+
|
|
200
|
+
|
|
201
|
+
|
|
202
|
+
|
|
203
|
+
|
|
143
204
|
/**
|
|
144
205
|
* Per-finding estimate for a single-stage waste claim. Returns null when the
|
|
145
206
|
* stage was excluded from the sweep (duration <= 0): callers must fall back
|
|
@@ -151,11 +212,19 @@ export function estimateSingleStage(
|
|
|
151
212
|
stageId ,
|
|
152
213
|
stages ,
|
|
153
214
|
info ,
|
|
215
|
+
{ shortensLongestTask = false, removedCoreWorkMs = 0, longestTaskAfterFixMs = 0 } = {},
|
|
154
216
|
) {
|
|
155
217
|
const stage = stages.get(stageId);
|
|
156
218
|
const stageInfo = info.get(stageId);
|
|
157
219
|
if (!stage || !stageInfo) return null;
|
|
158
|
-
const
|
|
220
|
+
const coreWork = stage.executorRunTime ?? 0;
|
|
221
|
+
const coreWorkFloor = removedCoreWorkMs > 0 && coreWork > 0
|
|
222
|
+
? stageInfo.coreWorkFloor * Math.max(0, 1 - removedCoreWorkMs / coreWork)
|
|
223
|
+
: stageInfo.coreWorkFloor;
|
|
224
|
+
const ceiling = shortensLongestTask
|
|
225
|
+
? Math.max((stage.taskDurationMax ?? 0) - wasteMsClaimed, longestTaskAfterFixMs, coreWorkFloor)
|
|
226
|
+
: stageInfo.ceiling;
|
|
227
|
+
const clipped = clipToCeiling(wasteMsClaimed, stage, ceiling);
|
|
159
228
|
if (stageInfo.gate >= SERIAL_GATE_THRESHOLD) {
|
|
160
229
|
return { basis: 'serial', wallClock: { low: clipped, high: clipped } };
|
|
161
230
|
}
|