sparkforensics-mcp 0.2.2 → 0.2.4

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (42) hide show
  1. package/bin/sparkforensics-mcp.mjs +11 -5
  2. package/package.json +3 -3
  3. package/vendor-core/cli/collect-run.js +3 -2
  4. package/vendor-core/cli/native-zstd.js +351 -0
  5. package/vendor-core/detectors.js +243 -53
  6. package/vendor-core/docs-config.js +34 -8
  7. package/vendor-core/docs-content/chapters/03-memory-model.md +39 -0
  8. package/vendor-core/docs-content/chapters/11-cluster-config.md +40 -0
  9. package/vendor-core/docs-content/detection/gc.md +2 -0
  10. package/vendor-core/docs-content/detection/host.md +2 -1
  11. package/vendor-core/docs-content/detection/plan.md +3 -1
  12. package/vendor-core/docs-content/detection/shape.md +2 -1
  13. package/vendor-core/docs-content/detection/shfl.md +2 -1
  14. package/vendor-core/docs-content/detection/spill.md +1 -1
  15. package/vendor-core/docs-content/detection/strag.md +2 -1
  16. package/vendor-core/docs-content/detection/tiny.md +2 -1
  17. package/vendor-core/docs-content/tuning/failures.md +1 -1
  18. package/vendor-core/docs-content/tuning/gc.md +11 -4
  19. package/vendor-core/docs-content/tuning/shuffle.md +25 -5
  20. package/vendor-core/docs-content/tuning/skew.md +14 -6
  21. package/vendor-core/docs-content/tuning/small-files.md +12 -7
  22. package/vendor-core/docs-content/tuning/straggler.md +34 -0
  23. package/vendor-core/docs-content/tuning/tiny-tasks.md +1 -1
  24. package/vendor-core/docs-content/tuning/utilization.md +57 -7
  25. package/vendor-core/docs-content/upstream.json +4 -0
  26. package/vendor-core/event-handlers.js +321 -69
  27. package/vendor-core/event-schemas.js +8 -6
  28. package/vendor-core/evidence-report.js +3 -1
  29. package/vendor-core/impact-estimator.js +170 -34
  30. package/vendor-core/mcp-tools.js +20 -7
  31. package/vendor-core/occupancy.js +71 -2
  32. package/vendor-core/parser-worker.js +56 -24
  33. package/vendor-core/plan-summary.js +5 -1
  34. package/vendor-core/run-comparison.js +1 -15
  35. package/vendor-core/shs-fetch.js +18 -7
  36. package/vendor-core/shs-load.js +2 -1
  37. package/vendor-core/stage-quantiles.js +111 -3
  38. package/vendor-core/string-hash.js +15 -0
  39. package/vendor-core/types.js +14 -1
  40. package/vendor-core/vendor/fzstd.js +94 -18
  41. package/vendor-core/zstd-worker-client.js +180 -0
  42. package/vendor-core/zstd-worker.js +103 -0
@@ -1,14 +1,76 @@
1
1
 
2
- import { computeOccupancy, estimateSingleStage, estimateMultiStage, } from './occupancy.js';
3
- import { detectorCatalog } from './detectors.js';
2
+ import { nsToMs } from './format-utils.js';
3
+ import {
4
+ computeOccupancy, estimateSingleStage, estimateMultiStage, tailRecoveryMs, tailRemovedWorkMs, stragglerFixLongestTaskMs,
5
+
6
+ } from './occupancy.js';
4
7
 
5
- // Assumed shuffle-network throughput, ~1 Gbps. Starting assumption, unvalidated.
8
+ // Assumed shuffle-network throughput per executor link, ~1 Gbps. Starting assumption, unvalidated.
6
9
  const SHUFFLE_THROUGHPUT_BPS = 125_000_000;
7
- // Assumed disk I/O throughput for spilled data, ~200 MB/s (conservative HDD/SSD blend).
10
+ // Assumed disk I/O throughput per executor for spilled data, ~200 MB/s (conservative HDD/SSD blend).
8
11
  const SPILL_IO_THROUGHPUT_BPS = 200_000_000;
12
+
13
+ // A stage's own per-task overhead: task wall time (launch to finish, summed per executor by
14
+ // finalizeStage) minus executorRunTime, i.e. deserialization, result serialization and the
15
+ // launch round trip that coalescing tasks removes. `concurrency` is the stage's achieved task
16
+ // concurrency (task time / stage duration). Null when the stage has no such data or no overhead.
17
+ function measuredTaskOverhead(stage ) {
18
+ const executorStats = Array.isArray(stage.executorStats)
19
+ ? (stage.executorStats ) : [];
20
+ const taskTimeMs = executorStats.reduce((sum, e) => sum + (e.totalDuration ?? 0), 0);
21
+ const taskCount = stage.taskCount ?? 0;
22
+ const durationMs = (stage.completedAt ?? 0) - (stage.submittedAt ?? 0);
23
+ const overheadMs = taskTimeMs - (stage.executorRunTime ?? 0);
24
+ if (taskTimeMs <= 0 || taskCount <= 0 || durationMs <= 0 || overheadMs <= 0) return null;
25
+ return { perTaskMs: overheadMs / taskCount, concurrency: taskTimeMs / durationMs };
26
+ }
27
+
28
+ // Both constants above are single-device figures (one NIC, one local disk). A stage's shuffle
29
+ // reads and spills are spread over every executor that ran its tasks, each moving its own share
30
+ // in parallel, so the stage's aggregate bandwidth scales with that executor count. Dividing a
31
+ // stage-wide byte total by one device's bandwidth modeled the whole cluster as one link: on 14
32
+ // real logs that claimed up to 1,870s of shuffle time on stages whose tasks measured ~0s of
33
+ // shuffle fetch wait (222 of 284 shuffle findings). No executor data falls back to one device.
34
+ function stageIoParallelism(stage ) {
35
+ const executors = Array.isArray(stage.executorStats) ? stage.executorStats.length : 0;
36
+ return Math.max(1, executors);
37
+ }
38
+
39
+ // Wall-clock the stage's tasks spent blocked fetching shuffle blocks: fetchWaitTime is a
40
+ // cross-task sum like executorRunTime, so dividing by the stage's average concurrency converts it
41
+ // (the gc estimate's conversion). Null when the stage has no run time or duration to convert with.
42
+ // On 14 real logs the link model claimed 2780s over 284 shuffle findings, 256s once capped at
43
+ // this, with about zero fetch wait on 5 of the 10 non-info ones: reads that overlapped compute
44
+ // stalled nothing.
45
+ function fetchWaitWallClockMs(stage ) {
46
+ const fetchWaitMs = stage.fetchWaitTime;
47
+ const runTimeMs = stage.executorRunTime ?? 0;
48
+ const durationMs = (stage.completedAt ?? 0) - (stage.submittedAt ?? 0);
49
+ if (typeof fetchWaitMs !== 'number' || runTimeMs <= 0 || durationMs <= 0) return null;
50
+ return fetchWaitMs / (runTimeMs / durationMs);
51
+ }
52
+
53
+ // Wall-clock a stage's wasted (retried) attempts cost it. Each one delayed only its own task, and
54
+ // attempts of different tasks ran side by side: one lost executor fails every task it was running
55
+ // at once (4 wasted attempts of 36.6s each, all first attempts, on a real stage that ran 41 tasks
56
+ // at once, claimed as 146.6s). So the stage lost at most its longest retry chain, the most attempts
57
+ // one task wasted (its highest attempt number + 1) times the mean wasted attempt, or their summed
58
+ // time spread over its slots, whichever is larger. Without a sample of every wasted attempt
59
+ // (retryTaskSamples is capped), the chain isn't known: the summed time.
60
+ function retryWallClockMs(stage ) {
61
+ const totalMs = (stage.retryWasteMs ) ?? 0;
62
+ const attempts = (stage.wastedAttempts ) ?? 0;
63
+ const samples = Array.isArray(stage.retryTaskSamples)
64
+ ? (stage.retryTaskSamples ) : [];
65
+ if (totalMs <= 0 || attempts <= 0 || samples.length < attempts) return totalMs;
66
+ const chain = samples.reduce((longest, s) => Math.max(longest, (s.attemptNumber ?? 0) + 1), 1);
67
+ const slots = Math.max(1, stage.peakConcurrentTasks ?? 1);
68
+ return Math.min(totalMs, Math.max((chain * totalMs) / attempts, totalMs / slots));
69
+ }
9
70
  // Spark's classic recommended shuffle partition size.
10
71
  const IDEAL_BYTES_PER_PARTITION_TASK = 128 * 1024 * 1024;
11
- // Assumed per-task scheduling/launch overhead.
72
+ // Assumed per-task scheduling/launch overhead: the fallback when a stage lacks the per-executor
73
+ // task-time sums measuredTaskOverhead needs.
12
74
  const TASK_SCHEDULING_OVERHEAD_MS = 50;
13
75
  // Assumed per-file open latency (small-file overhead).
14
76
  const FILE_OPEN_OVERHEAD_MS = 10;
@@ -21,15 +83,28 @@ const EXECUTOR_STARTUP_OVERHEAD_MS = 15000;
21
83
  // Assumed re-read throughput, shared by cachingOpportunity and cacheUtilization.
22
84
  const RE_READ_THROUGHPUT_BPS = 125_000_000;
23
85
 
24
- // stageSlowness flags a stage at `infoMin` minutes; that's the floor for the waste this estimate
25
- // reports. Read from the detector's catalog entry so the two stay in sync automatically.
26
- const STAGE_SLOWNESS_THRESHOLD_MINUTES = (() => {
27
- const infoMin = detectorCatalog().find((d) => d.type === 'stageSlowness')?.thresholds?.infoMin;
28
- if (typeof infoMin !== 'number') {
29
- throw new Error("impact-estimator: stageSlowness detector's 'infoMin' threshold not found in detectorCatalog()");
30
- }
31
- return infoMin;
32
- })();
86
+ // Below this share of executorRunTime spent on CPU, a stage's tasks were idle, waiting on something
87
+ // outside Spark: on 14 real logs (2026-09-23) every non-Python stage under 1% was a JDBC read, a
88
+ // file listing or a Delta log read, while file writes, which more partitions do parallelize,
89
+ // start at 2%.
90
+ const IDLE_CPU_SHARE_MAX = 0.01;
91
+
92
+ // True when the stage's tasks spent under IDLE_CPU_SHARE_MAX of their run time on CPU. False when
93
+ // the share can't be trusted: no CPU time recorded (older Spark logs omit the metric), or Python
94
+ // code run through PythonRDD, whose worker-process CPU executorCpuTime (the JVM task thread's)
95
+ // never counts (such stages read 0.1% on the same logs while computing).
96
+ function tasksMostlyIdle(stage ) {
97
+ const runMs = stage.executorRunTime ?? 0;
98
+ const cpuMs = nsToMs(stage.executorCpuTime ?? 0);
99
+ if (runMs <= 0 || cpuMs <= 0) return false;
100
+ if (/PythonRDD/.test(stage.name ?? '') || /org\.apache\.spark\.api\.python\./.test(stage.details ?? '')) return false;
101
+ return cpuMs / runMs < IDLE_CPU_SHARE_MAX;
102
+ }
103
+
104
+ // skew and straggler claim time off the stage's longest task itself, so the occupancy clip must
105
+ // not floor them at that same task (see estimateSingleStage). detectors.ts's clippedWasteMs gates
106
+ // both detectors on the same option so the firing floor and the displayed estimate agree.
107
+ const TAIL_CLAIM = { shortensLongestTask: true };
33
108
 
34
109
  // No quantifiable magnitude -> 'informational'; a rawWaste figure with no stage window ->
35
110
  // 'resourceOnly'. Never a fake {low:0, high:0}: wallClock is null in both cases.
@@ -46,8 +121,9 @@ function singleStageImpact(
46
121
  occupancy ,
47
122
  estimateMethod ,
48
123
  rawWaste ,
124
+ opts ,
49
125
  ) {
50
- const est = estimateSingleStage(wasteMs, stageId, stages , occupancy);
126
+ const est = estimateSingleStage(wasteMs, stageId, stages , occupancy, opts);
51
127
  if (est) return { basis: est.basis, wallClock: est.wallClock, estimateMethod, rawWaste };
52
128
  return costOnly(estimateMethod, rawWaste); // stage excluded from the sweep (duration <= 0)
53
129
  }
@@ -77,6 +153,7 @@ function computeEstimateForFinding(
77
153
  finding ,
78
154
  stages ,
79
155
  occupancy ,
156
+ totalCores ,
80
157
  ) {
81
158
  switch (finding.type) {
82
159
  case 'retryWaste': {
@@ -85,7 +162,9 @@ function computeEstimateForFinding(
85
162
  const stage = stages.get(finding.stageId);
86
163
  if (!stage) return null;
87
164
  const wasteMs = (stage.retryWasteMs ) ?? 0;
88
- return singleStageImpact(wasteMs, finding.stageId, stages, occupancy, 'measured', { value: wasteMs, unit: 'ms' });
165
+ const wallClockMs = retryWallClockMs(stage);
166
+ return singleStageImpact(wallClockMs, finding.stageId, stages, occupancy,
167
+ wallClockMs === wasteMs ? 'measured' : 'modeled', { value: wasteMs, unit: 'ms' });
89
168
  }
90
169
  case 'speculationWaste': {
91
170
  if (finding.stageId == null) return null;
@@ -103,6 +182,9 @@ function computeEstimateForFinding(
103
182
  return { basis: 'serial', wallClock: { low: wasteMs, high: wasteMs }, estimateMethod: 'measured' };
104
183
  }
105
184
  case 'gc': {
185
+ // The low-GC branch is an over-provisioning signal whose fix (less executor memory) raises
186
+ // GC rather than recovering it: the stage's GC time is no saving there, so no waste model.
187
+ if (finding.direction === 'low') return costOnly('none');
106
188
  if (finding.stageId == null) return null;
107
189
  const stage = stages.get(finding.stageId);
108
190
  if (!stage) return null;
@@ -128,8 +210,11 @@ function computeEstimateForFinding(
128
210
  const p50 = stage.taskDurationP50 ?? 0;
129
211
  // computeSkewRatio's own metric labels (src/detectors.ts): 'P95/median' or 'max/median'.
130
212
  const usesP95Branch = finding.metric === 'P95/median';
131
- const wasteMs = Math.max(0, usesP95Branch ? (stage.taskDurationP95 ?? 0) - p50 : (stage.taskDurationMax ?? 0) - p50);
132
- return singleStageImpact(wasteMs, finding.stageId, stages, occupancy, 'measured', { value: wasteMs, unit: 'ms' });
213
+ const singleDelta = Math.max(0, usesP95Branch ? (stage.taskDurationP95 ?? 0) - p50 : (stage.taskDurationMax ?? 0) - p50);
214
+ const wasteMs = tailRecoveryMs(stage, singleDelta);
215
+ // Fixing the skew still waits on the longest task it leaves, as for straggler.
216
+ return singleStageImpact(wasteMs, finding.stageId, stages, occupancy, 'measured', { value: wasteMs, unit: 'ms' },
217
+ { ...TAIL_CLAIM, removedCoreWorkMs: tailRemovedWorkMs(stage, singleDelta), longestTaskAfterFixMs: stragglerFixLongestTaskMs(stage) });
133
218
  }
134
219
  case 'straggler':
135
220
  case 'stageShape': {
@@ -171,8 +256,11 @@ function computeEstimateForFinding(
171
256
  if (finding.stageId == null) return null;
172
257
  const stage = stages.get(finding.stageId);
173
258
  if (!stage) return null;
174
- const wasteMs = Math.max(0, (stage.taskDurationMax ?? 0) - (stage.taskDurationP50 ?? 0));
175
- return singleStageImpact(wasteMs, finding.stageId, stages, occupancy, 'measured', { value: wasteMs, unit: 'ms' });
259
+ const longestTaskAfterFixMs = stragglerFixLongestTaskMs(stage);
260
+ const singleDelta = Math.max(0, (stage.taskDurationMax ?? 0) - longestTaskAfterFixMs);
261
+ const wasteMs = tailRecoveryMs(stage, singleDelta);
262
+ return singleStageImpact(wasteMs, finding.stageId, stages, occupancy, 'measured', { value: wasteMs, unit: 'ms' },
263
+ { ...TAIL_CLAIM, removedCoreWorkMs: tailRemovedWorkMs(stage, singleDelta), longestTaskAfterFixMs });
176
264
  }
177
265
  case 'slowHost': {
178
266
  // Three duration-based shapes, each carrying its absolute-ms figure under a different field
@@ -202,18 +290,31 @@ function computeEstimateForFinding(
202
290
  const occurrences = typeof finding.value === 'number' ? finding.value : 0;
203
291
  // Defensive only: minOccurrences guarantees occurrences >= 2 on real data; a malformed-value fallback.
204
292
  if (occurrences < 2) return costOnly('none');
293
+ // Same-shaped repeats whose details differ compute different data: nothing is known to be
294
+ // recomputed, so there is no time to claim.
295
+ if (finding.occurrencesIdentical === false) return costOnly('none');
296
+ // Each stage contributes the share of its operators inside the repeated subtree: a stage it
297
+ // shares with other operators (the consuming join, the join's other side) isn't all its
298
+ // time, and claiming whole stages let sibling groups claim the same stage twice. Findings
299
+ // built without the field (hand-made fixtures) count every linked stage whole.
300
+ const shares = (finding.stageShares ?? null) ;
205
301
  const redundantFraction = (occurrences - 1) / occurrences;
206
302
  const wasteMsByStage = new Map ();
207
303
  for (const id of stageIds) {
208
304
  const s = stages.get(id);
209
- if (s) {
210
- const durationMs = Math.max(0, (s.completedAt ?? 0) - (s.submittedAt ?? 0));
211
- wasteMsByStage.set(id, durationMs * redundantFraction);
305
+ const share = shares ? (shares[id] ?? 0) : 1;
306
+ if (s && share > 0) {
307
+ // Time with tasks running, not submit-to-complete: a stage left waiting for cores
308
+ // (2491s open, 60s of tasks on a real log) isn't recomputing anything while it waits.
309
+ const activeMs = s.taskActiveMs ?? Math.max(0, (s.completedAt ?? 0) - (s.submittedAt ?? 0));
310
+ wasteMsByStage.set(id, activeMs * share * redundantFraction);
212
311
  }
213
312
  }
313
+ // No operator of the subtree ran in a known stage: no time to attribute.
314
+ if (wasteMsByStage.size === 0) return costOnly('none');
214
315
  const totalWasteMs = [...wasteMsByStage.values()].reduce((sum, ms) => sum + ms, 0);
215
316
  const rawWaste = { value: totalWasteMs, unit: 'ms' };
216
- const est = estimateMultiStage(stageIds, wasteMsByStage, stages , occupancy);
317
+ const est = estimateMultiStage([...wasteMsByStage.keys()], wasteMsByStage, stages , occupancy);
217
318
  if (!est) return costOnly('measured', rawWaste);
218
319
  return { basis: est.basis, wallClock: est.wallClock, estimateMethod: 'measured', rawWaste };
219
320
  }
@@ -222,16 +323,21 @@ function computeEstimateForFinding(
222
323
  const stage = stages.get(finding.stageId);
223
324
  if (!stage) return null;
224
325
  const shuffleReadBytes = stage.shuffleReadBytes ?? 0;
225
- const wasteMs = (shuffleReadBytes / SHUFFLE_THROUGHPUT_BPS) * 1000;
226
- // The measured byte volume driving the modeled ms figure above.
227
- return singleStageImpact(wasteMs, finding.stageId, stages, occupancy, 'modeled', { value: shuffleReadBytes, unit: 'bytes' });
326
+ const modeledMs = (shuffleReadBytes / (SHUFFLE_THROUGHPUT_BPS * stageIoParallelism(stage))) * 1000;
327
+ // The link model can't see whether the reads stalled the tasks: capped at the fetch wait the
328
+ // tasks measured, the claim never exceeds what the stage spent blocked on the network.
329
+ const measuredMs = fetchWaitWallClockMs(stage);
330
+ const wasteMs = measuredMs == null ? modeledMs : Math.min(modeledMs, measuredMs);
331
+ // rawWaste: the measured byte volume behind the modeled figure.
332
+ return singleStageImpact(wasteMs, finding.stageId, stages, occupancy,
333
+ measuredMs != null && measuredMs < modeledMs ? 'measured' : 'modeled', { value: shuffleReadBytes, unit: 'bytes' });
228
334
  }
229
335
  case 'spill': {
230
336
  if (finding.stageId == null) return null;
231
337
  const stage = stages.get(finding.stageId);
232
338
  if (!stage) return null;
233
339
  const diskBytesSpilled = stage.diskBytesSpilled ?? 0;
234
- const wasteMs = (diskBytesSpilled / SPILL_IO_THROUGHPUT_BPS) * 1000;
340
+ const wasteMs = (diskBytesSpilled / (SPILL_IO_THROUGHPUT_BPS * stageIoParallelism(stage))) * 1000;
235
341
  // Surfaces the number the formula uses: the displayed metric is memoryBytesSpilled, but disk spill costs the I/O time.
236
342
  return singleStageImpact(wasteMs, finding.stageId, stages, occupancy, 'modeled', { value: diskBytesSpilled, unit: 'bytes' });
237
343
  }
@@ -239,9 +345,22 @@ function computeEstimateForFinding(
239
345
  if (finding.stageId == null) return null;
240
346
  const stage = stages.get(finding.stageId);
241
347
  if (!stage) return null;
242
- const stageDurationMs = (stage.completedAt ?? 0) - (stage.submittedAt ?? 0);
243
- const wasteMs = Math.max(0, stageDurationMs - STAGE_SLOWNESS_THRESHOLD_MINUTES * 60000);
244
- return singleStageImpact(wasteMs, finding.stageId, stages, occupancy, 'modeled', { value: wasteMs, unit: 'ms' });
348
+ // The recommended fix is more partitions, which only helps a stage that ran fewer tasks than
349
+ // the cluster has cores: the time its tasks were running could then spread over up to
350
+ // totalCores (lowShuffleParallelism's shape). Time the stage sat open with no task running
351
+ // is queueing no partition count recovers. Splitting partitions splits the longest task
352
+ // too, hence TAIL_CLAIM's post-fix floor. Unknown cluster size: no defensible figure.
353
+ if (totalCores <= 0) return costOnly('modeled');
354
+ // A stage that read no input and no shuffle, its tasks idle waiting on an external system,
355
+ // gains nothing from more partitions (a 1-task JDBC count stage open 27 minutes on 5s of CPU
356
+ // was claimed 99% recoverable): claim 0. Stages that read bytes keep their claim.
357
+ const readBytes = (stage.inputBytes ?? 0) + (stage.shuffleReadBytes ?? 0);
358
+ const activeMs = typeof stage.taskActiveMs === 'number'
359
+ ? stage.taskActiveMs
360
+ : Math.max(0, (stage.completedAt ?? 0) - (stage.submittedAt ?? 0));
361
+ const taskCount = stage.taskCount ?? 0;
362
+ const wasteMs = readBytes <= 0 && tasksMostlyIdle(stage) ? 0 : activeMs * Math.max(0, 1 - taskCount / totalCores);
363
+ return singleStageImpact(wasteMs, finding.stageId, stages, occupancy, 'modeled', { value: wasteMs, unit: 'ms' }, TAIL_CLAIM);
245
364
  }
246
365
  case 'partitionSizing': {
247
366
  if (finding.stageId == null) return null;
@@ -275,12 +394,29 @@ function computeEstimateForFinding(
275
394
  if (!stage) return null;
276
395
  const taskCount = stage.taskCount ?? 0;
277
396
  const excessTaskCount = Math.max(0, taskCount - Math.round(taskCount / 10));
397
+ const measured = measuredTaskOverhead(stage);
398
+ if (measured) {
399
+ // Coalescing to a tenth of the tasks removes the excess tasks' per-task overhead: core
400
+ // time spent in parallel, so wall-clock at the stage's achieved concurrency (floored at
401
+ // 1: a mostly-idle stage can't save more wall-clock than the task time it removes).
402
+ const wasteMs = (excessTaskCount * measured.perTaskMs) / Math.max(1, measured.concurrency);
403
+ return singleStageImpact(wasteMs, finding.stageId, stages, occupancy, 'measured', { value: wasteMs, unit: 'ms' });
404
+ }
278
405
  const wasteMs = excessTaskCount * TASK_SCHEDULING_OVERHEAD_MS;
279
406
  return singleStageImpact(wasteMs, finding.stageId, stages, occupancy, 'modeled', { value: wasteMs, unit: 'ms' });
280
407
  }
281
408
  case 'smallFiles': {
282
- const wasteMs = ((finding.fileCount ) ?? 0) * FILE_OPEN_OVERHEAD_MS;
283
- return stageMappableWasteOrCostOnly(wasteMs, finding.stageIds , stages, occupancy);
409
+ const fileMs = ((finding.fileCount ) ?? 0) * FILE_OPEN_OVERHEAD_MS;
410
+ // A read's files are opened by the scan's tasks, in parallel: spread the per-file cost over
411
+ // the most tasks its stages ran at once (91344 files x 10ms is 913s, claimed against a 117s
412
+ // stage that ran 314 tasks at once). A write keeps the serial sum: the job commit moves each
413
+ // output file on the driver, one after another.
414
+ const stageIds = finding.stageIds ;
415
+ let slots = 1;
416
+ if (finding.direction === 'read') {
417
+ for (const id of stageIds ?? []) slots = Math.max(slots, stages.get(id)?.peakConcurrentTasks ?? 1);
418
+ }
419
+ return stageMappableWasteOrCostOnly(fileMs / slots, stageIds, stages, occupancy);
284
420
  }
285
421
  case 'overBroadcast': {
286
422
  // metric: 'broadcastBytes', value: <bytes>.
@@ -385,7 +521,7 @@ function computeEstimateForFinding(
385
521
  export function estimateImpact(findings , stages , totalCores = 0) {
386
522
  const occupancy = computeOccupancy(stages , totalCores);
387
523
  for (const f of findings) {
388
- const estimate = computeEstimateForFinding(f, stages, occupancy);
524
+ const estimate = computeEstimateForFinding(f, stages, occupancy, totalCores);
389
525
  if (estimate) f.impactEstimate = estimate;
390
526
  }
391
527
  return findings;
@@ -228,6 +228,11 @@ export function getFindingDocumentation(type ) {
228
228
  if (tuningPath && anchor && existsSync(tuningPath)) {
229
229
  const tuningContent = readFileSync(tuningPath, 'utf8');
230
230
  tuningDoc = { anchor, title: extractDocTitle(tuningContent), content: tuningContent };
231
+ } else if (anchor && !slug) {
232
+ // A section hosted on a chapter page (e.g. autoscaling churn on cluster-config): return the
233
+ // owning chapter, the same page the web view's docs link opens.
234
+ const entry = findNavEntry(pageForAnchor(anchor.replace(/^#/, '')));
235
+ if (entry) tuningDoc = { anchor, title: entry.title, content: readNavEntryContent(entry) };
231
236
  }
232
237
 
233
238
  return { type, name: titleCase(label), detectionDoc, tuningDoc };
@@ -235,21 +240,29 @@ export function getFindingDocumentation(type ) {
235
240
 
236
241
  const CHAPTERS_NAV_FILE = join(DOCS_CONTENT_DIR, 'chapters', 'nav-index.json');
237
242
 
243
+
244
+
245
+ function findNavEntry(page ) {
246
+ const nav = JSON.parse(readFileSync(CHAPTERS_NAV_FILE, 'utf8')) ;
247
+ return nav.find((e) => e.anchor === page);
248
+ }
249
+
250
+ function readNavEntryContent(entry ) {
251
+ const dir = entry.store === 'tuning' ? 'tuning' : 'chapters';
252
+ return readFileSync(join(DOCS_CONTENT_DIR, dir, `${entry.slug}.md`), 'utf8');
253
+ }
254
+
238
255
 
239
256
 
240
257
  /** Full tuning-reference markdown for one doc anchor, run-independent. Resolves the anchor to its
241
258
  * owning page via pageForAnchor (so '#metric-task-duration' returns the 'metrics' page), looks it up
242
- * in the committed nav-index, and reads the markdown from the same docs-content store the website
259
+ * in the generated nav-index, and reads the markdown from the same docs-content store the website
243
260
  * renders from. The general-chapter counterpart to getFindingDocumentation (keyed by finding type). */
244
261
  export function getReferenceDoc(anchor ) {
245
262
  const page = pageForAnchor(String(anchor).replace(/^#/, ''));
246
- const nav = JSON.parse(readFileSync(CHAPTERS_NAV_FILE, 'utf8'))
247
- ;
248
- const entry = nav.find((e) => e.anchor === page);
263
+ const entry = findNavEntry(page);
249
264
  if (!entry) throw mcpError('invalid-anchor', `Unknown reference anchor: ${anchor}`);
250
- const dir = entry.store === 'tuning' ? 'tuning' : 'chapters';
251
- const content = readFileSync(join(DOCS_CONTENT_DIR, dir, `${entry.slug}.md`), 'utf8');
252
- return { anchor: page, title: entry.title, content };
265
+ return { anchor: page, title: entry.title, content: readNavEntryContent(entry) };
253
266
  }
254
267
 
255
268
  export function getFindingEvidence(
@@ -117,6 +117,9 @@ export function clipToCeiling(wasteMsClaimed , stage , cei
117
117
 
118
118
 
119
119
 
120
+
121
+
122
+
120
123
 
121
124
 
122
125
  export function computeOccupancy(
@@ -128,7 +131,11 @@ export function computeOccupancy(
128
131
  for (const s of stages.values()) {
129
132
  const g = gate.get(s.id);
130
133
  if (g === undefined) continue; // excluded from the sweep (duration <= 0)
131
- info.set(s.id, { gate: g, ceiling: computeCeiling(s, totalCores) });
134
+ info.set(s.id, {
135
+ gate: g,
136
+ ceiling: computeCeiling(s, totalCores),
137
+ coreWorkFloor: totalCores > 0 ? (s.executorRunTime ?? 0) / totalCores : 0,
138
+ });
132
139
  }
133
140
  return info;
134
141
  }
@@ -140,6 +147,60 @@ const SERIAL_GATE_THRESHOLD = 0.999;
140
147
 
141
148
 
142
149
 
150
+
151
+
152
+
153
+
154
+
155
+
156
+
157
+
158
+
159
+
160
+ // Wall-clock a skew/straggler fix recovers: finalizeStage's task-level replay
161
+ // (tailReplayRecoveryMs, computeTailReplayRecoveryMs) when the stage carries it. Without it (a
162
+ // stage built by hand, as in detector tests) an estimate from the slowest task's excess over the
163
+ // median: a lone straggler costs that excess; a tail of many slow tasks costs its summed excess
164
+ // (stragglerExcessMs) spread over the slots the stage had (peakConcurrentTasks), and the replay
165
+ // recovers about the larger of the two.
166
+ export function tailRecoveryMs(stage , singleTaskExcessMs ) {
167
+ if (stage.tailReplayRecoveryMs != null) return stage.tailReplayRecoveryMs;
168
+ const excessMs = stage.stragglerExcessMs ?? 0;
169
+ const slots = stage.peakConcurrentTasks ?? 0;
170
+ if (excessMs <= 0 || slots <= 0) return singleTaskExcessMs;
171
+ return Math.max(singleTaskExcessMs, excessMs / slots);
172
+ }
173
+
174
+ // Task time a skew/straggler fix removes, by the same measure: the tasks over 4x P50 capped at the
175
+ // median (stragglerExcessMs), or the slowest task's excess when that alone is larger.
176
+ export function tailRemovedWorkMs(stage , singleTaskExcessMs ) {
177
+ return Math.max(singleTaskExcessMs, stage.stragglerExcessMs ?? 0);
178
+ }
179
+
180
+ // Longest task a straggler fix leaves: every task over 4x P50 comes down to the median, so the
181
+ // longest one at or under that (finalizeStage's longestNonStragglerMs). Without stragglers (a
182
+ // speculation-driven finding) or the field (older snapshots), the median.
183
+ export function stragglerFixLongestTaskMs(stage ) {
184
+ const p50 = stage.taskDurationP50 ?? 0;
185
+ if (!((stage.stragglerCount ?? 0) > 0) || stage.longestNonStragglerMs == null) return p50;
186
+ return Math.max(p50, stage.longestNonStragglerMs);
187
+ }
188
+
189
+
190
+
191
+
192
+
193
+
194
+
195
+
196
+
197
+
198
+
199
+
200
+
201
+
202
+
203
+
143
204
  /**
144
205
  * Per-finding estimate for a single-stage waste claim. Returns null when the
145
206
  * stage was excluded from the sweep (duration <= 0): callers must fall back
@@ -151,11 +212,19 @@ export function estimateSingleStage(
151
212
  stageId ,
152
213
  stages ,
153
214
  info ,
215
+ { shortensLongestTask = false, removedCoreWorkMs = 0, longestTaskAfterFixMs = 0 } = {},
154
216
  ) {
155
217
  const stage = stages.get(stageId);
156
218
  const stageInfo = info.get(stageId);
157
219
  if (!stage || !stageInfo) return null;
158
- const clipped = clipToCeiling(wasteMsClaimed, stage, stageInfo.ceiling);
220
+ const coreWork = stage.executorRunTime ?? 0;
221
+ const coreWorkFloor = removedCoreWorkMs > 0 && coreWork > 0
222
+ ? stageInfo.coreWorkFloor * Math.max(0, 1 - removedCoreWorkMs / coreWork)
223
+ : stageInfo.coreWorkFloor;
224
+ const ceiling = shortensLongestTask
225
+ ? Math.max((stage.taskDurationMax ?? 0) - wasteMsClaimed, longestTaskAfterFixMs, coreWorkFloor)
226
+ : stageInfo.ceiling;
227
+ const clipped = clipToCeiling(wasteMsClaimed, stage, ceiling);
159
228
  if (stageInfo.gate >= SERIAL_GATE_THRESHOLD) {
160
229
  return { basis: 'serial', wallClock: { low: clipped, high: clipped } };
161
230
  }