sparkforensics-mcp 0.2.0 → 0.2.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (44) hide show
  1. package/package.json +5 -4
  2. package/vendor-core/cli/collect-run.js +3 -2
  3. package/vendor-core/cli/native-zstd.js +351 -0
  4. package/vendor-core/detectors.js +390 -75
  5. package/vendor-core/docs-config.js +34 -8
  6. package/vendor-core/docs-content/chapters/03-memory-model.md +39 -0
  7. package/vendor-core/docs-content/chapters/11-cluster-config.md +40 -0
  8. package/vendor-core/docs-content/detection/cache.md +4 -3
  9. package/vendor-core/docs-content/detection/chrn.md +4 -2
  10. package/vendor-core/docs-content/detection/gc.md +2 -0
  11. package/vendor-core/docs-content/detection/host.md +2 -1
  12. package/vendor-core/docs-content/detection/local.md +2 -3
  13. package/vendor-core/docs-content/detection/mem.md +3 -3
  14. package/vendor-core/docs-content/detection/plan.md +3 -1
  15. package/vendor-core/docs-content/detection/shape.md +2 -1
  16. package/vendor-core/docs-content/detection/shfl.md +2 -1
  17. package/vendor-core/docs-content/detection/spec.md +4 -3
  18. package/vendor-core/docs-content/detection/spill.md +1 -1
  19. package/vendor-core/docs-content/detection/strag.md +2 -1
  20. package/vendor-core/docs-content/detection/tiny.md +2 -1
  21. package/vendor-core/docs-content/tuning/failures.md +1 -1
  22. package/vendor-core/docs-content/tuning/gc.md +11 -4
  23. package/vendor-core/docs-content/tuning/shuffle.md +25 -5
  24. package/vendor-core/docs-content/tuning/skew.md +14 -6
  25. package/vendor-core/docs-content/tuning/small-files.md +12 -7
  26. package/vendor-core/docs-content/tuning/straggler.md +34 -0
  27. package/vendor-core/docs-content/tuning/tiny-tasks.md +1 -1
  28. package/vendor-core/docs-content/tuning/utilization.md +57 -7
  29. package/vendor-core/docs-content/upstream.json +4 -0
  30. package/vendor-core/event-handlers.js +321 -69
  31. package/vendor-core/event-schemas.js +8 -6
  32. package/vendor-core/evidence-report.js +4 -2
  33. package/vendor-core/impact-estimator.js +150 -34
  34. package/vendor-core/mcp-tools.js +20 -7
  35. package/vendor-core/occupancy.js +70 -2
  36. package/vendor-core/parser-worker.js +30 -14
  37. package/vendor-core/plan-summary.js +4 -0
  38. package/vendor-core/run-comparison.js +22 -17
  39. package/vendor-core/shs-fetch.js +18 -7
  40. package/vendor-core/shs-load.js +2 -1
  41. package/vendor-core/stage-quantiles.js +59 -3
  42. package/vendor-core/string-hash.js +15 -0
  43. package/vendor-core/types.js +9 -1
  44. package/vendor-core/vendor/fzstd.js +94 -18
@@ -1,14 +1,75 @@
1
1
 
2
- import { computeOccupancy, estimateSingleStage, estimateMultiStage, } from './occupancy.js';
3
- import { detectorCatalog } from './detectors.js';
2
+ import {
3
+ computeOccupancy, estimateSingleStage, estimateMultiStage, tailRecoveryMs, tailRemovedWorkMs, stragglerFixLongestTaskMs,
4
+
5
+ } from './occupancy.js';
4
6
 
5
- // Assumed shuffle-network throughput, ~1 Gbps. Starting assumption, unvalidated.
7
+ // Assumed shuffle-network throughput per executor link, ~1 Gbps. Starting assumption, unvalidated.
6
8
  const SHUFFLE_THROUGHPUT_BPS = 125_000_000;
7
- // Assumed disk I/O throughput for spilled data, ~200 MB/s (conservative HDD/SSD blend).
9
+ // Assumed disk I/O throughput per executor for spilled data, ~200 MB/s (conservative HDD/SSD blend).
8
10
  const SPILL_IO_THROUGHPUT_BPS = 200_000_000;
11
+
12
+ // A stage's own per-task overhead: task wall time (launch to finish, summed per executor by
13
+ // finalizeStage) minus executorRunTime, i.e. deserialization, result serialization and the
14
+ // launch round trip that coalescing tasks removes. `concurrency` is the stage's achieved task
15
+ // concurrency (task time / stage duration). Null when the stage has no such data or no overhead.
16
+ function measuredTaskOverhead(stage ) {
17
+ const executorStats = Array.isArray(stage.executorStats)
18
+ ? (stage.executorStats ) : [];
19
+ const taskTimeMs = executorStats.reduce((sum, e) => sum + (e.totalDuration ?? 0), 0);
20
+ const taskCount = stage.taskCount ?? 0;
21
+ const durationMs = (stage.completedAt ?? 0) - (stage.submittedAt ?? 0);
22
+ const overheadMs = taskTimeMs - (stage.executorRunTime ?? 0);
23
+ if (taskTimeMs <= 0 || taskCount <= 0 || durationMs <= 0 || overheadMs <= 0) return null;
24
+ return { perTaskMs: overheadMs / taskCount, concurrency: taskTimeMs / durationMs };
25
+ }
26
+
27
+ // Both constants above are single-device figures (one NIC, one local disk). A stage's shuffle
28
+ // reads and spills are spread over every executor that ran its tasks, each moving its own share
29
+ // in parallel, so the stage's aggregate bandwidth scales with that executor count. Dividing a
30
+ // stage-wide byte total by one device's bandwidth modeled the whole cluster as one link: on 14
31
+ // real logs that claimed up to 1,870s of shuffle time on stages whose tasks measured ~0s of
32
+ // shuffle fetch wait (222 of 284 shuffle findings). No executor data falls back to one device.
33
+ function stageIoParallelism(stage ) {
34
+ const executors = Array.isArray(stage.executorStats) ? stage.executorStats.length : 0;
35
+ return Math.max(1, executors);
36
+ }
37
+
38
+ // Wall-clock the stage's tasks spent blocked fetching shuffle blocks: fetchWaitTime is a
39
+ // cross-task sum like executorRunTime, so dividing by the stage's average concurrency converts it
40
+ // (the gc estimate's conversion). Null when the stage has no run time or duration to convert with.
41
+ // On 14 real logs the link model claimed 2780s over 284 shuffle findings, 256s once capped at
42
+ // this, with about zero fetch wait on 5 of the 10 non-info ones: reads that overlapped compute
43
+ // stalled nothing.
44
+ function fetchWaitWallClockMs(stage ) {
45
+ const fetchWaitMs = stage.fetchWaitTime;
46
+ const runTimeMs = stage.executorRunTime ?? 0;
47
+ const durationMs = (stage.completedAt ?? 0) - (stage.submittedAt ?? 0);
48
+ if (typeof fetchWaitMs !== 'number' || runTimeMs <= 0 || durationMs <= 0) return null;
49
+ return fetchWaitMs / (runTimeMs / durationMs);
50
+ }
51
+
52
+ // Wall-clock a stage's wasted (retried) attempts cost it. Each one delayed only its own task, and
53
+ // attempts of different tasks ran side by side: one lost executor fails every task it was running
54
+ // at once (4 wasted attempts of 36.6s each, all first attempts, on a real stage that ran 41 tasks
55
+ // at once, claimed as 146.6s). So the stage lost at most its longest retry chain, the most attempts
56
+ // one task wasted (its highest attempt number + 1) times the mean wasted attempt, or their summed
57
+ // time spread over its slots, whichever is larger. Without a sample of every wasted attempt
58
+ // (retryTaskSamples is capped), the chain isn't known: the summed time.
59
+ function retryWallClockMs(stage ) {
60
+ const totalMs = (stage.retryWasteMs ) ?? 0;
61
+ const attempts = (stage.wastedAttempts ) ?? 0;
62
+ const samples = Array.isArray(stage.retryTaskSamples)
63
+ ? (stage.retryTaskSamples ) : [];
64
+ if (totalMs <= 0 || attempts <= 0 || samples.length < attempts) return totalMs;
65
+ const chain = samples.reduce((longest, s) => Math.max(longest, (s.attemptNumber ?? 0) + 1), 1);
66
+ const slots = Math.max(1, stage.peakConcurrentTasks ?? 1);
67
+ return Math.min(totalMs, Math.max((chain * totalMs) / attempts, totalMs / slots));
68
+ }
9
69
  // Spark's classic recommended shuffle partition size.
10
70
  const IDEAL_BYTES_PER_PARTITION_TASK = 128 * 1024 * 1024;
11
- // Assumed per-task scheduling/launch overhead.
71
+ // Assumed per-task scheduling/launch overhead: the fallback when a stage lacks the per-executor
72
+ // task-time sums measuredTaskOverhead needs.
12
73
  const TASK_SCHEDULING_OVERHEAD_MS = 50;
13
74
  // Assumed per-file open latency (small-file overhead).
14
75
  const FILE_OPEN_OVERHEAD_MS = 10;
@@ -21,15 +82,10 @@ const EXECUTOR_STARTUP_OVERHEAD_MS = 15000;
21
82
  // Assumed re-read throughput, shared by cachingOpportunity and cacheUtilization.
22
83
  const RE_READ_THROUGHPUT_BPS = 125_000_000;
23
84
 
24
- // stageSlowness flags a stage at `infoMin` minutes; that's the floor for the waste this estimate
25
- // reports. Read from the detector's catalog entry so the two stay in sync automatically.
26
- const STAGE_SLOWNESS_THRESHOLD_MINUTES = (() => {
27
- const infoMin = detectorCatalog().find((d) => d.type === 'stageSlowness')?.thresholds?.infoMin;
28
- if (typeof infoMin !== 'number') {
29
- throw new Error("impact-estimator: stageSlowness detector's 'infoMin' threshold not found in detectorCatalog()");
30
- }
31
- return infoMin;
32
- })();
85
+ // skew and straggler claim time off the stage's longest task itself, so the occupancy clip must
86
+ // not floor them at that same task (see estimateSingleStage). detectors.ts's clippedWasteMs gates
87
+ // both detectors on the same option so the firing floor and the displayed estimate agree.
88
+ const TAIL_CLAIM = { shortensLongestTask: true };
33
89
 
34
90
  // No quantifiable magnitude -> 'informational'; a rawWaste figure with no stage window ->
35
91
  // 'resourceOnly'. Never a fake {low:0, high:0}: wallClock is null in both cases.
@@ -46,8 +102,9 @@ function singleStageImpact(
46
102
  occupancy ,
47
103
  estimateMethod ,
48
104
  rawWaste ,
105
+ opts ,
49
106
  ) {
50
- const est = estimateSingleStage(wasteMs, stageId, stages , occupancy);
107
+ const est = estimateSingleStage(wasteMs, stageId, stages , occupancy, opts);
51
108
  if (est) return { basis: est.basis, wallClock: est.wallClock, estimateMethod, rawWaste };
52
109
  return costOnly(estimateMethod, rawWaste); // stage excluded from the sweep (duration <= 0)
53
110
  }
@@ -77,6 +134,7 @@ function computeEstimateForFinding(
77
134
  finding ,
78
135
  stages ,
79
136
  occupancy ,
137
+ totalCores ,
80
138
  ) {
81
139
  switch (finding.type) {
82
140
  case 'retryWaste': {
@@ -85,7 +143,9 @@ function computeEstimateForFinding(
85
143
  const stage = stages.get(finding.stageId);
86
144
  if (!stage) return null;
87
145
  const wasteMs = (stage.retryWasteMs ) ?? 0;
88
- return singleStageImpact(wasteMs, finding.stageId, stages, occupancy, 'measured', { value: wasteMs, unit: 'ms' });
146
+ const wallClockMs = retryWallClockMs(stage);
147
+ return singleStageImpact(wallClockMs, finding.stageId, stages, occupancy,
148
+ wallClockMs === wasteMs ? 'measured' : 'modeled', { value: wasteMs, unit: 'ms' });
89
149
  }
90
150
  case 'speculationWaste': {
91
151
  if (finding.stageId == null) return null;
@@ -103,6 +163,9 @@ function computeEstimateForFinding(
103
163
  return { basis: 'serial', wallClock: { low: wasteMs, high: wasteMs }, estimateMethod: 'measured' };
104
164
  }
105
165
  case 'gc': {
166
+ // The low-GC branch is an over-provisioning signal whose fix (less executor memory) raises
167
+ // GC rather than recovering it: the stage's GC time is no saving there, so no waste model.
168
+ if (finding.direction === 'low') return costOnly('none');
106
169
  if (finding.stageId == null) return null;
107
170
  const stage = stages.get(finding.stageId);
108
171
  if (!stage) return null;
@@ -128,8 +191,11 @@ function computeEstimateForFinding(
128
191
  const p50 = stage.taskDurationP50 ?? 0;
129
192
  // computeSkewRatio's own metric labels (src/detectors.ts): 'P95/median' or 'max/median'.
130
193
  const usesP95Branch = finding.metric === 'P95/median';
131
- const wasteMs = Math.max(0, usesP95Branch ? (stage.taskDurationP95 ?? 0) - p50 : (stage.taskDurationMax ?? 0) - p50);
132
- return singleStageImpact(wasteMs, finding.stageId, stages, occupancy, 'measured', { value: wasteMs, unit: 'ms' });
194
+ const singleDelta = Math.max(0, usesP95Branch ? (stage.taskDurationP95 ?? 0) - p50 : (stage.taskDurationMax ?? 0) - p50);
195
+ const wasteMs = tailRecoveryMs(stage, singleDelta);
196
+ // Fixing the skew still waits on the longest task it leaves, as for straggler.
197
+ return singleStageImpact(wasteMs, finding.stageId, stages, occupancy, 'measured', { value: wasteMs, unit: 'ms' },
198
+ { ...TAIL_CLAIM, removedCoreWorkMs: tailRemovedWorkMs(stage, singleDelta), longestTaskAfterFixMs: stragglerFixLongestTaskMs(stage) });
133
199
  }
134
200
  case 'straggler':
135
201
  case 'stageShape': {
@@ -171,8 +237,11 @@ function computeEstimateForFinding(
171
237
  if (finding.stageId == null) return null;
172
238
  const stage = stages.get(finding.stageId);
173
239
  if (!stage) return null;
174
- const wasteMs = Math.max(0, (stage.taskDurationMax ?? 0) - (stage.taskDurationP50 ?? 0));
175
- return singleStageImpact(wasteMs, finding.stageId, stages, occupancy, 'measured', { value: wasteMs, unit: 'ms' });
240
+ const longestTaskAfterFixMs = stragglerFixLongestTaskMs(stage);
241
+ const singleDelta = Math.max(0, (stage.taskDurationMax ?? 0) - longestTaskAfterFixMs);
242
+ const wasteMs = tailRecoveryMs(stage, singleDelta);
243
+ return singleStageImpact(wasteMs, finding.stageId, stages, occupancy, 'measured', { value: wasteMs, unit: 'ms' },
244
+ { ...TAIL_CLAIM, removedCoreWorkMs: tailRemovedWorkMs(stage, singleDelta), longestTaskAfterFixMs });
176
245
  }
177
246
  case 'slowHost': {
178
247
  // Three duration-based shapes, each carrying its absolute-ms figure under a different field
@@ -202,18 +271,31 @@ function computeEstimateForFinding(
202
271
  const occurrences = typeof finding.value === 'number' ? finding.value : 0;
203
272
  // Defensive only: minOccurrences guarantees occurrences >= 2 on real data; a malformed-value fallback.
204
273
  if (occurrences < 2) return costOnly('none');
274
+ // Same-shaped repeats whose details differ compute different data: nothing is known to be
275
+ // recomputed, so there is no time to claim.
276
+ if (finding.occurrencesIdentical === false) return costOnly('none');
277
+ // Each stage contributes the share of its operators inside the repeated subtree: a stage it
278
+ // shares with other operators (the consuming join, the join's other side) isn't all its
279
+ // time, and claiming whole stages let sibling groups claim the same stage twice. Findings
280
+ // built without the field (hand-made fixtures) count every linked stage whole.
281
+ const shares = (finding.stageShares ?? null) ;
205
282
  const redundantFraction = (occurrences - 1) / occurrences;
206
283
  const wasteMsByStage = new Map ();
207
284
  for (const id of stageIds) {
208
285
  const s = stages.get(id);
209
- if (s) {
210
- const durationMs = Math.max(0, (s.completedAt ?? 0) - (s.submittedAt ?? 0));
211
- wasteMsByStage.set(id, durationMs * redundantFraction);
286
+ const share = shares ? (shares[id] ?? 0) : 1;
287
+ if (s && share > 0) {
288
+ // Time with tasks running, not submit-to-complete: a stage left waiting for cores
289
+ // (2491s open, 60s of tasks on a real log) isn't recomputing anything while it waits.
290
+ const activeMs = s.taskActiveMs ?? Math.max(0, (s.completedAt ?? 0) - (s.submittedAt ?? 0));
291
+ wasteMsByStage.set(id, activeMs * share * redundantFraction);
212
292
  }
213
293
  }
294
+ // No operator of the subtree ran in a known stage: no time to attribute.
295
+ if (wasteMsByStage.size === 0) return costOnly('none');
214
296
  const totalWasteMs = [...wasteMsByStage.values()].reduce((sum, ms) => sum + ms, 0);
215
297
  const rawWaste = { value: totalWasteMs, unit: 'ms' };
216
- const est = estimateMultiStage(stageIds, wasteMsByStage, stages , occupancy);
298
+ const est = estimateMultiStage([...wasteMsByStage.keys()], wasteMsByStage, stages , occupancy);
217
299
  if (!est) return costOnly('measured', rawWaste);
218
300
  return { basis: est.basis, wallClock: est.wallClock, estimateMethod: 'measured', rawWaste };
219
301
  }
@@ -222,16 +304,21 @@ function computeEstimateForFinding(
222
304
  const stage = stages.get(finding.stageId);
223
305
  if (!stage) return null;
224
306
  const shuffleReadBytes = stage.shuffleReadBytes ?? 0;
225
- const wasteMs = (shuffleReadBytes / SHUFFLE_THROUGHPUT_BPS) * 1000;
226
- // The measured byte volume driving the modeled ms figure above.
227
- return singleStageImpact(wasteMs, finding.stageId, stages, occupancy, 'modeled', { value: shuffleReadBytes, unit: 'bytes' });
307
+ const modeledMs = (shuffleReadBytes / (SHUFFLE_THROUGHPUT_BPS * stageIoParallelism(stage))) * 1000;
308
+ // The link model can't see whether the reads stalled the tasks: capped at the fetch wait the
309
+ // tasks measured, the claim never exceeds what the stage spent blocked on the network.
310
+ const measuredMs = fetchWaitWallClockMs(stage);
311
+ const wasteMs = measuredMs == null ? modeledMs : Math.min(modeledMs, measuredMs);
312
+ // rawWaste: the measured byte volume behind the modeled figure.
313
+ return singleStageImpact(wasteMs, finding.stageId, stages, occupancy,
314
+ measuredMs != null && measuredMs < modeledMs ? 'measured' : 'modeled', { value: shuffleReadBytes, unit: 'bytes' });
228
315
  }
229
316
  case 'spill': {
230
317
  if (finding.stageId == null) return null;
231
318
  const stage = stages.get(finding.stageId);
232
319
  if (!stage) return null;
233
320
  const diskBytesSpilled = stage.diskBytesSpilled ?? 0;
234
- const wasteMs = (diskBytesSpilled / SPILL_IO_THROUGHPUT_BPS) * 1000;
321
+ const wasteMs = (diskBytesSpilled / (SPILL_IO_THROUGHPUT_BPS * stageIoParallelism(stage))) * 1000;
235
322
  // Surfaces the number the formula uses: the displayed metric is memoryBytesSpilled, but disk spill costs the I/O time.
236
323
  return singleStageImpact(wasteMs, finding.stageId, stages, occupancy, 'modeled', { value: diskBytesSpilled, unit: 'bytes' });
237
324
  }
@@ -239,9 +326,21 @@ function computeEstimateForFinding(
239
326
  if (finding.stageId == null) return null;
240
327
  const stage = stages.get(finding.stageId);
241
328
  if (!stage) return null;
242
- const stageDurationMs = (stage.completedAt ?? 0) - (stage.submittedAt ?? 0);
243
- const wasteMs = Math.max(0, stageDurationMs - STAGE_SLOWNESS_THRESHOLD_MINUTES * 60000);
244
- return singleStageImpact(wasteMs, finding.stageId, stages, occupancy, 'modeled', { value: wasteMs, unit: 'ms' });
329
+ // The recommended fix is more partitions, which only helps a stage that ran fewer tasks than
330
+ // the cluster has cores: the time its tasks were running could then spread over up to
331
+ // totalCores (lowShuffleParallelism's shape). Time the stage sat open with no task running
332
+ // is queueing no partition count recovers. Splitting partitions splits the longest task
333
+ // too, hence TAIL_CLAIM's post-fix floor. Unknown cluster size: no defensible figure.
334
+ if (totalCores <= 0) return costOnly('modeled');
335
+ // A stage that read no input and no shuffle has no data for more partitions to split (a
336
+ // 1-task count stage open 27 minutes on 5s of CPU was claimed 99% recoverable): claim 0.
337
+ const readBytes = (stage.inputBytes ?? 0) + (stage.shuffleReadBytes ?? 0);
338
+ const activeMs = typeof stage.taskActiveMs === 'number'
339
+ ? stage.taskActiveMs
340
+ : Math.max(0, (stage.completedAt ?? 0) - (stage.submittedAt ?? 0));
341
+ const taskCount = stage.taskCount ?? 0;
342
+ const wasteMs = readBytes > 0 ? activeMs * Math.max(0, 1 - taskCount / totalCores) : 0;
343
+ return singleStageImpact(wasteMs, finding.stageId, stages, occupancy, 'modeled', { value: wasteMs, unit: 'ms' }, TAIL_CLAIM);
245
344
  }
246
345
  case 'partitionSizing': {
247
346
  if (finding.stageId == null) return null;
@@ -275,12 +374,29 @@ function computeEstimateForFinding(
275
374
  if (!stage) return null;
276
375
  const taskCount = stage.taskCount ?? 0;
277
376
  const excessTaskCount = Math.max(0, taskCount - Math.round(taskCount / 10));
377
+ const measured = measuredTaskOverhead(stage);
378
+ if (measured) {
379
+ // Coalescing to a tenth of the tasks removes the excess tasks' per-task overhead: core
380
+ // time spent in parallel, so wall-clock at the stage's achieved concurrency (floored at
381
+ // 1: a mostly-idle stage can't save more wall-clock than the task time it removes).
382
+ const wasteMs = (excessTaskCount * measured.perTaskMs) / Math.max(1, measured.concurrency);
383
+ return singleStageImpact(wasteMs, finding.stageId, stages, occupancy, 'measured', { value: wasteMs, unit: 'ms' });
384
+ }
278
385
  const wasteMs = excessTaskCount * TASK_SCHEDULING_OVERHEAD_MS;
279
386
  return singleStageImpact(wasteMs, finding.stageId, stages, occupancy, 'modeled', { value: wasteMs, unit: 'ms' });
280
387
  }
281
388
  case 'smallFiles': {
282
- const wasteMs = ((finding.fileCount ) ?? 0) * FILE_OPEN_OVERHEAD_MS;
283
- return stageMappableWasteOrCostOnly(wasteMs, finding.stageIds , stages, occupancy);
389
+ const fileMs = ((finding.fileCount ) ?? 0) * FILE_OPEN_OVERHEAD_MS;
390
+ // A read's files are opened by the scan's tasks, in parallel: spread the per-file cost over
391
+ // the most tasks its stages ran at once (91344 files x 10ms is 913s, claimed against a 117s
392
+ // stage that ran 314 tasks at once). A write keeps the serial sum: the job commit moves each
393
+ // output file on the driver, one after another.
394
+ const stageIds = finding.stageIds ;
395
+ let slots = 1;
396
+ if (finding.direction === 'read') {
397
+ for (const id of stageIds ?? []) slots = Math.max(slots, stages.get(id)?.peakConcurrentTasks ?? 1);
398
+ }
399
+ return stageMappableWasteOrCostOnly(fileMs / slots, stageIds, stages, occupancy);
284
400
  }
285
401
  case 'overBroadcast': {
286
402
  // metric: 'broadcastBytes', value: <bytes>.
@@ -385,7 +501,7 @@ function computeEstimateForFinding(
385
501
  export function estimateImpact(findings , stages , totalCores = 0) {
386
502
  const occupancy = computeOccupancy(stages , totalCores);
387
503
  for (const f of findings) {
388
- const estimate = computeEstimateForFinding(f, stages, occupancy);
504
+ const estimate = computeEstimateForFinding(f, stages, occupancy, totalCores);
389
505
  if (estimate) f.impactEstimate = estimate;
390
506
  }
391
507
  return findings;
@@ -228,6 +228,11 @@ export function getFindingDocumentation(type ) {
228
228
  if (tuningPath && anchor && existsSync(tuningPath)) {
229
229
  const tuningContent = readFileSync(tuningPath, 'utf8');
230
230
  tuningDoc = { anchor, title: extractDocTitle(tuningContent), content: tuningContent };
231
+ } else if (anchor && !slug) {
232
+ // A section hosted on a chapter page (e.g. autoscaling churn on cluster-config): return the
233
+ // owning chapter, the same page the web view's docs link opens.
234
+ const entry = findNavEntry(pageForAnchor(anchor.replace(/^#/, '')));
235
+ if (entry) tuningDoc = { anchor, title: entry.title, content: readNavEntryContent(entry) };
231
236
  }
232
237
 
233
238
  return { type, name: titleCase(label), detectionDoc, tuningDoc };
@@ -235,21 +240,29 @@ export function getFindingDocumentation(type ) {
235
240
 
236
241
  const CHAPTERS_NAV_FILE = join(DOCS_CONTENT_DIR, 'chapters', 'nav-index.json');
237
242
 
243
+
244
+
245
+ function findNavEntry(page ) {
246
+ const nav = JSON.parse(readFileSync(CHAPTERS_NAV_FILE, 'utf8')) ;
247
+ return nav.find((e) => e.anchor === page);
248
+ }
249
+
250
+ function readNavEntryContent(entry ) {
251
+ const dir = entry.store === 'tuning' ? 'tuning' : 'chapters';
252
+ return readFileSync(join(DOCS_CONTENT_DIR, dir, `${entry.slug}.md`), 'utf8');
253
+ }
254
+
238
255
 
239
256
 
240
257
  /** Full tuning-reference markdown for one doc anchor, run-independent. Resolves the anchor to its
241
258
  * owning page via pageForAnchor (so '#metric-task-duration' returns the 'metrics' page), looks it up
242
- * in the committed nav-index, and reads the markdown from the same docs-content store the website
259
+ * in the generated nav-index, and reads the markdown from the same docs-content store the website
243
260
  * renders from. The general-chapter counterpart to getFindingDocumentation (keyed by finding type). */
244
261
  export function getReferenceDoc(anchor ) {
245
262
  const page = pageForAnchor(String(anchor).replace(/^#/, ''));
246
- const nav = JSON.parse(readFileSync(CHAPTERS_NAV_FILE, 'utf8'))
247
- ;
248
- const entry = nav.find((e) => e.anchor === page);
263
+ const entry = findNavEntry(page);
249
264
  if (!entry) throw mcpError('invalid-anchor', `Unknown reference anchor: ${anchor}`);
250
- const dir = entry.store === 'tuning' ? 'tuning' : 'chapters';
251
- const content = readFileSync(join(DOCS_CONTENT_DIR, dir, `${entry.slug}.md`), 'utf8');
252
- return { anchor: page, title: entry.title, content };
265
+ return { anchor: page, title: entry.title, content: readNavEntryContent(entry) };
253
266
  }
254
267
 
255
268
  export function getFindingEvidence(
@@ -117,6 +117,9 @@ export function clipToCeiling(wasteMsClaimed , stage , cei
117
117
 
118
118
 
119
119
 
120
+
121
+
122
+
120
123
 
121
124
 
122
125
  export function computeOccupancy(
@@ -128,7 +131,11 @@ export function computeOccupancy(
128
131
  for (const s of stages.values()) {
129
132
  const g = gate.get(s.id);
130
133
  if (g === undefined) continue; // excluded from the sweep (duration <= 0)
131
- info.set(s.id, { gate: g, ceiling: computeCeiling(s, totalCores) });
134
+ info.set(s.id, {
135
+ gate: g,
136
+ ceiling: computeCeiling(s, totalCores),
137
+ coreWorkFloor: totalCores > 0 ? (s.executorRunTime ?? 0) / totalCores : 0,
138
+ });
132
139
  }
133
140
  return info;
134
141
  }
@@ -140,6 +147,59 @@ const SERIAL_GATE_THRESHOLD = 0.999;
140
147
 
141
148
 
142
149
 
150
+
151
+
152
+
153
+
154
+
155
+
156
+
157
+
158
+
159
+ // Wall-clock a skew/straggler fix recovers, given the slowest task's excess over the median. A
160
+ // lone straggler costs that excess; a tail of many slow tasks (a bimodal stage: 26% of 1400
161
+ // tasks over 4x P50 on a real log) costs its summed excess (stragglerExcessMs) spread over the
162
+ // slots the stage had (peakConcurrentTasks), far more than one task's. The task-level replay in
163
+ // dev/eval-tail-replay.mjs recovers about the larger of the two. Average concurrency would be the
164
+ // wrong divisor: a tail-dominated stage runs few tasks for most of its span (5.7 average vs 14
165
+ // peak on one real stage), which doubled the claim.
166
+ export function tailRecoveryMs(stage , singleTaskExcessMs ) {
167
+ const excessMs = stage.stragglerExcessMs ?? 0;
168
+ const slots = stage.peakConcurrentTasks ?? 0;
169
+ if (excessMs <= 0 || slots <= 0) return singleTaskExcessMs;
170
+ return Math.max(singleTaskExcessMs, excessMs / slots);
171
+ }
172
+
173
+ // Task time a skew/straggler fix removes, by the same measure: the tasks over 4x P50 capped at the
174
+ // median (stragglerExcessMs), or the slowest task's excess when that alone is larger.
175
+ export function tailRemovedWorkMs(stage , singleTaskExcessMs ) {
176
+ return Math.max(singleTaskExcessMs, stage.stragglerExcessMs ?? 0);
177
+ }
178
+
179
+ // Longest task a straggler fix leaves: every task over 4x P50 comes down to the median, so the
180
+ // longest one at or under that (finalizeStage's longestNonStragglerMs). Without stragglers (a
181
+ // speculation-driven finding) or the field (older snapshots), the median.
182
+ export function stragglerFixLongestTaskMs(stage ) {
183
+ const p50 = stage.taskDurationP50 ?? 0;
184
+ if (!((stage.stragglerCount ?? 0) > 0) || stage.longestNonStragglerMs == null) return p50;
185
+ return Math.max(p50, stage.longestNonStragglerMs);
186
+ }
187
+
188
+
189
+
190
+
191
+
192
+
193
+
194
+
195
+
196
+
197
+
198
+
199
+
200
+
201
+
202
+
143
203
  /**
144
204
  * Per-finding estimate for a single-stage waste claim. Returns null when the
145
205
  * stage was excluded from the sweep (duration <= 0): callers must fall back
@@ -151,11 +211,19 @@ export function estimateSingleStage(
151
211
  stageId ,
152
212
  stages ,
153
213
  info ,
214
+ { shortensLongestTask = false, removedCoreWorkMs = 0, longestTaskAfterFixMs = 0 } = {},
154
215
  ) {
155
216
  const stage = stages.get(stageId);
156
217
  const stageInfo = info.get(stageId);
157
218
  if (!stage || !stageInfo) return null;
158
- const clipped = clipToCeiling(wasteMsClaimed, stage, stageInfo.ceiling);
219
+ const coreWork = stage.executorRunTime ?? 0;
220
+ const coreWorkFloor = removedCoreWorkMs > 0 && coreWork > 0
221
+ ? stageInfo.coreWorkFloor * Math.max(0, 1 - removedCoreWorkMs / coreWork)
222
+ : stageInfo.coreWorkFloor;
223
+ const ceiling = shortensLongestTask
224
+ ? Math.max((stage.taskDurationMax ?? 0) - wasteMsClaimed, longestTaskAfterFixMs, coreWorkFloor)
225
+ : stageInfo.ceiling;
226
+ const clipped = clipToCeiling(wasteMsClaimed, stage, ceiling);
159
227
  if (stageInfo.gate >= SERIAL_GATE_THRESHOLD) {
160
228
  return { basis: 'serial', wallClock: { low: clipped, high: clipped } };
161
229
  }
@@ -2,7 +2,7 @@ import { Gunzip } from './vendor/fflate.js';
2
2
  import { createLz4BlockDecoder } from './lz4-block.js';
3
3
  import { Decompress as ZstdDecompress } from './vendor/fzstd.js';
4
4
  import { createSnappyBlockDecoder } from './snappy-block.js';
5
- import { createState, dispatchLine, buildChunkDecoder, emitParseCompletion, } from './event-handlers.js';
5
+ import { createState, dispatchLine, buildChunkDecoder, emitParseCompletion, } from './event-handlers.js';
6
6
  import { TASK_FIELD_NAMES } from './stage-quantiles.js';
7
7
  import { runParseFromUrl, sniffCodec } from './shs-fetch.js';
8
8
 
@@ -34,7 +34,9 @@ const MIN_PROGRESS_STEPS = 100;
34
34
  const PROGRESS_EMIT_LINES = 300;
35
35
 
36
36
 
37
-
37
+ // zstdDecoder replaces the vendored fzstd for zstd input; the Node CLI/MCP path passes
38
+ // cli/native-zstd.ts's native-zlib decoder, which a browser bundle can't import.
39
+
38
40
 
39
41
  // Minimal shape streamFile/runParse/runParseFiles read off `file` (name, size,
40
42
  // slice(start,end).arrayBuffer()): narrower than the full DOM `File`. A real
@@ -59,6 +61,10 @@ const PROGRESS_EMIT_LINES = 300;
59
61
  // without touching the vendored files (mirrors shs-fetch.ts's shim).
60
62
 
61
63
 
64
+ // A Node decoder may decompress off the main thread: streamFile awaits each push.
65
+
66
+
67
+
62
68
 
63
69
  // Stream one File's (possibly compressed) bytes through the codec dispatch,
64
70
  // in `chunkSize` slices, invoking `onChunk` with each decompressed buffer as
@@ -69,6 +75,7 @@ export async function streamFile(
69
75
  file ,
70
76
  onChunk ,
71
77
  chunkSize ,
78
+ zstdDecoder ,
72
79
  ) {
73
80
  const header = new Uint8Array(await file.slice(0, Math.min(8, file.size)).arrayBuffer());
74
81
  const codec = sniffCodec(header);
@@ -81,7 +88,10 @@ export async function streamFile(
81
88
  let currentPct = 0;
82
89
  const gunzip = codec === 'gz' ? new (Gunzip )((inflated) => onChunk(inflated, currentPct)) : null;
83
90
  const lz4 = codec === 'lz4' ? createLz4BlockDecoder((inflated) => onChunk(inflated, currentPct)) : null;
84
- const zstd = codec === 'zstd' ? new (ZstdDecompress )((inflated) => onChunk(inflated, currentPct)) : null;
91
+ const onZstdChunk = (inflated ) => onChunk(inflated, currentPct);
92
+ const zstd = codec !== 'zstd' ? null
93
+ : zstdDecoder ? zstdDecoder(onZstdChunk)
94
+ : new (ZstdDecompress )(onZstdChunk);
85
95
  const snappy = codec === 'snappy' ? createSnappyBlockDecoder((inflated) => onChunk(inflated, currentPct)) : null;
86
96
 
87
97
  // A fixed read size gives too few progress checkpoints on smaller files
@@ -99,7 +109,7 @@ export async function streamFile(
99
109
  currentPct = start / file.size;
100
110
  if (gunzip) gunzip.push(slice, final);
101
111
  else if (lz4) lz4.push(slice);
102
- else if (zstd) zstd.push(slice, final);
112
+ else if (zstd) await zstd.push(slice, final);
103
113
  else if (snappy) snappy.push(slice);
104
114
  else onChunk(slice, currentPct);
105
115
  }
@@ -115,7 +125,7 @@ export async function streamFile(
115
125
  export async function runParse(
116
126
  file ,
117
127
  state ,
118
- { emit = (msg ) => self.postMessage(msg), chunkSize = CHUNK_SIZE } = {},
128
+ { emit = (msg ) => self.postMessage(msg), chunkSize = CHUNK_SIZE, zstdDecoder } = {},
119
129
  ) {
120
130
  if (file.size === 0) {
121
131
  emit({ type: 'error', message: 'File is empty.' });
@@ -123,10 +133,13 @@ export async function runParse(
123
133
  }
124
134
 
125
135
  const decoder = buildChunkDecoder();
136
+ const joined = [];
126
137
  let linesProcessed = 0;
127
138
  const feed = (bytes , pct ) => {
128
- for (const line of decoder.decode(bytes)) {
129
- dispatchLine(line, state, emit);
139
+ joined.length = 0;
140
+ const lines = decoder.decode(bytes, joined);
141
+ for (let i = 0, j = 0; i < lines.length; i++) {
142
+ dispatchLine(lines[i], state, emit, joined[j]?.index === i ? joined[j++] : undefined);
130
143
  linesProcessed++;
131
144
  if (linesProcessed % PROGRESS_EMIT_LINES === 0) {
132
145
  emit({ type: 'progress', pct: pct ?? null, linesProcessed });
@@ -135,7 +148,7 @@ export async function runParse(
135
148
  };
136
149
 
137
150
  try {
138
- await streamFile(file, feed, chunkSize);
151
+ await streamFile(file, feed, chunkSize, zstdDecoder);
139
152
  } catch (e) {
140
153
  const message = e instanceof Error ? e.message : String(e);
141
154
  emit({ type: 'error', message: `Could not decompress "${file.name}": ${message}` });
@@ -147,7 +160,7 @@ export async function runParse(
147
160
  }
148
161
 
149
162
  if (!state.app) {
150
- emit({ type: 'error', message: 'Not a Spark event log: SparkListenerApplicationStart not found.' });
163
+ emit({ type: 'error', message: 'Not a Spark event log: no application-start event found. Choose a Spark event log file, or check the docs for supported formats.' });
151
164
  return;
152
165
  }
153
166
 
@@ -163,7 +176,7 @@ export async function runParse(
163
176
  export async function runParseFiles(
164
177
  files ,
165
178
  state ,
166
- { emit = (msg ) => self.postMessage(msg), chunkSize = CHUNK_SIZE } = {},
179
+ { emit = (msg ) => self.postMessage(msg), chunkSize = CHUNK_SIZE, zstdDecoder } = {},
167
180
  ) {
168
181
  if (files.length === 0) {
169
182
  emit({ type: 'error', message: 'Rolling event-log directory contained no event files.' });
@@ -171,13 +184,16 @@ export async function runParseFiles(
171
184
  }
172
185
 
173
186
  const decoder = buildChunkDecoder();
187
+ const joined = [];
174
188
  let linesProcessed = 0;
175
189
  const totalSize = files.reduce((sum, f) => sum + f.size, 0);
176
190
  let bytesBeforeCurrentFile = 0;
177
191
  let currentFileSize = 0;
178
192
  const feed = (bytes , pct ) => {
179
- for (const line of decoder.decode(bytes)) {
180
- dispatchLine(line, state, emit);
193
+ joined.length = 0;
194
+ const lines = decoder.decode(bytes, joined);
195
+ for (let i = 0, j = 0; i < lines.length; i++) {
196
+ dispatchLine(lines[i], state, emit, joined[j]?.index === i ? joined[j++] : undefined);
181
197
  linesProcessed++;
182
198
  if (linesProcessed % PROGRESS_EMIT_LINES === 0) {
183
199
  const overallPct = totalSize > 0 ? (bytesBeforeCurrentFile + (pct ?? 0) * currentFileSize) / totalSize : null;
@@ -189,7 +205,7 @@ export async function runParseFiles(
189
205
  for (const file of files) {
190
206
  currentFileSize = file.size;
191
207
  try {
192
- await streamFile(file, feed, chunkSize);
208
+ await streamFile(file, feed, chunkSize, zstdDecoder);
193
209
  } catch (e) {
194
210
  const message = e instanceof Error ? e.message : String(e);
195
211
  emit({ type: 'error', message: `Could not decompress "${file.name}": ${message}` });
@@ -203,7 +219,7 @@ export async function runParseFiles(
203
219
  }
204
220
 
205
221
  if (!state.app) {
206
- emit({ type: 'error', message: 'Not a Spark event log: SparkListenerApplicationStart not found.' });
222
+ emit({ type: 'error', message: 'Not a Spark event log: no application-start event found. Choose a Spark event log file, or check the docs for supported formats.' });
207
223
  return;
208
224
  }
209
225
 
@@ -55,6 +55,10 @@ export function scanRelationId(name , detail ) {
55
55
  // Delta tables). All catalog tables surface as `spark_catalog.<db>.<table>`.
56
56
  let format = null, table = null;
57
57
  const nameM = name.match(/^Scan\s+(parquet|orc|csv|json)\s+(spark_catalog\.\S+)$/i);
58
+ // Fast reject: every branch below that returns a key needs this name match, a FileScan
59
+ // detail or a JDBCRelation detail. Most plan nodes (Project, Filter, joins) have none, and
60
+ // their long details otherwise pay every regex below.
61
+ if (!nameM && !/FileScan/i.test(detail) && !detail.includes('JDBCRelation')) return null;
58
62
  if (nameM) { format = nameM[1].toLowerCase(); table = nameM[2]; }
59
63
  else {
60
64
  const detM = detail.match(/FileScan\s+(parquet|orc|csv|json)\s+(spark_catalog\.[^[\s]+)\[/i);