sparkforensics-mcp 0.2.3 → 0.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/LICENSE ADDED
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Agustin Recoba
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
package/README.md CHANGED
@@ -2,5 +2,11 @@
2
2
 
3
3
  MCP server exposing Apache Spark event-log diagnostics over stdio.
4
4
 
5
+ ```bash
6
+ npx sparkforensics-mcp
7
+ ```
8
+
9
+ Point your MCP client at that command.
10
+
5
11
  See the [main README](https://github.com/shuffle-works/sparkforensics#readme)
6
- for install instructions and the full tool reference.
12
+ for the full tool reference.
@@ -19,20 +19,26 @@ async function loadCreateMcpServer() {
19
19
  return mod.createMcpServer;
20
20
  }
21
21
 
22
- const USAGE = `Usage: sparkforensics-mcp
22
+ // Tool names come from the server the bin actually builds, so --help can't drift
23
+ // from the registered set. Building the server registers tools only: no transport,
24
+ // no I/O. _registeredTools is the SDK's registry; the MCP package test checks that
25
+ // this list matches what listTools reports.
26
+ function usage(toolNames) {
27
+ return `Usage: sparkforensics-mcp
23
28
 
24
29
  Starts the SparkForensics MCP server, speaking the MCP protocol over
25
30
  stdio. Point an MCP client (Claude Desktop, Claude Code, etc.) at this
26
- command; it exposes five tools for diagnosing Apache Spark event logs:
27
- diagnose_run, get_run_summary, compare_runs, evaluate_budgets,
28
- get_finding_evidence.
31
+ command; it exposes ${toolNames.length} tools for diagnosing Apache Spark event logs:
32
+ ${toolNames.join(', ')}.
29
33
 
30
34
  See https://github.com/shuffle-works/sparkforensics#readme for details.
31
35
  `;
36
+ }
32
37
 
33
38
  async function main() {
34
39
  if (process.argv.includes('--help') || process.argv.includes('-h')) {
35
- process.stderr.write(USAGE);
40
+ const createMcpServer = await loadCreateMcpServer();
41
+ process.stderr.write(usage(Object.keys(createMcpServer()._registeredTools)));
36
42
  return;
37
43
  }
38
44
  const createMcpServer = await loadCreateMcpServer();
package/package.json CHANGED
@@ -1,10 +1,10 @@
1
1
  {
2
2
  "name": "sparkforensics-mcp",
3
- "version": "0.2.3",
3
+ "version": "0.3.0",
4
4
  "mcpName": "io.github.shuffle-works/sparkforensics-mcp",
5
5
  "type": "module",
6
6
  "license": "MIT",
7
- "description": "MCP server exposing Apache Spark event-log diagnostics (diagnose_run, get_run_summary, compare_runs, get_finding_evidence, evaluate_budgets) over stdio.",
7
+ "description": "MCP server exposing Apache Spark event-log diagnostics over stdio.",
8
8
  "repository": {
9
9
  "type": "git",
10
10
  "url": "git+https://github.com/shuffle-works/sparkforensics.git"
@@ -23,7 +23,7 @@
23
23
  "performance"
24
24
  ],
25
25
  "bin": {
26
- "sparkforensics-mcp": "./bin/sparkforensics-mcp.mjs"
26
+ "sparkforensics-mcp": "bin/sparkforensics-mcp.mjs"
27
27
  },
28
28
  "engines": {
29
29
  "node": ">=18"
@@ -343,8 +343,8 @@ export function createThreadedZstdDecoder(
343
343
  }
344
344
 
345
345
  // Node's native zstd where this Node has it (22.15+/23.8+); older Nodes keep the vendored fzstd.
346
- // runParse/runParseFiles (collectRun) take the threaded decoder. decodeShsArchive decodes each
347
- // archive entry in one synchronous call, so the SHS archive loader takes the inline one.
346
+ // runParse/runParseFiles (collectRun) take the threaded decoder; the SHS archive loader
347
+ // (decodeShsArchive over an in-memory download) takes the inline one.
348
348
  export const nodeParseCodecs =
349
349
  nativeZstdAvailable ? { zstdDecoder: createThreadedZstdDecoder } : {};
350
350
  export const nodeArchiveCodecs =
@@ -6,6 +6,7 @@ import { computeCoreLocalityRatio } from './core-locality-ratio.js';
6
6
  import { estimateSingleStage, tailRecoveryMs, tailRemovedWorkMs, stragglerFixLongestTaskMs, } from './occupancy.js';
7
7
  import { isExchangeNode, isBroadcastExchangeNode } from './plan-node-detail.js';
8
8
  import { cyrb53 } from './string-hash.js';
9
+ import { MAX_FAILURE_GROUPS, describeTaskFailure, } from './task-failure.js';
9
10
 
10
11
 
11
12
  const MB = 1024 * 1024;
@@ -73,6 +74,7 @@ const TB = 1024 * GB;
73
74
 
74
75
 
75
76
 
77
+
76
78
 
77
79
 
78
80
 
@@ -1206,7 +1208,7 @@ export const DETECTORS = [
1206
1208
  },
1207
1209
  },
1208
1210
  {
1209
- type: 'failures', scope: 'stage', order: 40, fixEffort: 'code', version: 1,
1211
+ type: 'failures', scope: 'stage', order: 40, fixEffort: 'code', version: 2,
1210
1212
  docAnchor: '#bottleneck-failures',
1211
1213
  thresholds: { minTasks: 10, warnRate: 0.05, critRate: 0.20 },
1212
1214
  detect(
@@ -1219,13 +1221,25 @@ export const DETECTORS = [
1219
1221
  if (failureRate <= this.thresholds.warnRate) return null;
1220
1222
  const value = Math.round(failureRate * 1000) / 10;
1221
1223
  const dominantReason = pickDominantReason(stage.failureReasons);
1224
+ // Groups arrive most frequent first. Name the dominant error from the largest group under the
1225
+ // dominant tag, so the error and the tag agree even when one tag splits into many messages.
1226
+ const allGroups = stage.failureGroups ?? [];
1227
+ const dominantGroup = allGroups.find((g) => g.reason === dominantReason);
1228
+ const dominantError = (dominantGroup ? describeTaskFailure(dominantGroup) : null) ?? dominantReason;
1229
+ const failureGroups = allGroups.slice(0, MAX_FAILURE_GROUPS);
1230
+ const groupedTasks = failureGroups.reduce((sum, g) => sum + g.count, 0);
1222
1231
  return {
1223
1232
  type: 'failures', stageId: stage.id,
1224
1233
  impactBand: failureRate > this.thresholds.critRate ? 'critical' : 'warning',
1225
1234
  metric: 'failureRate', value,
1226
1235
  failedTasks: stage.failedTasks,
1227
1236
  dominantReason,
1228
- recommendation: `${value}% of tasks failed${dominantReason ? ` (dominant reason: ${dominantReason})` : ''}: investigate driver logs for executor instability or data-driven errors.`,
1237
+ dominantError,
1238
+ // One entry per distinct error (tag, class, message, loss reason), each with one bounded
1239
+ // stack excerpt; otherFailedTasks counts the failed tasks no shown group covers.
1240
+ failureGroups,
1241
+ otherFailedTasks: Math.max(0, stage.failedTasks - groupedTasks),
1242
+ recommendation: `${value}% of tasks failed${dominantError ? ` (dominant error: ${dominantError})` : ''}: investigate driver logs for executor instability or data-driven errors.`,
1229
1243
  };
1230
1244
  },
1231
1245
  },
@@ -1,5 +1,9 @@
1
1
  ### `FAIL`: Failed tasks {#fail}
2
2
 
3
3
  Tasks fail often enough to affect the stage. Failed tasks point to executor
4
- instability or data-driven errors: check driver logs for the dominant
5
- failure reason.
4
+ instability or data-driven errors. The finding names the dominant error: the
5
+ exception class, or the executor loss reason (for example "Container killed
6
+ by YARN for exceeding memory limits"). It lists up to five distinct failures,
7
+ each with its message and a short stack excerpt. With redaction on, messages
8
+ and the message text inside excerpts are replaced, since they can carry file
9
+ paths and data values; class names and stack frames stay.
@@ -19,6 +19,7 @@ import {
19
19
  } from './event-schemas.js';
20
20
  import { assertNever } from './assert-never.js';
21
21
  import { finalizeStage } from './stage-quantiles.js';
22
+ import { MAX_FAILURE_DETAILS_PER_STAGE, extractTaskFailureDetail, taskFailureKey, } from './task-failure.js';
22
23
  import { computeRunAggregates } from './run-aggregates.js';
23
24
 
24
25
 
@@ -94,6 +95,9 @@ const MAX_TASK_SAMPLES = 20;
94
95
 
95
96
 
96
97
 
98
+
99
+
100
+
97
101
 
98
102
 
99
103
 
@@ -136,6 +140,8 @@ const MAX_TASK_SAMPLES = 20;
136
140
 
137
141
 
138
142
 
143
+
144
+
139
145
 
140
146
 
141
147
 
@@ -485,6 +491,19 @@ function taskRecordToSample(t ) {
485
491
  };
486
492
  }
487
493
 
494
+ // One shared detail object per distinct failure, so a stage with thousands of failed attempts
495
+ // holds each bounded stack excerpt once. Past the cap an attempt keeps only its reason tag.
496
+ function internTaskFailure(stage , endReason ) {
497
+ const detail = extractTaskFailureDetail(endReason);
498
+ if (!detail || !stage.failureDetails) return null;
499
+ const key = taskFailureKey(detail);
500
+ const known = stage.failureDetails.get(key);
501
+ if (known) return known;
502
+ if (stage.failureDetails.size >= MAX_FAILURE_DETAILS_PER_STAGE) return null;
503
+ stage.failureDetails.set(key, detail);
504
+ return detail;
505
+ }
506
+
488
507
  export function accumulateTask(event , state ) {
489
508
  const stageId = event['Stage ID'];
490
509
  const stage = state.stages.get(stageId);
@@ -521,6 +540,7 @@ export function accumulateTask(event , state
521
540
  launchTime: info['Launch Time'] ?? 0,
522
541
  finishTime: info['Finish Time'] ?? 0,
523
542
  reason: event['Task End Reason']?.['Reason'] ?? null,
543
+ failure: failed ? internTaskFailure(stage, event['Task End Reason']) : null,
524
544
  speculative: info['Speculative'] === true,
525
545
  host: info['Host'] ?? '',
526
546
  executorId: info['Executor ID'] ?? '',
@@ -802,6 +822,7 @@ export function submitStage(event , st
802
822
  failureReasons: new Map(),
803
823
  stageFailureReason: null,
804
824
  taskAttempts: new Map(),
825
+ failureDetails: new Map(),
805
826
  retryTaskSamples: [],
806
827
  retryWasteMs: 0,
807
828
  wastedAttempts: 0,
@@ -238,6 +238,14 @@ export const TaskEndEventSchema = z.object({
238
238
  'Stage Attempt ID': z.number().optional(),
239
239
  'Task End Reason': z.object({
240
240
  Reason: z.string().optional(),
241
+ // The error behind a failed attempt, read by extractTaskFailureDetail (task-failure.ts), which
242
+ // keeps only string values: unknown here so an unexpected shape drops the detail, never the task.
243
+ 'Class Name': z.unknown().optional(),
244
+ Description: z.unknown().optional(),
245
+ 'Full Stack Trace': z.unknown().optional(),
246
+ 'Loss Reason': z.unknown().optional(),
247
+ Message: z.unknown().optional(),
248
+ 'Kill Reason': z.unknown().optional(),
241
249
  }).optional(),
242
250
  'Task Info': z.object({
243
251
  'Launch Time': z.number().optional(),
@@ -6,6 +6,7 @@ import { detectorCatalog } from './detectors.js';
6
6
  import { typeTag, formatBytes, formatDuration, IMPACT_BAND_ORDER } from './format-utils.js';
7
7
  import { FINDING_NAMES, titleCase } from './finding-names.js';
8
8
  import { redactReport } from './redact.js';
9
+ import { formatTaskFailureHeadline, } from './task-failure.js';
9
10
  import { coreFindingActionLabel } from './finding-action-label.js';
10
11
  import { matchesFindingFilterCriteria } from './finding-filter-predicate.js';
11
12
  import { buildRecommendationRollup, isEligible, rankFindings, } from './recommendation-rollup.js';
@@ -269,6 +270,21 @@ function renderEvidenceValue(key , value ) {
269
270
  return String(value);
270
271
  }
271
272
 
273
+ // The `failures` finding's distinct errors: a headline per group, its stack excerpt as an indented
274
+ // code block (indented, not fenced, so no excerpt content can close it early).
275
+ function renderFailureGroups(groups ) {
276
+ const lines = [` - failureGroups: ${groups.length}`];
277
+ for (const g of groups) {
278
+ lines.push(` - ${g.count} task(s): ${formatTaskFailureHeadline(g)}`);
279
+ if (g.stackExcerpt) {
280
+ lines.push('');
281
+ for (const l of g.stackExcerpt.split('\n')) lines.push(` ${l}`);
282
+ lines.push('');
283
+ }
284
+ }
285
+ return lines;
286
+ }
287
+
272
288
  function formatWallClockRange(low , high ) {
273
289
  const fmtMs = (ms ) => (ms === 0 ? '0s' : formatDuration(ms));
274
290
  return low === high ? `Estimated ${fmtMs(high)}` : `Estimated ${fmtMs(low)}-${fmtMs(high)}`;
@@ -336,7 +352,10 @@ function renderMarkdown(json ) {
336
352
  const evidence = Object.entries(r.evidence ?? {}).sort(([a], [b]) => (a < b ? -1 : a > b ? 1 : 0));
337
353
  if (evidence.length) {
338
354
  lines.push('- evidence:');
339
- for (const [k, v] of evidence) lines.push(` - ${k}: ${renderEvidenceValue(k, v)}`);
355
+ for (const [k, v] of evidence) {
356
+ if (k === 'failureGroups' && Array.isArray(v)) lines.push(...renderFailureGroups(v ));
357
+ else lines.push(` - ${k}: ${renderEvidenceValue(k, v)}`);
358
+ }
340
359
  }
341
360
  lines.push('');
342
361
  }
@@ -1,4 +1,5 @@
1
1
 
2
+ import { nsToMs } from './format-utils.js';
2
3
  import {
3
4
  computeOccupancy, estimateSingleStage, estimateMultiStage, tailRecoveryMs, tailRemovedWorkMs, stragglerFixLongestTaskMs,
4
5
 
@@ -82,6 +83,24 @@ const EXECUTOR_STARTUP_OVERHEAD_MS = 15000;
82
83
  // Assumed re-read throughput, shared by cachingOpportunity and cacheUtilization.
83
84
  const RE_READ_THROUGHPUT_BPS = 125_000_000;
84
85
 
86
+ // Below this share of executorRunTime spent on CPU, a stage's tasks were idle, waiting on something
87
+ // outside Spark: on 14 real logs (2026-09-23) every non-Python stage under 1% was a JDBC read, a
88
+ // file listing or a Delta log read, while file writes, which more partitions do parallelize,
89
+ // start at 2%.
90
+ const IDLE_CPU_SHARE_MAX = 0.01;
91
+
92
+ // True when the stage's tasks spent under IDLE_CPU_SHARE_MAX of their run time on CPU. False when
93
+ // the share can't be trusted: no CPU time recorded (older Spark logs omit the metric), or Python
94
+ // code run through PythonRDD, whose worker-process CPU executorCpuTime (the JVM task thread's)
95
+ // never counts (such stages read 0.1% on the same logs while computing).
96
+ function tasksMostlyIdle(stage ) {
97
+ const runMs = stage.executorRunTime ?? 0;
98
+ const cpuMs = nsToMs(stage.executorCpuTime ?? 0);
99
+ if (runMs <= 0 || cpuMs <= 0) return false;
100
+ if (/PythonRDD/.test(stage.name ?? '') || /org\.apache\.spark\.api\.python\./.test(stage.details ?? '')) return false;
101
+ return cpuMs / runMs < IDLE_CPU_SHARE_MAX;
102
+ }
103
+
85
104
  // skew and straggler claim time off the stage's longest task itself, so the occupancy clip must
86
105
  // not floor them at that same task (see estimateSingleStage). detectors.ts's clippedWasteMs gates
87
106
  // both detectors on the same option so the firing floor and the displayed estimate agree.
@@ -332,14 +351,15 @@ function computeEstimateForFinding(
332
351
  // is queueing no partition count recovers. Splitting partitions splits the longest task
333
352
  // too, hence TAIL_CLAIM's post-fix floor. Unknown cluster size: no defensible figure.
334
353
  if (totalCores <= 0) return costOnly('modeled');
335
- // A stage that read no input and no shuffle has no data for more partitions to split (a
336
- // 1-task count stage open 27 minutes on 5s of CPU was claimed 99% recoverable): claim 0.
354
+ // A stage that read no input and no shuffle, its tasks idle waiting on an external system,
355
+ // gains nothing from more partitions (a 1-task JDBC count stage open 27 minutes on 5s of CPU
356
+ // was claimed 99% recoverable): claim 0. Stages that read bytes keep their claim.
337
357
  const readBytes = (stage.inputBytes ?? 0) + (stage.shuffleReadBytes ?? 0);
338
358
  const activeMs = typeof stage.taskActiveMs === 'number'
339
359
  ? stage.taskActiveMs
340
360
  : Math.max(0, (stage.completedAt ?? 0) - (stage.submittedAt ?? 0));
341
361
  const taskCount = stage.taskCount ?? 0;
342
- const wasteMs = readBytes > 0 ? activeMs * Math.max(0, 1 - taskCount / totalCores) : 0;
362
+ const wasteMs = readBytes <= 0 && tasksMostlyIdle(stage) ? 0 : activeMs * Math.max(0, 1 - taskCount / totalCores);
343
363
  return singleStageImpact(wasteMs, finding.stageId, stages, occupancy, 'modeled', { value: wasteMs, unit: 'ms' }, TAIL_CLAIM);
344
364
  }
345
365
  case 'partitionSizing': {
@@ -152,18 +152,19 @@ const SERIAL_GATE_THRESHOLD = 0.999;
152
152
 
153
153
 
154
154
 
155
+
155
156
 
156
157
 
157
158
 
158
159
 
159
- // Wall-clock a skew/straggler fix recovers, given the slowest task's excess over the median. A
160
- // lone straggler costs that excess; a tail of many slow tasks (a bimodal stage: 26% of 1400
161
- // tasks over 4x P50 on a real log) costs its summed excess (stragglerExcessMs) spread over the
162
- // slots the stage had (peakConcurrentTasks), far more than one task's. The task-level replay in
163
- // dev/eval-tail-replay.mjs recovers about the larger of the two. Average concurrency would be the
164
- // wrong divisor: a tail-dominated stage runs few tasks for most of its span (5.7 average vs 14
165
- // peak on one real stage), which doubled the claim.
160
+ // Wall-clock a skew/straggler fix recovers: finalizeStage's task-level replay
161
+ // (tailReplayRecoveryMs, computeTailReplayRecoveryMs) when the stage carries it. Without it (a
162
+ // stage built by hand, as in detector tests) an estimate from the slowest task's excess over the
163
+ // median: a lone straggler costs that excess; a tail of many slow tasks costs its summed excess
164
+ // (stragglerExcessMs) spread over the slots the stage had (peakConcurrentTasks), and the replay
165
+ // recovers about the larger of the two.
166
166
  export function tailRecoveryMs(stage , singleTaskExcessMs ) {
167
+ if (stage.tailReplayRecoveryMs != null) return stage.tailReplayRecoveryMs;
167
168
  const excessMs = stage.stragglerExcessMs ?? 0;
168
169
  const slots = stage.peakConcurrentTasks ?? 0;
169
170
  if (excessMs <= 0 || slots <= 0) return singleTaskExcessMs;
@@ -4,7 +4,9 @@ import { Decompress as ZstdDecompress } from './vendor/fzstd.js';
4
4
  import { createSnappyBlockDecoder } from './snappy-block.js';
5
5
  import { createState, dispatchLine, buildChunkDecoder, emitParseCompletion, } from './event-handlers.js';
6
6
  import { TASK_FIELD_NAMES } from './stage-quantiles.js';
7
- import { runParseFromUrl, sniffCodec } from './shs-fetch.js';
7
+ import { runParseFromUrl, sniffCodec, parseZipArchive } from './shs-fetch.js';
8
+ import { isZip } from './zip-archive.js';
9
+ import { createWorkerZstdDecoders } from './zstd-worker-client.js';
8
10
 
9
11
  export {
10
12
  buildChunkDecoder, createState, normalizeSparkProperties, parseSparkMemoryMB, extractResources,
@@ -34,6 +36,8 @@ const MIN_PROGRESS_STEPS = 100;
34
36
  const PROGRESS_EMIT_LINES = 300;
35
37
 
36
38
 
39
+ const NOT_AN_EVENT_LOG = 'Not a Spark event log: no application-start event found. Choose a Spark event log file, or check the docs for supported formats.';
40
+
37
41
  // zstdDecoder replaces the vendored fzstd for zstd input; the Node CLI/MCP path passes
38
42
  // cli/native-zstd.ts's native-zlib decoder, which a browser bundle can't import.
39
43
 
@@ -61,9 +65,11 @@ const PROGRESS_EMIT_LINES = 300;
61
65
  // without touching the vendored files (mirrors shs-fetch.ts's shim).
62
66
 
63
67
 
64
- // A Node decoder may decompress off the main thread: streamFile awaits each push.
68
+ // A Node decoder, or the browser's decompress worker (zstd-worker-client.ts), may decompress off
69
+ // the calling thread: streamFile awaits each push, and calls cancel() when it abandons the stream.
65
70
 
66
71
 
72
+
67
73
 
68
74
 
69
75
  // Stream one File's (possibly compressed) bytes through the codec dispatch,
@@ -89,7 +95,7 @@ export async function streamFile(
89
95
  const gunzip = codec === 'gz' ? new (Gunzip )((inflated) => onChunk(inflated, currentPct)) : null;
90
96
  const lz4 = codec === 'lz4' ? createLz4BlockDecoder((inflated) => onChunk(inflated, currentPct)) : null;
91
97
  const onZstdChunk = (inflated ) => onChunk(inflated, currentPct);
92
- const zstd = codec !== 'zstd' ? null
98
+ const zstd = codec !== 'zstd' ? null
93
99
  : zstdDecoder ? zstdDecoder(onZstdChunk)
94
100
  : new (ZstdDecompress )(onZstdChunk);
95
101
  const snappy = codec === 'snappy' ? createSnappyBlockDecoder((inflated) => onChunk(inflated, currentPct)) : null;
@@ -101,17 +107,22 @@ export async function streamFile(
101
107
  const stepSize = Math.max(1, Math.min(chunkSize, Math.ceil(file.size / MIN_PROGRESS_STEPS)));
102
108
 
103
109
  let offset = 0;
104
- while (offset < file.size) {
105
- const start = offset;
106
- const slice = new Uint8Array(await file.slice(start, start + stepSize).arrayBuffer());
107
- offset += stepSize;
108
- const final = offset >= file.size;
109
- currentPct = start / file.size;
110
- if (gunzip) gunzip.push(slice, final);
111
- else if (lz4) lz4.push(slice);
112
- else if (zstd) await zstd.push(slice, final);
113
- else if (snappy) snappy.push(slice);
114
- else onChunk(slice, currentPct);
110
+ try {
111
+ while (offset < file.size) {
112
+ const start = offset;
113
+ const slice = new Uint8Array(await file.slice(start, start + stepSize).arrayBuffer());
114
+ offset += stepSize;
115
+ const final = offset >= file.size;
116
+ currentPct = start / file.size;
117
+ if (gunzip) gunzip.push(slice, final);
118
+ else if (lz4) lz4.push(slice);
119
+ else if (zstd) await zstd.push(slice, final);
120
+ else if (snappy) snappy.push(slice);
121
+ else onChunk(slice, currentPct);
122
+ }
123
+ } catch (e) {
124
+ zstd?.cancel?.();
125
+ throw e;
115
126
  }
116
127
  if (lz4) lz4.end();
117
128
  if (snappy) snappy.end();
@@ -132,6 +143,20 @@ export async function runParse(
132
143
  return;
133
144
  }
134
145
 
146
+ // A Spark History Server download (the UI's download link, or GET
147
+ // /api/v1/applications/<id>/logs) is a zip holding the log file, or a
148
+ // rolling log's parts: unwrap it through the same path the SHS fetch uses.
149
+ if (isZip(new Uint8Array(await file.slice(0, Math.min(4, file.size)).arrayBuffer()))) {
150
+ await parseZipArchive(file, state, emit, {
151
+ zstdDecoder,
152
+ chunkSize,
153
+ progressEvery: PROGRESS_EMIT_LINES,
154
+ reportPct: true,
155
+ onInvalid: (detail) => emit({ type: 'error', message: detail ?? NOT_AN_EVENT_LOG }),
156
+ });
157
+ return;
158
+ }
159
+
135
160
  const decoder = buildChunkDecoder();
136
161
  const joined = [];
137
162
  let linesProcessed = 0;
@@ -160,7 +185,7 @@ export async function runParse(
160
185
  }
161
186
 
162
187
  if (!state.app) {
163
- emit({ type: 'error', message: 'Not a Spark event log: no application-start event found. Choose a Spark event log file, or check the docs for supported formats.' });
188
+ emit({ type: 'error', message: NOT_AN_EVENT_LOG });
164
189
  return;
165
190
  }
166
191
 
@@ -219,7 +244,7 @@ export async function runParseFiles(
219
244
  }
220
245
 
221
246
  if (!state.app) {
222
- emit({ type: 'error', message: 'Not a Spark event log: no application-start event found. Choose a Spark event log file, or check the docs for supported formats.' });
247
+ emit({ type: 'error', message: NOT_AN_EVENT_LOG });
223
248
  return;
224
249
  }
225
250
 
@@ -232,17 +257,25 @@ const isWorker = typeof WorkerGlobalScope !== 'undefined' && self instanceof Wor
232
257
 
233
258
  if (isWorker) {
234
259
  let workerState = null;
260
+ // Dropped zstd files decompress in a second worker, overlapping with parsing here. The
261
+ // `new Worker(new URL(...))` stays inline for Vite's worker detection (see ingest.ts). The SHS
262
+ // path (runParseFromUrl) keeps in-thread fzstd.
263
+ const zstdDecoder = createWorkerZstdDecoders(
264
+ () => new Worker(new URL('./zstd-worker.js', import.meta.url), { type: 'module' }),
265
+ (onChunk) => new (ZstdDecompress )(onChunk),
266
+ { onFallback: (reason) => console.warn(`zstd: decompressing on the parse worker: ${reason}`) },
267
+ );
235
268
 
236
269
  self.onmessage = async ({ data } ) => {
237
270
  if (data.type === 'parse') {
238
271
  workerState = createState();
239
- await runParse(data.file, workerState);
272
+ await runParse(data.file, workerState, { zstdDecoder });
240
273
  } else if (data.type === 'parseFromUrl') {
241
274
  workerState = createState();
242
275
  await runParseFromUrl(data.request, workerState);
243
276
  } else if (data.type === 'parseFiles') {
244
277
  workerState = createState();
245
- await runParseFiles(data.files, workerState);
278
+ await runParseFiles(data.files, workerState, { zstdDecoder });
246
279
  } else if (data.type === 'getTaskData') {
247
280
  const { stageId, reqId } = data;
248
281
  const stored = workerState?.taskStore.get(stageId) ?? new Float64Array(0);
@@ -45,7 +45,7 @@ function pushLongFilterWarning(result , len ) {
45
45
  }
46
46
 
47
47
  // Stable relation-identity key for a scan node: "<format>:<relation>" (e.g.
48
- // "delta:mx.t_emp_whitelist", "parquet:business_prd.sales", "jdbc:dw.d_producto")
48
+ // "delta:mx.store_map", "parquet:warehouse.sales", "jdbc:dw.dim_product")
49
49
  // or null when the node is internal Delta metadata / a non-scan / un-nameable.
50
50
  // The single source of truth for scan identity, shared by `visitScan` (summary)
51
51
  // and the `cachingOpportunity` detector so the regexes live in one place.
@@ -8,6 +8,7 @@
8
8
  // and non-mutating (returns a fresh, deep-copied tree).
9
9
 
10
10
 
11
+ import { redactTaskFailureGroup, } from './task-failure.js';
11
12
 
12
13
  // Host / IP identifier patterns. Used to enumerate host names that surface only
13
14
  // inside free text: recommendation strings, `stageFailed`'s failure-reason
@@ -79,6 +80,25 @@ function collectHostFields(node , hosts ) {
79
80
  }
80
81
  }
81
82
 
83
+ // A failed-task error's message and stack text can carry file paths and data values that no
84
+ // host/app-id pattern recognizes, so they are dropped rather than pseudonymized: every array under
85
+ // a key named `failureGroups` (a `failures` finding's, or a stage's in the HTML export) gets
86
+ // redactTaskFailureGroup applied. Walking by key name, like collectHostFields, needs no path list.
87
+ // Returns a fresh tree.
88
+ function redactFailureGroups (node ) {
89
+ if (Array.isArray(node)) return node.map((n) => redactFailureGroups(n)) ;
90
+ if (node && typeof node === 'object') {
91
+ const out = {};
92
+ for (const [k, v] of Object.entries(node)) {
93
+ out[k] = k === 'failureGroups' && Array.isArray(v)
94
+ ? v.map((g) => (g && typeof g === 'object' ? redactTaskFailureGroup(g ) : g))
95
+ : redactFailureGroups(v);
96
+ }
97
+ return out ;
98
+ }
99
+ return node;
100
+ }
101
+
82
102
  // Spark config keys ending in `host`/`hostname` (e.g. spark.driver.host,
83
103
  // spark.yarn.am.hostname) carry plain FQDN host names that neither
84
104
  // HOST_PATTERNS matches (no IP/EC2 shape) nor collectHostFields's by-key-name
@@ -172,7 +192,8 @@ function applyReplacements (node , ids
172
192
  return deepReplace(node, merged) ;
173
193
  }
174
194
 
175
- export function redactReport (report ) {
195
+ export function redactReport (input ) {
196
+ const report = redactFailureGroups(input);
176
197
  const { appIds, hosts } = collectIds(report);
177
198
  return applyReplacements(report, { appIds, hosts });
178
199
  }
@@ -216,7 +237,8 @@ export function redactComparison (comparison ) {
216
237
  // this also walks executors.added/removed for their literal `host` field
217
238
  // (ExecutorAddedEvent.host), since raw executor records: not just findings
218
239
  //: reach data.js.
219
- export function redactExportData(data ) {
240
+ export function redactExportData(input ) {
241
+ const data = redactFailureGroups(input);
220
242
  const appIds = new Set ();
221
243
  const hosts = new Set ();
222
244
  const appId = data.app?.id;
@@ -180,6 +180,14 @@ function skewRatios(stages ) {
180
180
  return out;
181
181
  }
182
182
 
183
+ // Every key metricDeltas() emits, in emission order. The CLI validates
184
+ // --regression-metric against it, so a misspelled key is a usage error rather
185
+ // than an inconclusive budget. A contract test keeps it in sync.
186
+ export const COMPARISON_METRIC_KEYS = [
187
+ 'wallClock', 'shuffleSpill', 'taskSkew', 'failedTaskRate', 'diskSpill', 'gcTime',
188
+ 'inputBytes', 'outputBytes', 'executorRunTime', 'taskCount', 'executorsAdded',
189
+ ];
190
+
183
191
  // Volume/count metrics, not cost metrics: more or less input/output data, or
184
192
  // tasks/executors, isn't inherently better or worse (it may just reflect a
185
193
  // differently-sized job), unlike wall-clock, spill, GC, etc. Exported as the