sparkforensics-mcp 0.2.3 → 0.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -0
- package/README.md +7 -1
- package/bin/sparkforensics-mcp.mjs +11 -5
- package/package.json +3 -3
- package/vendor-core/cli/native-zstd.js +2 -2
- package/vendor-core/detectors.js +16 -2
- package/vendor-core/docs-content/detection/fail.md +6 -2
- package/vendor-core/event-handlers.js +21 -0
- package/vendor-core/event-schemas.js +8 -0
- package/vendor-core/evidence-report.js +20 -1
- package/vendor-core/impact-estimator.js +23 -3
- package/vendor-core/occupancy.js +8 -7
- package/vendor-core/parser-worker.js +51 -18
- package/vendor-core/plan-summary.js +1 -1
- package/vendor-core/redact.js +24 -2
- package/vendor-core/run-comparison.js +8 -0
- package/vendor-core/shs-fetch.js +175 -65
- package/vendor-core/shs-load.js +1 -1
- package/vendor-core/stage-quantiles.js +65 -1
- package/vendor-core/task-failure.js +151 -0
- package/vendor-core/types.js +5 -0
- package/vendor-core/vendor/fflate.js +1 -1
- package/vendor-core/zip-archive.js +167 -0
- package/vendor-core/zstd-worker-client.js +180 -0
- package/vendor-core/zstd-worker.js +103 -0
package/LICENSE
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Agustin Recoba
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
package/README.md
CHANGED
|
@@ -2,5 +2,11 @@
|
|
|
2
2
|
|
|
3
3
|
MCP server exposing Apache Spark event-log diagnostics over stdio.
|
|
4
4
|
|
|
5
|
+
```bash
|
|
6
|
+
npx sparkforensics-mcp
|
|
7
|
+
```
|
|
8
|
+
|
|
9
|
+
Point your MCP client at that command.
|
|
10
|
+
|
|
5
11
|
See the [main README](https://github.com/shuffle-works/sparkforensics#readme)
|
|
6
|
-
for
|
|
12
|
+
for the full tool reference.
|
|
@@ -19,20 +19,26 @@ async function loadCreateMcpServer() {
|
|
|
19
19
|
return mod.createMcpServer;
|
|
20
20
|
}
|
|
21
21
|
|
|
22
|
-
|
|
22
|
+
// Tool names come from the server the bin actually builds, so --help can't drift
|
|
23
|
+
// from the registered set. Building the server registers tools only: no transport,
|
|
24
|
+
// no I/O. _registeredTools is the SDK's registry; the MCP package test checks that
|
|
25
|
+
// this list matches what listTools reports.
|
|
26
|
+
function usage(toolNames) {
|
|
27
|
+
return `Usage: sparkforensics-mcp
|
|
23
28
|
|
|
24
29
|
Starts the SparkForensics MCP server, speaking the MCP protocol over
|
|
25
30
|
stdio. Point an MCP client (Claude Desktop, Claude Code, etc.) at this
|
|
26
|
-
command; it exposes
|
|
27
|
-
|
|
28
|
-
get_finding_evidence.
|
|
31
|
+
command; it exposes ${toolNames.length} tools for diagnosing Apache Spark event logs:
|
|
32
|
+
${toolNames.join(', ')}.
|
|
29
33
|
|
|
30
34
|
See https://github.com/shuffle-works/sparkforensics#readme for details.
|
|
31
35
|
`;
|
|
36
|
+
}
|
|
32
37
|
|
|
33
38
|
async function main() {
|
|
34
39
|
if (process.argv.includes('--help') || process.argv.includes('-h')) {
|
|
35
|
-
|
|
40
|
+
const createMcpServer = await loadCreateMcpServer();
|
|
41
|
+
process.stderr.write(usage(Object.keys(createMcpServer()._registeredTools)));
|
|
36
42
|
return;
|
|
37
43
|
}
|
|
38
44
|
const createMcpServer = await loadCreateMcpServer();
|
package/package.json
CHANGED
|
@@ -1,10 +1,10 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "sparkforensics-mcp",
|
|
3
|
-
"version": "0.
|
|
3
|
+
"version": "0.3.0",
|
|
4
4
|
"mcpName": "io.github.shuffle-works/sparkforensics-mcp",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"license": "MIT",
|
|
7
|
-
"description": "MCP server exposing Apache Spark event-log diagnostics
|
|
7
|
+
"description": "MCP server exposing Apache Spark event-log diagnostics over stdio.",
|
|
8
8
|
"repository": {
|
|
9
9
|
"type": "git",
|
|
10
10
|
"url": "git+https://github.com/shuffle-works/sparkforensics.git"
|
|
@@ -23,7 +23,7 @@
|
|
|
23
23
|
"performance"
|
|
24
24
|
],
|
|
25
25
|
"bin": {
|
|
26
|
-
"sparkforensics-mcp": "
|
|
26
|
+
"sparkforensics-mcp": "bin/sparkforensics-mcp.mjs"
|
|
27
27
|
},
|
|
28
28
|
"engines": {
|
|
29
29
|
"node": ">=18"
|
|
@@ -343,8 +343,8 @@ export function createThreadedZstdDecoder(
|
|
|
343
343
|
}
|
|
344
344
|
|
|
345
345
|
// Node's native zstd where this Node has it (22.15+/23.8+); older Nodes keep the vendored fzstd.
|
|
346
|
-
// runParse/runParseFiles (collectRun) take the threaded decoder
|
|
347
|
-
//
|
|
346
|
+
// runParse/runParseFiles (collectRun) take the threaded decoder; the SHS archive loader
|
|
347
|
+
// (decodeShsArchive over an in-memory download) takes the inline one.
|
|
348
348
|
export const nodeParseCodecs =
|
|
349
349
|
nativeZstdAvailable ? { zstdDecoder: createThreadedZstdDecoder } : {};
|
|
350
350
|
export const nodeArchiveCodecs =
|
package/vendor-core/detectors.js
CHANGED
|
@@ -6,6 +6,7 @@ import { computeCoreLocalityRatio } from './core-locality-ratio.js';
|
|
|
6
6
|
import { estimateSingleStage, tailRecoveryMs, tailRemovedWorkMs, stragglerFixLongestTaskMs, } from './occupancy.js';
|
|
7
7
|
import { isExchangeNode, isBroadcastExchangeNode } from './plan-node-detail.js';
|
|
8
8
|
import { cyrb53 } from './string-hash.js';
|
|
9
|
+
import { MAX_FAILURE_GROUPS, describeTaskFailure, } from './task-failure.js';
|
|
9
10
|
|
|
10
11
|
|
|
11
12
|
const MB = 1024 * 1024;
|
|
@@ -73,6 +74,7 @@ const TB = 1024 * GB;
|
|
|
73
74
|
|
|
74
75
|
|
|
75
76
|
|
|
77
|
+
|
|
76
78
|
|
|
77
79
|
|
|
78
80
|
|
|
@@ -1206,7 +1208,7 @@ export const DETECTORS = [
|
|
|
1206
1208
|
},
|
|
1207
1209
|
},
|
|
1208
1210
|
{
|
|
1209
|
-
type: 'failures', scope: 'stage', order: 40, fixEffort: 'code', version:
|
|
1211
|
+
type: 'failures', scope: 'stage', order: 40, fixEffort: 'code', version: 2,
|
|
1210
1212
|
docAnchor: '#bottleneck-failures',
|
|
1211
1213
|
thresholds: { minTasks: 10, warnRate: 0.05, critRate: 0.20 },
|
|
1212
1214
|
detect(
|
|
@@ -1219,13 +1221,25 @@ export const DETECTORS = [
|
|
|
1219
1221
|
if (failureRate <= this.thresholds.warnRate) return null;
|
|
1220
1222
|
const value = Math.round(failureRate * 1000) / 10;
|
|
1221
1223
|
const dominantReason = pickDominantReason(stage.failureReasons);
|
|
1224
|
+
// Groups arrive most frequent first. Name the dominant error from the largest group under the
|
|
1225
|
+
// dominant tag, so the error and the tag agree even when one tag splits into many messages.
|
|
1226
|
+
const allGroups = stage.failureGroups ?? [];
|
|
1227
|
+
const dominantGroup = allGroups.find((g) => g.reason === dominantReason);
|
|
1228
|
+
const dominantError = (dominantGroup ? describeTaskFailure(dominantGroup) : null) ?? dominantReason;
|
|
1229
|
+
const failureGroups = allGroups.slice(0, MAX_FAILURE_GROUPS);
|
|
1230
|
+
const groupedTasks = failureGroups.reduce((sum, g) => sum + g.count, 0);
|
|
1222
1231
|
return {
|
|
1223
1232
|
type: 'failures', stageId: stage.id,
|
|
1224
1233
|
impactBand: failureRate > this.thresholds.critRate ? 'critical' : 'warning',
|
|
1225
1234
|
metric: 'failureRate', value,
|
|
1226
1235
|
failedTasks: stage.failedTasks,
|
|
1227
1236
|
dominantReason,
|
|
1228
|
-
|
|
1237
|
+
dominantError,
|
|
1238
|
+
// One entry per distinct error (tag, class, message, loss reason), each with one bounded
|
|
1239
|
+
// stack excerpt; otherFailedTasks counts the failed tasks no shown group covers.
|
|
1240
|
+
failureGroups,
|
|
1241
|
+
otherFailedTasks: Math.max(0, stage.failedTasks - groupedTasks),
|
|
1242
|
+
recommendation: `${value}% of tasks failed${dominantError ? ` (dominant error: ${dominantError})` : ''}: investigate driver logs for executor instability or data-driven errors.`,
|
|
1229
1243
|
};
|
|
1230
1244
|
},
|
|
1231
1245
|
},
|
|
@@ -1,5 +1,9 @@
|
|
|
1
1
|
### `FAIL`: Failed tasks {#fail}
|
|
2
2
|
|
|
3
3
|
Tasks fail often enough to affect the stage. Failed tasks point to executor
|
|
4
|
-
instability or data-driven errors
|
|
5
|
-
|
|
4
|
+
instability or data-driven errors. The finding names the dominant error: the
|
|
5
|
+
exception class, or the executor loss reason (for example "Container killed
|
|
6
|
+
by YARN for exceeding memory limits"). It lists up to five distinct failures,
|
|
7
|
+
each with its message and a short stack excerpt. With redaction on, messages
|
|
8
|
+
and the message text inside excerpts are replaced, since they can carry file
|
|
9
|
+
paths and data values; class names and stack frames stay.
|
|
@@ -19,6 +19,7 @@ import {
|
|
|
19
19
|
} from './event-schemas.js';
|
|
20
20
|
import { assertNever } from './assert-never.js';
|
|
21
21
|
import { finalizeStage } from './stage-quantiles.js';
|
|
22
|
+
import { MAX_FAILURE_DETAILS_PER_STAGE, extractTaskFailureDetail, taskFailureKey, } from './task-failure.js';
|
|
22
23
|
import { computeRunAggregates } from './run-aggregates.js';
|
|
23
24
|
|
|
24
25
|
|
|
@@ -94,6 +95,9 @@ const MAX_TASK_SAMPLES = 20;
|
|
|
94
95
|
|
|
95
96
|
|
|
96
97
|
|
|
98
|
+
|
|
99
|
+
|
|
100
|
+
|
|
97
101
|
|
|
98
102
|
|
|
99
103
|
|
|
@@ -136,6 +140,8 @@ const MAX_TASK_SAMPLES = 20;
|
|
|
136
140
|
|
|
137
141
|
|
|
138
142
|
|
|
143
|
+
|
|
144
|
+
|
|
139
145
|
|
|
140
146
|
|
|
141
147
|
|
|
@@ -485,6 +491,19 @@ function taskRecordToSample(t ) {
|
|
|
485
491
|
};
|
|
486
492
|
}
|
|
487
493
|
|
|
494
|
+
// One shared detail object per distinct failure, so a stage with thousands of failed attempts
|
|
495
|
+
// holds each bounded stack excerpt once. Past the cap an attempt keeps only its reason tag.
|
|
496
|
+
function internTaskFailure(stage , endReason ) {
|
|
497
|
+
const detail = extractTaskFailureDetail(endReason);
|
|
498
|
+
if (!detail || !stage.failureDetails) return null;
|
|
499
|
+
const key = taskFailureKey(detail);
|
|
500
|
+
const known = stage.failureDetails.get(key);
|
|
501
|
+
if (known) return known;
|
|
502
|
+
if (stage.failureDetails.size >= MAX_FAILURE_DETAILS_PER_STAGE) return null;
|
|
503
|
+
stage.failureDetails.set(key, detail);
|
|
504
|
+
return detail;
|
|
505
|
+
}
|
|
506
|
+
|
|
488
507
|
export function accumulateTask(event , state ) {
|
|
489
508
|
const stageId = event['Stage ID'];
|
|
490
509
|
const stage = state.stages.get(stageId);
|
|
@@ -521,6 +540,7 @@ export function accumulateTask(event , state
|
|
|
521
540
|
launchTime: info['Launch Time'] ?? 0,
|
|
522
541
|
finishTime: info['Finish Time'] ?? 0,
|
|
523
542
|
reason: event['Task End Reason']?.['Reason'] ?? null,
|
|
543
|
+
failure: failed ? internTaskFailure(stage, event['Task End Reason']) : null,
|
|
524
544
|
speculative: info['Speculative'] === true,
|
|
525
545
|
host: info['Host'] ?? '',
|
|
526
546
|
executorId: info['Executor ID'] ?? '',
|
|
@@ -802,6 +822,7 @@ export function submitStage(event , st
|
|
|
802
822
|
failureReasons: new Map(),
|
|
803
823
|
stageFailureReason: null,
|
|
804
824
|
taskAttempts: new Map(),
|
|
825
|
+
failureDetails: new Map(),
|
|
805
826
|
retryTaskSamples: [],
|
|
806
827
|
retryWasteMs: 0,
|
|
807
828
|
wastedAttempts: 0,
|
|
@@ -238,6 +238,14 @@ export const TaskEndEventSchema = z.object({
|
|
|
238
238
|
'Stage Attempt ID': z.number().optional(),
|
|
239
239
|
'Task End Reason': z.object({
|
|
240
240
|
Reason: z.string().optional(),
|
|
241
|
+
// The error behind a failed attempt, read by extractTaskFailureDetail (task-failure.ts), which
|
|
242
|
+
// keeps only string values: unknown here so an unexpected shape drops the detail, never the task.
|
|
243
|
+
'Class Name': z.unknown().optional(),
|
|
244
|
+
Description: z.unknown().optional(),
|
|
245
|
+
'Full Stack Trace': z.unknown().optional(),
|
|
246
|
+
'Loss Reason': z.unknown().optional(),
|
|
247
|
+
Message: z.unknown().optional(),
|
|
248
|
+
'Kill Reason': z.unknown().optional(),
|
|
241
249
|
}).optional(),
|
|
242
250
|
'Task Info': z.object({
|
|
243
251
|
'Launch Time': z.number().optional(),
|
|
@@ -6,6 +6,7 @@ import { detectorCatalog } from './detectors.js';
|
|
|
6
6
|
import { typeTag, formatBytes, formatDuration, IMPACT_BAND_ORDER } from './format-utils.js';
|
|
7
7
|
import { FINDING_NAMES, titleCase } from './finding-names.js';
|
|
8
8
|
import { redactReport } from './redact.js';
|
|
9
|
+
import { formatTaskFailureHeadline, } from './task-failure.js';
|
|
9
10
|
import { coreFindingActionLabel } from './finding-action-label.js';
|
|
10
11
|
import { matchesFindingFilterCriteria } from './finding-filter-predicate.js';
|
|
11
12
|
import { buildRecommendationRollup, isEligible, rankFindings, } from './recommendation-rollup.js';
|
|
@@ -269,6 +270,21 @@ function renderEvidenceValue(key , value ) {
|
|
|
269
270
|
return String(value);
|
|
270
271
|
}
|
|
271
272
|
|
|
273
|
+
// The `failures` finding's distinct errors: a headline per group, its stack excerpt as an indented
|
|
274
|
+
// code block (indented, not fenced, so no excerpt content can close it early).
|
|
275
|
+
function renderFailureGroups(groups ) {
|
|
276
|
+
const lines = [` - failureGroups: ${groups.length}`];
|
|
277
|
+
for (const g of groups) {
|
|
278
|
+
lines.push(` - ${g.count} task(s): ${formatTaskFailureHeadline(g)}`);
|
|
279
|
+
if (g.stackExcerpt) {
|
|
280
|
+
lines.push('');
|
|
281
|
+
for (const l of g.stackExcerpt.split('\n')) lines.push(` ${l}`);
|
|
282
|
+
lines.push('');
|
|
283
|
+
}
|
|
284
|
+
}
|
|
285
|
+
return lines;
|
|
286
|
+
}
|
|
287
|
+
|
|
272
288
|
function formatWallClockRange(low , high ) {
|
|
273
289
|
const fmtMs = (ms ) => (ms === 0 ? '0s' : formatDuration(ms));
|
|
274
290
|
return low === high ? `Estimated ${fmtMs(high)}` : `Estimated ${fmtMs(low)}-${fmtMs(high)}`;
|
|
@@ -336,7 +352,10 @@ function renderMarkdown(json ) {
|
|
|
336
352
|
const evidence = Object.entries(r.evidence ?? {}).sort(([a], [b]) => (a < b ? -1 : a > b ? 1 : 0));
|
|
337
353
|
if (evidence.length) {
|
|
338
354
|
lines.push('- evidence:');
|
|
339
|
-
for (const [k, v] of evidence)
|
|
355
|
+
for (const [k, v] of evidence) {
|
|
356
|
+
if (k === 'failureGroups' && Array.isArray(v)) lines.push(...renderFailureGroups(v ));
|
|
357
|
+
else lines.push(` - ${k}: ${renderEvidenceValue(k, v)}`);
|
|
358
|
+
}
|
|
340
359
|
}
|
|
341
360
|
lines.push('');
|
|
342
361
|
}
|
|
@@ -1,4 +1,5 @@
|
|
|
1
1
|
|
|
2
|
+
import { nsToMs } from './format-utils.js';
|
|
2
3
|
import {
|
|
3
4
|
computeOccupancy, estimateSingleStage, estimateMultiStage, tailRecoveryMs, tailRemovedWorkMs, stragglerFixLongestTaskMs,
|
|
4
5
|
|
|
@@ -82,6 +83,24 @@ const EXECUTOR_STARTUP_OVERHEAD_MS = 15000;
|
|
|
82
83
|
// Assumed re-read throughput, shared by cachingOpportunity and cacheUtilization.
|
|
83
84
|
const RE_READ_THROUGHPUT_BPS = 125_000_000;
|
|
84
85
|
|
|
86
|
+
// Below this share of executorRunTime spent on CPU, a stage's tasks were idle, waiting on something
|
|
87
|
+
// outside Spark: on 14 real logs (2026-09-23) every non-Python stage under 1% was a JDBC read, a
|
|
88
|
+
// file listing or a Delta log read, while file writes, which more partitions do parallelize,
|
|
89
|
+
// start at 2%.
|
|
90
|
+
const IDLE_CPU_SHARE_MAX = 0.01;
|
|
91
|
+
|
|
92
|
+
// True when the stage's tasks spent under IDLE_CPU_SHARE_MAX of their run time on CPU. False when
|
|
93
|
+
// the share can't be trusted: no CPU time recorded (older Spark logs omit the metric), or Python
|
|
94
|
+
// code run through PythonRDD, whose worker-process CPU executorCpuTime (the JVM task thread's)
|
|
95
|
+
// never counts (such stages read 0.1% on the same logs while computing).
|
|
96
|
+
function tasksMostlyIdle(stage ) {
|
|
97
|
+
const runMs = stage.executorRunTime ?? 0;
|
|
98
|
+
const cpuMs = nsToMs(stage.executorCpuTime ?? 0);
|
|
99
|
+
if (runMs <= 0 || cpuMs <= 0) return false;
|
|
100
|
+
if (/PythonRDD/.test(stage.name ?? '') || /org\.apache\.spark\.api\.python\./.test(stage.details ?? '')) return false;
|
|
101
|
+
return cpuMs / runMs < IDLE_CPU_SHARE_MAX;
|
|
102
|
+
}
|
|
103
|
+
|
|
85
104
|
// skew and straggler claim time off the stage's longest task itself, so the occupancy clip must
|
|
86
105
|
// not floor them at that same task (see estimateSingleStage). detectors.ts's clippedWasteMs gates
|
|
87
106
|
// both detectors on the same option so the firing floor and the displayed estimate agree.
|
|
@@ -332,14 +351,15 @@ function computeEstimateForFinding(
|
|
|
332
351
|
// is queueing no partition count recovers. Splitting partitions splits the longest task
|
|
333
352
|
// too, hence TAIL_CLAIM's post-fix floor. Unknown cluster size: no defensible figure.
|
|
334
353
|
if (totalCores <= 0) return costOnly('modeled');
|
|
335
|
-
// A stage that read no input and no shuffle
|
|
336
|
-
// 1-task count stage open 27 minutes on 5s of CPU
|
|
354
|
+
// A stage that read no input and no shuffle, its tasks idle waiting on an external system,
|
|
355
|
+
// gains nothing from more partitions (a 1-task JDBC count stage open 27 minutes on 5s of CPU
|
|
356
|
+
// was claimed 99% recoverable): claim 0. Stages that read bytes keep their claim.
|
|
337
357
|
const readBytes = (stage.inputBytes ?? 0) + (stage.shuffleReadBytes ?? 0);
|
|
338
358
|
const activeMs = typeof stage.taskActiveMs === 'number'
|
|
339
359
|
? stage.taskActiveMs
|
|
340
360
|
: Math.max(0, (stage.completedAt ?? 0) - (stage.submittedAt ?? 0));
|
|
341
361
|
const taskCount = stage.taskCount ?? 0;
|
|
342
|
-
const wasteMs = readBytes
|
|
362
|
+
const wasteMs = readBytes <= 0 && tasksMostlyIdle(stage) ? 0 : activeMs * Math.max(0, 1 - taskCount / totalCores);
|
|
343
363
|
return singleStageImpact(wasteMs, finding.stageId, stages, occupancy, 'modeled', { value: wasteMs, unit: 'ms' }, TAIL_CLAIM);
|
|
344
364
|
}
|
|
345
365
|
case 'partitionSizing': {
|
package/vendor-core/occupancy.js
CHANGED
|
@@ -152,18 +152,19 @@ const SERIAL_GATE_THRESHOLD = 0.999;
|
|
|
152
152
|
|
|
153
153
|
|
|
154
154
|
|
|
155
|
+
|
|
155
156
|
|
|
156
157
|
|
|
157
158
|
|
|
158
159
|
|
|
159
|
-
// Wall-clock a skew/straggler fix recovers
|
|
160
|
-
//
|
|
161
|
-
//
|
|
162
|
-
//
|
|
163
|
-
//
|
|
164
|
-
//
|
|
165
|
-
// peak on one real stage), which doubled the claim.
|
|
160
|
+
// Wall-clock a skew/straggler fix recovers: finalizeStage's task-level replay
|
|
161
|
+
// (tailReplayRecoveryMs, computeTailReplayRecoveryMs) when the stage carries it. Without it (a
|
|
162
|
+
// stage built by hand, as in detector tests) an estimate from the slowest task's excess over the
|
|
163
|
+
// median: a lone straggler costs that excess; a tail of many slow tasks costs its summed excess
|
|
164
|
+
// (stragglerExcessMs) spread over the slots the stage had (peakConcurrentTasks), and the replay
|
|
165
|
+
// recovers about the larger of the two.
|
|
166
166
|
export function tailRecoveryMs(stage , singleTaskExcessMs ) {
|
|
167
|
+
if (stage.tailReplayRecoveryMs != null) return stage.tailReplayRecoveryMs;
|
|
167
168
|
const excessMs = stage.stragglerExcessMs ?? 0;
|
|
168
169
|
const slots = stage.peakConcurrentTasks ?? 0;
|
|
169
170
|
if (excessMs <= 0 || slots <= 0) return singleTaskExcessMs;
|
|
@@ -4,7 +4,9 @@ import { Decompress as ZstdDecompress } from './vendor/fzstd.js';
|
|
|
4
4
|
import { createSnappyBlockDecoder } from './snappy-block.js';
|
|
5
5
|
import { createState, dispatchLine, buildChunkDecoder, emitParseCompletion, } from './event-handlers.js';
|
|
6
6
|
import { TASK_FIELD_NAMES } from './stage-quantiles.js';
|
|
7
|
-
import { runParseFromUrl, sniffCodec } from './shs-fetch.js';
|
|
7
|
+
import { runParseFromUrl, sniffCodec, parseZipArchive } from './shs-fetch.js';
|
|
8
|
+
import { isZip } from './zip-archive.js';
|
|
9
|
+
import { createWorkerZstdDecoders } from './zstd-worker-client.js';
|
|
8
10
|
|
|
9
11
|
export {
|
|
10
12
|
buildChunkDecoder, createState, normalizeSparkProperties, parseSparkMemoryMB, extractResources,
|
|
@@ -34,6 +36,8 @@ const MIN_PROGRESS_STEPS = 100;
|
|
|
34
36
|
const PROGRESS_EMIT_LINES = 300;
|
|
35
37
|
|
|
36
38
|
|
|
39
|
+
const NOT_AN_EVENT_LOG = 'Not a Spark event log: no application-start event found. Choose a Spark event log file, or check the docs for supported formats.';
|
|
40
|
+
|
|
37
41
|
// zstdDecoder replaces the vendored fzstd for zstd input; the Node CLI/MCP path passes
|
|
38
42
|
// cli/native-zstd.ts's native-zlib decoder, which a browser bundle can't import.
|
|
39
43
|
|
|
@@ -61,9 +65,11 @@ const PROGRESS_EMIT_LINES = 300;
|
|
|
61
65
|
// without touching the vendored files (mirrors shs-fetch.ts's shim).
|
|
62
66
|
|
|
63
67
|
|
|
64
|
-
// A Node decoder
|
|
68
|
+
// A Node decoder, or the browser's decompress worker (zstd-worker-client.ts), may decompress off
|
|
69
|
+
// the calling thread: streamFile awaits each push, and calls cancel() when it abandons the stream.
|
|
65
70
|
|
|
66
71
|
|
|
72
|
+
|
|
67
73
|
|
|
68
74
|
|
|
69
75
|
// Stream one File's (possibly compressed) bytes through the codec dispatch,
|
|
@@ -89,7 +95,7 @@ export async function streamFile(
|
|
|
89
95
|
const gunzip = codec === 'gz' ? new (Gunzip )((inflated) => onChunk(inflated, currentPct)) : null;
|
|
90
96
|
const lz4 = codec === 'lz4' ? createLz4BlockDecoder((inflated) => onChunk(inflated, currentPct)) : null;
|
|
91
97
|
const onZstdChunk = (inflated ) => onChunk(inflated, currentPct);
|
|
92
|
-
const zstd
|
|
98
|
+
const zstd = codec !== 'zstd' ? null
|
|
93
99
|
: zstdDecoder ? zstdDecoder(onZstdChunk)
|
|
94
100
|
: new (ZstdDecompress )(onZstdChunk);
|
|
95
101
|
const snappy = codec === 'snappy' ? createSnappyBlockDecoder((inflated) => onChunk(inflated, currentPct)) : null;
|
|
@@ -101,17 +107,22 @@ export async function streamFile(
|
|
|
101
107
|
const stepSize = Math.max(1, Math.min(chunkSize, Math.ceil(file.size / MIN_PROGRESS_STEPS)));
|
|
102
108
|
|
|
103
109
|
let offset = 0;
|
|
104
|
-
|
|
105
|
-
|
|
106
|
-
|
|
107
|
-
|
|
108
|
-
|
|
109
|
-
|
|
110
|
-
|
|
111
|
-
|
|
112
|
-
|
|
113
|
-
|
|
114
|
-
|
|
110
|
+
try {
|
|
111
|
+
while (offset < file.size) {
|
|
112
|
+
const start = offset;
|
|
113
|
+
const slice = new Uint8Array(await file.slice(start, start + stepSize).arrayBuffer());
|
|
114
|
+
offset += stepSize;
|
|
115
|
+
const final = offset >= file.size;
|
|
116
|
+
currentPct = start / file.size;
|
|
117
|
+
if (gunzip) gunzip.push(slice, final);
|
|
118
|
+
else if (lz4) lz4.push(slice);
|
|
119
|
+
else if (zstd) await zstd.push(slice, final);
|
|
120
|
+
else if (snappy) snappy.push(slice);
|
|
121
|
+
else onChunk(slice, currentPct);
|
|
122
|
+
}
|
|
123
|
+
} catch (e) {
|
|
124
|
+
zstd?.cancel?.();
|
|
125
|
+
throw e;
|
|
115
126
|
}
|
|
116
127
|
if (lz4) lz4.end();
|
|
117
128
|
if (snappy) snappy.end();
|
|
@@ -132,6 +143,20 @@ export async function runParse(
|
|
|
132
143
|
return;
|
|
133
144
|
}
|
|
134
145
|
|
|
146
|
+
// A Spark History Server download (the UI's download link, or GET
|
|
147
|
+
// /api/v1/applications/<id>/logs) is a zip holding the log file, or a
|
|
148
|
+
// rolling log's parts: unwrap it through the same path the SHS fetch uses.
|
|
149
|
+
if (isZip(new Uint8Array(await file.slice(0, Math.min(4, file.size)).arrayBuffer()))) {
|
|
150
|
+
await parseZipArchive(file, state, emit, {
|
|
151
|
+
zstdDecoder,
|
|
152
|
+
chunkSize,
|
|
153
|
+
progressEvery: PROGRESS_EMIT_LINES,
|
|
154
|
+
reportPct: true,
|
|
155
|
+
onInvalid: (detail) => emit({ type: 'error', message: detail ?? NOT_AN_EVENT_LOG }),
|
|
156
|
+
});
|
|
157
|
+
return;
|
|
158
|
+
}
|
|
159
|
+
|
|
135
160
|
const decoder = buildChunkDecoder();
|
|
136
161
|
const joined = [];
|
|
137
162
|
let linesProcessed = 0;
|
|
@@ -160,7 +185,7 @@ export async function runParse(
|
|
|
160
185
|
}
|
|
161
186
|
|
|
162
187
|
if (!state.app) {
|
|
163
|
-
emit({ type: 'error', message:
|
|
188
|
+
emit({ type: 'error', message: NOT_AN_EVENT_LOG });
|
|
164
189
|
return;
|
|
165
190
|
}
|
|
166
191
|
|
|
@@ -219,7 +244,7 @@ export async function runParseFiles(
|
|
|
219
244
|
}
|
|
220
245
|
|
|
221
246
|
if (!state.app) {
|
|
222
|
-
emit({ type: 'error', message:
|
|
247
|
+
emit({ type: 'error', message: NOT_AN_EVENT_LOG });
|
|
223
248
|
return;
|
|
224
249
|
}
|
|
225
250
|
|
|
@@ -232,17 +257,25 @@ const isWorker = typeof WorkerGlobalScope !== 'undefined' && self instanceof Wor
|
|
|
232
257
|
|
|
233
258
|
if (isWorker) {
|
|
234
259
|
let workerState = null;
|
|
260
|
+
// Dropped zstd files decompress in a second worker, overlapping with parsing here. The
|
|
261
|
+
// `new Worker(new URL(...))` stays inline for Vite's worker detection (see ingest.ts). The SHS
|
|
262
|
+
// path (runParseFromUrl) keeps in-thread fzstd.
|
|
263
|
+
const zstdDecoder = createWorkerZstdDecoders(
|
|
264
|
+
() => new Worker(new URL('./zstd-worker.js', import.meta.url), { type: 'module' }),
|
|
265
|
+
(onChunk) => new (ZstdDecompress )(onChunk),
|
|
266
|
+
{ onFallback: (reason) => console.warn(`zstd: decompressing on the parse worker: ${reason}`) },
|
|
267
|
+
);
|
|
235
268
|
|
|
236
269
|
self.onmessage = async ({ data } ) => {
|
|
237
270
|
if (data.type === 'parse') {
|
|
238
271
|
workerState = createState();
|
|
239
|
-
await runParse(data.file, workerState);
|
|
272
|
+
await runParse(data.file, workerState, { zstdDecoder });
|
|
240
273
|
} else if (data.type === 'parseFromUrl') {
|
|
241
274
|
workerState = createState();
|
|
242
275
|
await runParseFromUrl(data.request, workerState);
|
|
243
276
|
} else if (data.type === 'parseFiles') {
|
|
244
277
|
workerState = createState();
|
|
245
|
-
await runParseFiles(data.files, workerState);
|
|
278
|
+
await runParseFiles(data.files, workerState, { zstdDecoder });
|
|
246
279
|
} else if (data.type === 'getTaskData') {
|
|
247
280
|
const { stageId, reqId } = data;
|
|
248
281
|
const stored = workerState?.taskStore.get(stageId) ?? new Float64Array(0);
|
|
@@ -45,7 +45,7 @@ function pushLongFilterWarning(result , len ) {
|
|
|
45
45
|
}
|
|
46
46
|
|
|
47
47
|
// Stable relation-identity key for a scan node: "<format>:<relation>" (e.g.
|
|
48
|
-
// "delta:mx.
|
|
48
|
+
// "delta:mx.store_map", "parquet:warehouse.sales", "jdbc:dw.dim_product")
|
|
49
49
|
// or null when the node is internal Delta metadata / a non-scan / un-nameable.
|
|
50
50
|
// The single source of truth for scan identity, shared by `visitScan` (summary)
|
|
51
51
|
// and the `cachingOpportunity` detector so the regexes live in one place.
|
package/vendor-core/redact.js
CHANGED
|
@@ -8,6 +8,7 @@
|
|
|
8
8
|
// and non-mutating (returns a fresh, deep-copied tree).
|
|
9
9
|
|
|
10
10
|
|
|
11
|
+
import { redactTaskFailureGroup, } from './task-failure.js';
|
|
11
12
|
|
|
12
13
|
// Host / IP identifier patterns. Used to enumerate host names that surface only
|
|
13
14
|
// inside free text: recommendation strings, `stageFailed`'s failure-reason
|
|
@@ -79,6 +80,25 @@ function collectHostFields(node , hosts ) {
|
|
|
79
80
|
}
|
|
80
81
|
}
|
|
81
82
|
|
|
83
|
+
// A failed-task error's message and stack text can carry file paths and data values that no
|
|
84
|
+
// host/app-id pattern recognizes, so they are dropped rather than pseudonymized: every array under
|
|
85
|
+
// a key named `failureGroups` (a `failures` finding's, or a stage's in the HTML export) gets
|
|
86
|
+
// redactTaskFailureGroup applied. Walking by key name, like collectHostFields, needs no path list.
|
|
87
|
+
// Returns a fresh tree.
|
|
88
|
+
function redactFailureGroups (node ) {
|
|
89
|
+
if (Array.isArray(node)) return node.map((n) => redactFailureGroups(n)) ;
|
|
90
|
+
if (node && typeof node === 'object') {
|
|
91
|
+
const out = {};
|
|
92
|
+
for (const [k, v] of Object.entries(node)) {
|
|
93
|
+
out[k] = k === 'failureGroups' && Array.isArray(v)
|
|
94
|
+
? v.map((g) => (g && typeof g === 'object' ? redactTaskFailureGroup(g ) : g))
|
|
95
|
+
: redactFailureGroups(v);
|
|
96
|
+
}
|
|
97
|
+
return out ;
|
|
98
|
+
}
|
|
99
|
+
return node;
|
|
100
|
+
}
|
|
101
|
+
|
|
82
102
|
// Spark config keys ending in `host`/`hostname` (e.g. spark.driver.host,
|
|
83
103
|
// spark.yarn.am.hostname) carry plain FQDN host names that neither
|
|
84
104
|
// HOST_PATTERNS matches (no IP/EC2 shape) nor collectHostFields's by-key-name
|
|
@@ -172,7 +192,8 @@ function applyReplacements (node , ids
|
|
|
172
192
|
return deepReplace(node, merged) ;
|
|
173
193
|
}
|
|
174
194
|
|
|
175
|
-
export function redactReport (
|
|
195
|
+
export function redactReport (input ) {
|
|
196
|
+
const report = redactFailureGroups(input);
|
|
176
197
|
const { appIds, hosts } = collectIds(report);
|
|
177
198
|
return applyReplacements(report, { appIds, hosts });
|
|
178
199
|
}
|
|
@@ -216,7 +237,8 @@ export function redactComparison (comparison ) {
|
|
|
216
237
|
// this also walks executors.added/removed for their literal `host` field
|
|
217
238
|
// (ExecutorAddedEvent.host), since raw executor records: not just findings
|
|
218
239
|
//: reach data.js.
|
|
219
|
-
export function redactExportData(
|
|
240
|
+
export function redactExportData(input ) {
|
|
241
|
+
const data = redactFailureGroups(input);
|
|
220
242
|
const appIds = new Set ();
|
|
221
243
|
const hosts = new Set ();
|
|
222
244
|
const appId = data.app?.id;
|
|
@@ -180,6 +180,14 @@ function skewRatios(stages ) {
|
|
|
180
180
|
return out;
|
|
181
181
|
}
|
|
182
182
|
|
|
183
|
+
// Every key metricDeltas() emits, in emission order. The CLI validates
|
|
184
|
+
// --regression-metric against it, so a misspelled key is a usage error rather
|
|
185
|
+
// than an inconclusive budget. A contract test keeps it in sync.
|
|
186
|
+
export const COMPARISON_METRIC_KEYS = [
|
|
187
|
+
'wallClock', 'shuffleSpill', 'taskSkew', 'failedTaskRate', 'diskSpill', 'gcTime',
|
|
188
|
+
'inputBytes', 'outputBytes', 'executorRunTime', 'taskCount', 'executorsAdded',
|
|
189
|
+
];
|
|
190
|
+
|
|
183
191
|
// Volume/count metrics, not cost metrics: more or less input/output data, or
|
|
184
192
|
// tasks/executors, isn't inherently better or worse (it may just reflect a
|
|
185
193
|
// differently-sized job), unlike wall-clock, spill, GC, etc. Exported as the
|