sparkforensics-mcp 0.2.4 → 0.4.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -0
- package/README.md +7 -1
- package/bin/sparkforensics-mcp.mjs +41 -10
- package/package.json +1 -1
- package/vendor-core/analyzer.js +156 -48
- package/vendor-core/check-coverage.js +88 -0
- package/vendor-core/cli/budgets.js +31 -18
- package/vendor-core/cli/collect-run.js +76 -31
- package/vendor-core/cli/native-zstd.js +2 -2
- package/vendor-core/cli/threshold-config.js +28 -0
- package/vendor-core/comparison-verdict.js +177 -0
- package/vendor-core/core-source-hash.txt +1 -0
- package/vendor-core/core-usage-locality.js +56 -2
- package/vendor-core/detector-docs.js +58 -0
- package/vendor-core/detectors.js +933 -459
- package/vendor-core/docs-config.js +0 -36
- package/vendor-core/docs-content/chapters/nav-index.json +31 -0
- package/vendor-core/docs-content/detection/cstor.md +9 -0
- package/vendor-core/docs-content/detection/fail.md +6 -2
- package/vendor-core/docs-site-config.js +3 -0
- package/vendor-core/event-handlers.js +191 -6
- package/vendor-core/event-schemas.js +29 -0
- package/vendor-core/evidence-report.js +440 -112
- package/vendor-core/export-data.js +79 -6
- package/vendor-core/finding-action-label.js +9 -88
- package/vendor-core/finding-filter-predicate.js +9 -0
- package/vendor-core/finding-generic-recommendation.js +6 -104
- package/vendor-core/finding-names.js +21 -45
- package/vendor-core/finding-presentation.js +333 -0
- package/vendor-core/finding-tag-help.js +110 -0
- package/vendor-core/finding-types.js +361 -0
- package/vendor-core/findings-of-type.js +11 -0
- package/vendor-core/format-utils.js +92 -27
- package/vendor-core/html-export.js +51 -0
- package/vendor-core/impact-band.js +21 -8
- package/vendor-core/impact-estimator.js +8 -521
- package/vendor-core/impact-format.js +114 -0
- package/vendor-core/impact-model.js +175 -0
- package/vendor-core/ingest.js +2 -0
- package/vendor-core/intervals.js +13 -0
- package/vendor-core/list-runs.js +2 -3
- package/vendor-core/load-vendored.js +70 -5
- package/vendor-core/mcp-server-factory.js +14 -10
- package/vendor-core/mcp-tools.js +105 -45
- package/vendor-core/model-assembler.js +12 -0
- package/vendor-core/occupancy.js +1 -1
- package/vendor-core/parser-worker.js +22 -5
- package/vendor-core/plan-graph-model.js +3 -2
- package/vendor-core/plan-node-detail.js +1 -1
- package/vendor-core/recommendation-rollup.js +63 -3
- package/vendor-core/redact.js +68 -28
- package/vendor-core/run-comparison.js +40 -7
- package/vendor-core/run-interpretation.js +290 -0
- package/vendor-core/run-outcome.js +74 -0
- package/vendor-core/run-payload.js +17 -0
- package/vendor-core/run-shape.js +40 -0
- package/vendor-core/run-verdict.js +353 -0
- package/vendor-core/scaling-sim.js +4 -5
- package/vendor-core/scorecard-estimates.js +62 -0
- package/vendor-core/shs-fetch.js +175 -65
- package/vendor-core/shs-load.js +1 -1
- package/vendor-core/sql-stages.js +11 -0
- package/vendor-core/stage-quantiles.js +14 -0
- package/vendor-core/task-failure.js +151 -0
- package/vendor-core/threshold-overrides.js +160 -0
- package/vendor-core/threshold-summary.js +11 -33
- package/vendor-core/types.js +6 -42
- package/vendor-core/vendor/fflate.js +1 -1
- package/vendor-core/wall-clock.js +1 -12
- package/vendor-core/wasted-core-hours.js +2 -2
- package/vendor-core/zip-archive.js +167 -0
package/vendor-core/shs-fetch.js
CHANGED
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import {
|
|
1
|
+
import { Gunzip } from './vendor/fflate.js';
|
|
2
2
|
import { createLz4BlockDecoder } from './lz4-block.js';
|
|
3
3
|
import { Decompress as ZstdDecompress } from './vendor/fzstd.js';
|
|
4
4
|
import { createSnappyBlockDecoder } from './snappy-block.js';
|
|
@@ -6,6 +6,8 @@ import { buildProxyRequestUrl, isShsErrorCode } from './shs-request.js';
|
|
|
6
6
|
import { dispatchLine, buildChunkDecoder, emitParseCompletion, } from './event-handlers.js';
|
|
7
7
|
import { ShsProxyErrorBodySchema } from './shs-schemas.js';
|
|
8
8
|
import { naturalCompare, reassembleRollingEntries } from './rolling-log-reassembly.js';
|
|
9
|
+
import { listZipEntries, streamZipEntry, } from './zip-archive.js';
|
|
10
|
+
|
|
9
11
|
|
|
10
12
|
export { naturalCompare, reassembleRollingEntries };
|
|
11
13
|
|
|
@@ -24,48 +26,84 @@ export function sniffCodec(bytes )
|
|
|
24
26
|
return null;
|
|
25
27
|
}
|
|
26
28
|
|
|
27
|
-
//
|
|
28
|
-
//
|
|
29
|
-
// whole decompressed entry as one buffer. A giant
|
|
30
|
-
// (or gzip member) declares its full content size in
|
|
31
|
-
// it (fzstd's `decompress()`, fflate's `gunzipSync()`)
|
|
32
|
-
// size as a single ArrayBuffer up front, which can exceed
|
|
33
|
-
// will allocate for a multi-GB event log ("Array buffer
|
|
29
|
+
// Decodes one zip entry's (possibly compressed) bytes as they stream out of
|
|
30
|
+
// the archive, block by block, invoking `onChunk` per decompressed piece;
|
|
31
|
+
// never materializes the whole decompressed entry as one buffer. A giant
|
|
32
|
+
// single-segment zstd frame (or gzip member) declares its full content size in
|
|
33
|
+
// the header; one-shotting it (fzstd's `decompress()`, fflate's `gunzipSync()`)
|
|
34
|
+
// allocates that entire size as a single ArrayBuffer up front, which can exceed
|
|
35
|
+
// what the browser will allocate for a multi-GB event log ("Array buffer
|
|
36
|
+
// allocation failed"). The codec is sniffed from the entry's first 8 bytes,
|
|
37
|
+
// falling back to the entry name's suffix.
|
|
34
38
|
// fflate's Gunzip and fzstd's Decompress are untyped vendor JS, so TS can't
|
|
35
39
|
// infer a construct signature for them; this local shape types the call sites
|
|
36
40
|
// without touching the vendored files.
|
|
37
41
|
|
|
38
42
|
|
|
39
|
-
|
|
40
|
-
|
|
41
|
-
|
|
42
|
-
|
|
43
|
-
|
|
44
|
-
function decodeEntry(
|
|
45
|
-
name , raw , onChunk , zstdDecoder ,
|
|
46
|
-
) {
|
|
47
|
-
const codec = sniffCodec(raw);
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
function openCodecSink(
|
|
46
|
+
codec , name , onChunk , zstdDecoder ,
|
|
47
|
+
) {
|
|
48
48
|
if (codec === 'lz4' || name.endsWith('.lz4')) {
|
|
49
49
|
const lz4 = createLz4BlockDecoder(onChunk);
|
|
50
|
-
lz4.push(
|
|
51
|
-
|
|
52
|
-
|
|
53
|
-
new (Gunzip )(onChunk)
|
|
54
|
-
|
|
55
|
-
|
|
56
|
-
|
|
50
|
+
return { push(chunk, final) { lz4.push(chunk); if (final) lz4.end(); } };
|
|
51
|
+
}
|
|
52
|
+
if (codec === 'gz' || name.endsWith('.gz')) {
|
|
53
|
+
const gunzip = new (Gunzip )(onChunk);
|
|
54
|
+
return { push: (chunk, final) => gunzip.push(chunk, final) };
|
|
55
|
+
}
|
|
56
|
+
if (codec === 'zstd' || name.endsWith('.zstd') || name.endsWith('.zst')) {
|
|
57
|
+
const zstd = zstdDecoder ? zstdDecoder(onChunk) : new (ZstdDecompress )(onChunk);
|
|
58
|
+
return { push: (chunk, final) => zstd.push(chunk, final), cancel: () => (zstd ).cancel?.() };
|
|
59
|
+
}
|
|
60
|
+
if (codec === 'snappy' || name.endsWith('.snappy')) {
|
|
57
61
|
const snappy = createSnappyBlockDecoder(onChunk);
|
|
58
|
-
snappy.push(
|
|
59
|
-
snappy.end();
|
|
60
|
-
} else {
|
|
61
|
-
onChunk(raw);
|
|
62
|
+
return { push(chunk, final) { snappy.push(chunk); if (final) snappy.end(); } };
|
|
62
63
|
}
|
|
64
|
+
return { push: (chunk) => onChunk(chunk) };
|
|
65
|
+
}
|
|
66
|
+
|
|
67
|
+
// Wraps openCodecSink for a stream of unknown length: holds back the latest
|
|
68
|
+
// chunk so the last one can be pushed with `final` set (codecs reject a
|
|
69
|
+
// trailing empty push less uniformly than a real last chunk), and buffers the
|
|
70
|
+
// start of the entry until it has the 8 bytes sniffCodec needs.
|
|
71
|
+
function createEntryDecoder(name , onChunk , zstdDecoder ) {
|
|
72
|
+
let head = new Uint8Array(0);
|
|
73
|
+
let held = null;
|
|
74
|
+
let sink = null;
|
|
75
|
+
const open = () => {
|
|
76
|
+
sink = openCodecSink(sniffCodec(head), name, onChunk, zstdDecoder);
|
|
77
|
+
held = head;
|
|
78
|
+
};
|
|
79
|
+
return {
|
|
80
|
+
async push(chunk ) {
|
|
81
|
+
if (!sink) {
|
|
82
|
+
const joined = new Uint8Array(head.length + chunk.length);
|
|
83
|
+
joined.set(head);
|
|
84
|
+
joined.set(chunk, head.length);
|
|
85
|
+
head = joined;
|
|
86
|
+
if (head.length >= 8) open();
|
|
87
|
+
return;
|
|
88
|
+
}
|
|
89
|
+
await sink.push(held , false);
|
|
90
|
+
held = chunk;
|
|
91
|
+
},
|
|
92
|
+
async end() {
|
|
93
|
+
if (!sink) {
|
|
94
|
+
if (!head.length) return;
|
|
95
|
+
open();
|
|
96
|
+
}
|
|
97
|
+
await sink .push(held , true);
|
|
98
|
+
},
|
|
99
|
+
cancel() { sink?.cancel?.(); },
|
|
100
|
+
};
|
|
63
101
|
}
|
|
64
102
|
|
|
65
103
|
|
|
66
104
|
|
|
67
|
-
function emitShsError(emit , code ) {
|
|
68
|
-
emit({ type: 'error', source: 'shs', code });
|
|
105
|
+
function emitShsError(emit , code , message ) {
|
|
106
|
+
emit({ type: 'error', source: 'shs', code, ...(message ? { message } : {}) });
|
|
69
107
|
}
|
|
70
108
|
|
|
71
109
|
export async function runParseFromUrl(
|
|
@@ -141,55 +179,127 @@ export async function runParseFromUrl(
|
|
|
141
179
|
if (total) {
|
|
142
180
|
emit({ type: 'progress', pct: 0.5, linesProcessed: 0 });
|
|
143
181
|
}
|
|
144
|
-
decodeShsArchive(zipBytes, state, emit);
|
|
182
|
+
await decodeShsArchive(zipBytes, state, emit);
|
|
183
|
+
}
|
|
184
|
+
|
|
185
|
+
// Reads the in-memory SHS download through the same random-access interface a
|
|
186
|
+
// dropped File offers, so both paths share parseZipArchive.
|
|
187
|
+
function bytesZipSource(bytes ) {
|
|
188
|
+
return {
|
|
189
|
+
size: bytes.length,
|
|
190
|
+
slice: (start, end) => ({ arrayBuffer: async () => bytes.slice(start, end).buffer }),
|
|
191
|
+
};
|
|
145
192
|
}
|
|
146
193
|
|
|
147
194
|
export function decodeShsArchive(
|
|
148
195
|
zipBytes , state , emit , { zstdDecoder } = {},
|
|
149
|
-
)
|
|
150
|
-
|
|
196
|
+
) {
|
|
197
|
+
return parseZipArchive(bytesZipSource(zipBytes), state, emit, {
|
|
198
|
+
zstdDecoder,
|
|
199
|
+
onInvalid: (detail) => emitShsError(emit, 'invalid-event-log', detail ?? undefined),
|
|
200
|
+
});
|
|
201
|
+
}
|
|
202
|
+
|
|
203
|
+
const baseName = (e ) => e.name.slice(e.name.lastIndexOf('/') + 1);
|
|
204
|
+
const isRollingPart = (e ) => /^events_\d+_/.test(baseName(e));
|
|
205
|
+
|
|
206
|
+
// The application attempts a History Server zip holds. Without an attempt ID
|
|
207
|
+
// the History Server packs every attempt into one download: each single-file
|
|
208
|
+
// log is its own root entry, each rolling log its own `eventlog_v2_` directory.
|
|
209
|
+
// Directory entries and `appstatus` markers are not attempts.
|
|
210
|
+
function countLogAttempts(entries ) {
|
|
211
|
+
const attempts = new Set ();
|
|
212
|
+
for (const e of entries) {
|
|
213
|
+
if (e.name.endsWith('/') || baseName(e).toLowerCase().startsWith('appstatus')) continue;
|
|
214
|
+
const dir = e.name.slice(0, e.name.indexOf('/') + 1);
|
|
215
|
+
attempts.add(dir || (isRollingPart(e) ? '' : e.name));
|
|
216
|
+
}
|
|
217
|
+
return attempts.size;
|
|
218
|
+
}
|
|
219
|
+
|
|
220
|
+
// The event-log entries a single-attempt History Server zip holds, in parse
|
|
221
|
+
// order. A rolling log's parts sit under an `eventlog_v2_<appId>/` directory
|
|
222
|
+
// entry; they are matched and reassembled by base name. A single-file log is
|
|
223
|
+
// one entry at the root. Directory entries and the rolling log's `appstatus`
|
|
224
|
+
// marker are skipped.
|
|
225
|
+
function selectLogEntries(entries ) {
|
|
226
|
+
const files = entries.filter((e) => !e.name.endsWith('/'));
|
|
227
|
+
if (files.some(isRollingPart)) {
|
|
228
|
+
const byBase = new Map(files.map((e) => [baseName(e), e]));
|
|
229
|
+
return reassembleRollingEntries(files.map(baseName)).map((n) => byBase.get(n) );
|
|
230
|
+
}
|
|
231
|
+
return files.filter((e) => baseName(e).toLowerCase() !== 'appstatus');
|
|
232
|
+
}
|
|
233
|
+
|
|
234
|
+
|
|
235
|
+
|
|
236
|
+
|
|
237
|
+
|
|
238
|
+
|
|
239
|
+
|
|
240
|
+
|
|
241
|
+
|
|
242
|
+
|
|
243
|
+
|
|
244
|
+
|
|
245
|
+
|
|
246
|
+
|
|
247
|
+
// Parses a single-entry or rolling Spark History Server zip, streaming each
|
|
248
|
+
// entry out of `source` in `chunkSize` slices. One NDJSON line decoder spans
|
|
249
|
+
// all entries, since a file-roll boundary need not fall on a line boundary.
|
|
250
|
+
export async function parseZipArchive(
|
|
251
|
+
source ,
|
|
252
|
+
state ,
|
|
253
|
+
emit ,
|
|
254
|
+
{ zstdDecoder, chunkSize = 512 * 1024, onInvalid, progressEvery = 2000, reportPct = false } ,
|
|
255
|
+
) {
|
|
256
|
+
let logEntries ;
|
|
257
|
+
let attempts ;
|
|
151
258
|
try {
|
|
152
|
-
entries =
|
|
153
|
-
|
|
154
|
-
|
|
259
|
+
const entries = await listZipEntries(source);
|
|
260
|
+
attempts = countLogAttempts(entries);
|
|
261
|
+
logEntries = attempts > 1 ? [] : selectLogEntries(entries);
|
|
262
|
+
} catch (e) {
|
|
263
|
+
onInvalid(`Could not read the zip archive: ${e instanceof Error ? e.message : String(e)}`);
|
|
155
264
|
return;
|
|
156
265
|
}
|
|
157
|
-
|
|
158
|
-
|
|
159
|
-
|
|
160
|
-
if (isRolling) {
|
|
161
|
-
try {
|
|
162
|
-
names = reassembleRollingEntries(allNames);
|
|
163
|
-
} catch {
|
|
164
|
-
emitShsError(emit, 'invalid-event-log');
|
|
165
|
-
return;
|
|
166
|
-
}
|
|
167
|
-
} else {
|
|
168
|
-
names = allNames.filter(n => n.toLowerCase() !== 'appstatus').sort(naturalCompare);
|
|
266
|
+
if (attempts > 1) {
|
|
267
|
+
onInvalid(`The zip archive holds ${attempts} application attempts. Download a single attempt, for example GET /api/v1/applications/<appId>/<attemptId>/logs.`);
|
|
268
|
+
return;
|
|
169
269
|
}
|
|
170
|
-
if (
|
|
171
|
-
|
|
270
|
+
if (logEntries.length === 0) {
|
|
271
|
+
onInvalid('The zip archive contains no event log.');
|
|
172
272
|
return;
|
|
173
273
|
}
|
|
174
274
|
|
|
275
|
+
const totalBytes = logEntries.reduce((sum, e) => sum + e.compressedSize, 0);
|
|
276
|
+
let bytesRead = 0;
|
|
175
277
|
const decoder = buildChunkDecoder();
|
|
176
278
|
const joined = [];
|
|
177
279
|
let linesProcessed = 0;
|
|
178
|
-
|
|
280
|
+
const feed = (bytes ) => {
|
|
281
|
+
joined.length = 0;
|
|
282
|
+
const lines = decoder.decode(bytes, joined);
|
|
283
|
+
for (let i = 0, j = 0; i < lines.length; i++) {
|
|
284
|
+
dispatchLine(lines[i], state, emit, joined[j]?.index === i ? joined[j++] : undefined);
|
|
285
|
+
linesProcessed++;
|
|
286
|
+
if (linesProcessed % progressEvery === 0) {
|
|
287
|
+
emit({ type: 'progress', pct: reportPct && totalBytes ? bytesRead / totalBytes : null, linesProcessed });
|
|
288
|
+
}
|
|
289
|
+
}
|
|
290
|
+
};
|
|
291
|
+
|
|
292
|
+
for (const entry of logEntries) {
|
|
293
|
+
const entryDecoder = createEntryDecoder(entry.name, feed, zstdDecoder);
|
|
179
294
|
try {
|
|
180
|
-
|
|
181
|
-
|
|
182
|
-
|
|
183
|
-
|
|
184
|
-
|
|
185
|
-
|
|
186
|
-
|
|
187
|
-
|
|
188
|
-
}
|
|
189
|
-
}
|
|
190
|
-
}, zstdDecoder);
|
|
191
|
-
} catch {
|
|
192
|
-
emitShsError(emit, 'invalid-event-log');
|
|
295
|
+
await streamZipEntry(source, entry, (chunk) => entryDecoder.push(chunk), {
|
|
296
|
+
chunkSize,
|
|
297
|
+
onRead: (bytes) => { bytesRead += bytes; },
|
|
298
|
+
});
|
|
299
|
+
await entryDecoder.end();
|
|
300
|
+
} catch (e) {
|
|
301
|
+
entryDecoder.cancel();
|
|
302
|
+
onInvalid(`Could not decompress "${entry.name}" in the zip archive: ${e instanceof Error ? e.message : String(e)}`);
|
|
193
303
|
return;
|
|
194
304
|
}
|
|
195
305
|
}
|
|
@@ -198,7 +308,7 @@ export function decodeShsArchive(
|
|
|
198
308
|
}
|
|
199
309
|
|
|
200
310
|
if (!state.app) {
|
|
201
|
-
|
|
311
|
+
onInvalid(null);
|
|
202
312
|
return;
|
|
203
313
|
}
|
|
204
314
|
emitParseCompletion(state, emit, linesProcessed);
|
package/vendor-core/shs-load.js
CHANGED
|
@@ -19,7 +19,7 @@ export const DEFAULT_IDLE_TIMEOUT_MS = envInt('SPARKFORENSICS_SHS_TIMEOUT_MS', 3
|
|
|
19
19
|
|
|
20
20
|
function collectShsAppModel(zipBytes ) {
|
|
21
21
|
return collectViaDispatch(
|
|
22
|
-
(state, emit) => decodeShsArchive(zipBytes, state, emit, nodeArchiveCodecs),
|
|
22
|
+
(state, emit, reject) => { decodeShsArchive(zipBytes, state, emit, nodeArchiveCodecs).catch(reject); },
|
|
23
23
|
(msg) => {
|
|
24
24
|
const m = msg ;
|
|
25
25
|
return mcpError(m?.code ?? 'invalid-event-log', m?.message ?? 'Failed to decode SHS archive.');
|
|
@@ -0,0 +1,11 @@
|
|
|
1
|
+
// Shared by every scope:'sql' detector and the plan views. sql.get(id).stageIds is always empty (parser-worker
|
|
2
|
+
// never populates it), so stage linkage is derived from each stage's own sqlExecutionId.
|
|
3
|
+
// Minimal param shape (not DetectorStage): external callers pass Map<StageId, Stage>.
|
|
4
|
+
export function stageIdsForSqlExec(
|
|
5
|
+
executionId ,
|
|
6
|
+
stages ,
|
|
7
|
+
) {
|
|
8
|
+
const out = [];
|
|
9
|
+
for (const s of stages.values()) if (s.sqlExecutionId === executionId) out.push(s.id);
|
|
10
|
+
return out;
|
|
11
|
+
}
|
|
@@ -1,4 +1,5 @@
|
|
|
1
1
|
|
|
2
|
+
|
|
2
3
|
|
|
3
4
|
// Single source of truth for the packed per-task numeric array: FIELDS (offset
|
|
4
5
|
// constants), TASK_FIELD_NAMES (display labels), and the finalizeStage hot-loop
|
|
@@ -74,6 +75,8 @@ export function finalizeStage(
|
|
|
74
75
|
const executorStats = new Map();
|
|
75
76
|
const failureReasons = new Map();
|
|
76
77
|
const failedTaskSamples = [];
|
|
78
|
+
// Keyed by the interned detail object: accumulateTask shares one per distinct failure.
|
|
79
|
+
const failureGroups = new Map ();
|
|
77
80
|
const localityStats = new Map();
|
|
78
81
|
let peakExecutionMemoryMax = 0;
|
|
79
82
|
|
|
@@ -82,6 +85,8 @@ export function finalizeStage(
|
|
|
82
85
|
if (t.failed) {
|
|
83
86
|
failedTasks++;
|
|
84
87
|
if (t.reason) failureReasons.set(t.reason, (failureReasons.get(t.reason) ?? 0) + 1);
|
|
88
|
+
const failure = (t ).failure;
|
|
89
|
+
if (failure) failureGroups.set(failure, (failureGroups.get(failure) ?? 0) + 1);
|
|
85
90
|
if (failedTaskSamples.length < MAX_TASK_SAMPLES) {
|
|
86
91
|
failedTaskSamples.push({
|
|
87
92
|
taskId: t.taskId, attemptNumber: t.attemptNumber, host: t.host, executorId: t.executorId,
|
|
@@ -125,6 +130,7 @@ export function finalizeStage(
|
|
|
125
130
|
stage.failedTasks = failedTasks;
|
|
126
131
|
stage.speculativeTasks = speculativeTasks;
|
|
127
132
|
stage.taskAttempts = null; // no longer needed after finalize, freeing memory
|
|
133
|
+
stage.failureDetails = null;
|
|
128
134
|
|
|
129
135
|
const arr = new Float64Array(buf);
|
|
130
136
|
state.taskStore.set(stageId, arr);
|
|
@@ -164,6 +170,10 @@ export function finalizeStage(
|
|
|
164
170
|
const failureReasonsArr = [...failureReasons.entries()].map(
|
|
165
171
|
([reason, count]) => ({ reason, count })
|
|
166
172
|
);
|
|
173
|
+
// Most frequent first; ties keep first-seen order (Array.prototype.sort is stable).
|
|
174
|
+
const failureGroupsArr = [...failureGroups.entries()]
|
|
175
|
+
.map(([detail, count]) => ({ ...detail, count }))
|
|
176
|
+
.sort((a, b) => b.count - a.count);
|
|
167
177
|
const localityStatsArr = [...localityStats.entries()].map(
|
|
168
178
|
([locality, count]) => ({ locality, count })
|
|
169
179
|
);
|
|
@@ -173,6 +183,7 @@ export function finalizeStage(
|
|
|
173
183
|
hostStats: hostStatsArr, executorStats: executorStatsArr, failureReasons: failureReasonsArr, localityStats: localityStatsArr, stragglerCount, stragglerExcessMs, longestNonStragglerMs,
|
|
174
184
|
tailReplayRecoveryMs,
|
|
175
185
|
failedTaskSamples,
|
|
186
|
+
failureGroups: failureGroupsArr,
|
|
176
187
|
peakExecutionMemoryMax,
|
|
177
188
|
taskActiveMs: computeTaskActiveMs(arr),
|
|
178
189
|
peakConcurrentTasks,
|
|
@@ -187,6 +198,9 @@ export function finalizeStage(
|
|
|
187
198
|
stageType: acc.shuffleReadBytes > 0 ? 'REDUCE' : 'MAP',
|
|
188
199
|
};
|
|
189
200
|
delete data.taskAttempts; // internal-only field, already nulled above; never part of the public message
|
|
201
|
+
delete data.failureDetails; // internal-only intern table, summarized by failureGroups
|
|
202
|
+
delete data.speculativeWinners; // internal-only late-TaskEnd pairing state, kept worker-side
|
|
203
|
+
delete data.lateSpeculationWaste;
|
|
190
204
|
|
|
191
205
|
return { type: 'stage', data };
|
|
192
206
|
}
|
|
@@ -0,0 +1,151 @@
|
|
|
1
|
+
// What actually went wrong behind a failed task attempt, read from its TaskEnd "Task End Reason".
|
|
2
|
+
// Spark's `Reason` is only the end-reason tag (ExceptionFailure, ExecutorLostFailure, ...); the
|
|
3
|
+
// real error lives in tag-specific fields (JsonProtocol.taskEndReasonToJson):
|
|
4
|
+
// ExceptionFailure Class Name, Description, Full Stack Trace
|
|
5
|
+
// ExecutorLostFailure Loss Reason (e.g. "Container killed by YARN for exceeding memory limits")
|
|
6
|
+
// FetchFailed Message (often a whole exception string, stack included)
|
|
7
|
+
// TaskKilled Kill Reason
|
|
8
|
+
// Every text field is bounded here, at ingest, so neither parser memory nor a report grows with
|
|
9
|
+
// the size of the traces Spark wrote.
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
const MAX_TEXT_CHARS = 300;
|
|
24
|
+
const MAX_EXCERPT_FRAMES = 8;
|
|
25
|
+
const MAX_EXCERPT_LINE_CHARS = 300;
|
|
26
|
+
const MAX_EXCERPT_CHARS = 2000;
|
|
27
|
+
// Distinct failures kept per stage while parsing, and shown per finding.
|
|
28
|
+
export const MAX_FAILURE_DETAILS_PER_STAGE = 50;
|
|
29
|
+
export const MAX_FAILURE_GROUPS = 5;
|
|
30
|
+
|
|
31
|
+
export const REDACTED_TEXT = '[redacted]';
|
|
32
|
+
|
|
33
|
+
function truncate(s , max ) {
|
|
34
|
+
return s.length > max ? `${s.slice(0, max - 3)}...` : s;
|
|
35
|
+
}
|
|
36
|
+
|
|
37
|
+
function text(v ) {
|
|
38
|
+
return typeof v === 'string' && v.trim().length > 0 ? v : null;
|
|
39
|
+
}
|
|
40
|
+
|
|
41
|
+
// First non-blank line, whitespace collapsed: a message can carry a whole stack trace after it.
|
|
42
|
+
function firstLine(s ) {
|
|
43
|
+
if (s == null) return null;
|
|
44
|
+
const line = s.split('\n').map((l) => l.trim()).find((l) => l.length > 0);
|
|
45
|
+
return line ? truncate(line.replace(/\s+/g, ' '), MAX_TEXT_CHARS) : null;
|
|
46
|
+
}
|
|
47
|
+
|
|
48
|
+
const PYTHON_TRACEBACK_HEAD = /(?:^|: )(?:Traceback \(most recent call last\):|An exception was thrown from the Python worker)/;
|
|
49
|
+
const JAVA_FRAME = /^\s*at \S/;
|
|
50
|
+
|
|
51
|
+
// A PySpark error's text is a whole Python traceback whose own error line (`ValueError: bad row`)
|
|
52
|
+
// comes last, before any JVM frames: index of that line among non-blank `lines`, or -1.
|
|
53
|
+
function pythonErrorLineIndex(lines ) {
|
|
54
|
+
if (lines.length < 2 || !PYTHON_TRACEBACK_HEAD.test(lines[0].trim())) return -1;
|
|
55
|
+
let last = -1;
|
|
56
|
+
for (let i = 1; i < lines.length && !JAVA_FRAME.test(lines[i]); i++) last = i;
|
|
57
|
+
return last;
|
|
58
|
+
}
|
|
59
|
+
|
|
60
|
+
// The message's headline: its first line, or a Python traceback's error line.
|
|
61
|
+
function messageLine(s ) {
|
|
62
|
+
if (s == null) return null;
|
|
63
|
+
const lines = s.split('\n').filter((l) => l.trim().length > 0);
|
|
64
|
+
const pyIdx = pythonErrorLineIndex(lines);
|
|
65
|
+
return pyIdx >= 0 ? firstLine(lines[pyIdx]) : firstLine(s);
|
|
66
|
+
}
|
|
67
|
+
|
|
68
|
+
/** Header line, the first frames, and (when cut off) a Python traceback's error line and the last
|
|
69
|
+
* `Caused by:` line, which usually name the root cause. Bounded by frame count, line length and
|
|
70
|
+
* total length. */
|
|
71
|
+
export function buildStackExcerpt(trace ) {
|
|
72
|
+
if (trace == null) return null;
|
|
73
|
+
const lines = trace.split('\n').map((l) => l.replace(/\s+$/, '')).filter((l) => l.trim().length > 0);
|
|
74
|
+
if (lines.length === 0) return null;
|
|
75
|
+
const headCount = 1 + MAX_EXCERPT_FRAMES;
|
|
76
|
+
const kept = lines.slice(0, headCount);
|
|
77
|
+
if (lines.length > headCount) {
|
|
78
|
+
const pyIdx = pythonErrorLineIndex(lines);
|
|
79
|
+
let causeIdx = -1;
|
|
80
|
+
for (let i = lines.length - 1; i >= headCount; i--) {
|
|
81
|
+
if (lines[i].startsWith('Caused by:')) { causeIdx = i; break; }
|
|
82
|
+
}
|
|
83
|
+
if (pyIdx >= headCount) kept.push('\t...', lines[pyIdx]);
|
|
84
|
+
if (pyIdx < lines.length - 1) kept.push('\t...');
|
|
85
|
+
if (causeIdx >= 0) kept.push(...lines.slice(causeIdx, causeIdx + 2));
|
|
86
|
+
}
|
|
87
|
+
return truncate(kept.map((l) => truncate(l, MAX_EXCERPT_LINE_CHARS)).join('\n'), MAX_EXCERPT_CHARS);
|
|
88
|
+
}
|
|
89
|
+
|
|
90
|
+
/** Bounded failure detail for a failed attempt's end reason, or null when there is no end reason. */
|
|
91
|
+
export function extractTaskFailureDetail(endReason ) {
|
|
92
|
+
if (!endReason) return null;
|
|
93
|
+
const reason = text(endReason['Reason']);
|
|
94
|
+
const rawMessage = text(endReason['Description']) ?? text(endReason['Message']) ?? text(endReason['Kill Reason']);
|
|
95
|
+
// FetchFailed has no Full Stack Trace field, but its Message is a full exception string.
|
|
96
|
+
const trace = text(endReason['Full Stack Trace'])
|
|
97
|
+
?? (rawMessage != null && rawMessage.trim().includes('\n') ? rawMessage : null);
|
|
98
|
+
return {
|
|
99
|
+
reason,
|
|
100
|
+
className: firstLine(text(endReason['Class Name'])),
|
|
101
|
+
message: messageLine(rawMessage),
|
|
102
|
+
lossReason: firstLine(text(endReason['Loss Reason'])),
|
|
103
|
+
stackExcerpt: buildStackExcerpt(trace),
|
|
104
|
+
};
|
|
105
|
+
}
|
|
106
|
+
|
|
107
|
+
/** Grouping key: one group per distinct error. The excerpt is left out, so two traces of the same
|
|
108
|
+
* error share a group and the first one seen is kept. */
|
|
109
|
+
export function taskFailureKey(d ) {
|
|
110
|
+
return JSON.stringify([d.reason, d.className, d.message, d.lossReason]);
|
|
111
|
+
}
|
|
112
|
+
|
|
113
|
+
/** Short name for the error: the exception class, or the end-reason tag with its loss reason. */
|
|
114
|
+
export function describeTaskFailure(d ) {
|
|
115
|
+
if (d.className) return d.className;
|
|
116
|
+
if (d.lossReason) return d.reason ? `${d.reason}: ${d.lossReason}` : d.lossReason;
|
|
117
|
+
return d.reason;
|
|
118
|
+
}
|
|
119
|
+
|
|
120
|
+
const FRAME_LINE = /^\s*(?:at \S.*|\.\.\.(?: \d+ more)?)$/;
|
|
121
|
+
const CLASS_HEADER = /^((?:Caused by: |Suppressed: )?[\w$]+(?:\.[\w$]+)+)(?::.*)?$/;
|
|
122
|
+
|
|
123
|
+
// Keeps stack frames and exception class names only: a message (the text after "Class: ", and any
|
|
124
|
+
// continuation line such as a Python traceback's `File "/path"` lines) is dropped.
|
|
125
|
+
function stripExcerptMessages(excerpt ) {
|
|
126
|
+
const kept = [];
|
|
127
|
+
for (const line of excerpt.split('\n')) {
|
|
128
|
+
if (FRAME_LINE.test(line)) { kept.push(line); continue; }
|
|
129
|
+
const header = CLASS_HEADER.exec(line.trim());
|
|
130
|
+
if (header) kept.push(header[1]);
|
|
131
|
+
}
|
|
132
|
+
return kept.join('\n');
|
|
133
|
+
}
|
|
134
|
+
|
|
135
|
+
/** Redacted copy of a failure group: the message and the message text inside the stack excerpt
|
|
136
|
+
* can carry file paths and data values. Class names, frames and the loss reason are kept (hosts in
|
|
137
|
+
* a loss reason are pseudonymized by the caller's host scan like any other free text). */
|
|
138
|
+
export function redactTaskFailureGroup (g ) {
|
|
139
|
+
return {
|
|
140
|
+
...g,
|
|
141
|
+
message: g.message == null ? null : REDACTED_TEXT,
|
|
142
|
+
stackExcerpt: g.stackExcerpt == null ? null : stripExcerptMessages(g.stackExcerpt),
|
|
143
|
+
};
|
|
144
|
+
}
|
|
145
|
+
|
|
146
|
+
/** One-line headline for a failure group, for reports and the Failed Tasks widget. */
|
|
147
|
+
export function formatTaskFailureHeadline(d ) {
|
|
148
|
+
const name = d.className ?? d.reason ?? 'Unknown failure';
|
|
149
|
+
const detail = [d.message, d.lossReason].filter((s) => s != null).join(' · ');
|
|
150
|
+
return detail ? `${name}: ${detail}` : name;
|
|
151
|
+
}
|