sparkforensics-mcp 0.2.3 → 0.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -0
- package/README.md +7 -1
- package/bin/sparkforensics-mcp.mjs +11 -5
- package/package.json +3 -3
- package/vendor-core/cli/native-zstd.js +2 -2
- package/vendor-core/detectors.js +16 -2
- package/vendor-core/docs-content/detection/fail.md +6 -2
- package/vendor-core/event-handlers.js +21 -0
- package/vendor-core/event-schemas.js +8 -0
- package/vendor-core/evidence-report.js +20 -1
- package/vendor-core/impact-estimator.js +23 -3
- package/vendor-core/occupancy.js +8 -7
- package/vendor-core/parser-worker.js +51 -18
- package/vendor-core/plan-summary.js +1 -1
- package/vendor-core/redact.js +24 -2
- package/vendor-core/run-comparison.js +8 -0
- package/vendor-core/shs-fetch.js +175 -65
- package/vendor-core/shs-load.js +1 -1
- package/vendor-core/stage-quantiles.js +65 -1
- package/vendor-core/task-failure.js +151 -0
- package/vendor-core/types.js +5 -0
- package/vendor-core/vendor/fflate.js +1 -1
- package/vendor-core/zip-archive.js +167 -0
- package/vendor-core/zstd-worker-client.js +180 -0
- package/vendor-core/zstd-worker.js +103 -0
package/vendor-core/shs-fetch.js
CHANGED
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import {
|
|
1
|
+
import { Gunzip } from './vendor/fflate.js';
|
|
2
2
|
import { createLz4BlockDecoder } from './lz4-block.js';
|
|
3
3
|
import { Decompress as ZstdDecompress } from './vendor/fzstd.js';
|
|
4
4
|
import { createSnappyBlockDecoder } from './snappy-block.js';
|
|
@@ -6,6 +6,8 @@ import { buildProxyRequestUrl, isShsErrorCode } from './shs-request.js';
|
|
|
6
6
|
import { dispatchLine, buildChunkDecoder, emitParseCompletion, } from './event-handlers.js';
|
|
7
7
|
import { ShsProxyErrorBodySchema } from './shs-schemas.js';
|
|
8
8
|
import { naturalCompare, reassembleRollingEntries } from './rolling-log-reassembly.js';
|
|
9
|
+
import { listZipEntries, streamZipEntry, } from './zip-archive.js';
|
|
10
|
+
|
|
9
11
|
|
|
10
12
|
export { naturalCompare, reassembleRollingEntries };
|
|
11
13
|
|
|
@@ -24,48 +26,84 @@ export function sniffCodec(bytes )
|
|
|
24
26
|
return null;
|
|
25
27
|
}
|
|
26
28
|
|
|
27
|
-
//
|
|
28
|
-
//
|
|
29
|
-
// whole decompressed entry as one buffer. A giant
|
|
30
|
-
// (or gzip member) declares its full content size in
|
|
31
|
-
// it (fzstd's `decompress()`, fflate's `gunzipSync()`)
|
|
32
|
-
// size as a single ArrayBuffer up front, which can exceed
|
|
33
|
-
// will allocate for a multi-GB event log ("Array buffer
|
|
29
|
+
// Decodes one zip entry's (possibly compressed) bytes as they stream out of
|
|
30
|
+
// the archive, block by block, invoking `onChunk` per decompressed piece;
|
|
31
|
+
// never materializes the whole decompressed entry as one buffer. A giant
|
|
32
|
+
// single-segment zstd frame (or gzip member) declares its full content size in
|
|
33
|
+
// the header; one-shotting it (fzstd's `decompress()`, fflate's `gunzipSync()`)
|
|
34
|
+
// allocates that entire size as a single ArrayBuffer up front, which can exceed
|
|
35
|
+
// what the browser will allocate for a multi-GB event log ("Array buffer
|
|
36
|
+
// allocation failed"). The codec is sniffed from the entry's first 8 bytes,
|
|
37
|
+
// falling back to the entry name's suffix.
|
|
34
38
|
// fflate's Gunzip and fzstd's Decompress are untyped vendor JS, so TS can't
|
|
35
39
|
// infer a construct signature for them; this local shape types the call sites
|
|
36
40
|
// without touching the vendored files.
|
|
37
41
|
|
|
38
42
|
|
|
39
|
-
|
|
40
|
-
|
|
41
|
-
|
|
42
|
-
|
|
43
|
-
|
|
44
|
-
function decodeEntry(
|
|
45
|
-
name , raw , onChunk , zstdDecoder ,
|
|
46
|
-
) {
|
|
47
|
-
const codec = sniffCodec(raw);
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
function openCodecSink(
|
|
46
|
+
codec , name , onChunk , zstdDecoder ,
|
|
47
|
+
) {
|
|
48
48
|
if (codec === 'lz4' || name.endsWith('.lz4')) {
|
|
49
49
|
const lz4 = createLz4BlockDecoder(onChunk);
|
|
50
|
-
lz4.push(
|
|
51
|
-
|
|
52
|
-
|
|
53
|
-
new (Gunzip )(onChunk)
|
|
54
|
-
|
|
55
|
-
|
|
56
|
-
|
|
50
|
+
return { push(chunk, final) { lz4.push(chunk); if (final) lz4.end(); } };
|
|
51
|
+
}
|
|
52
|
+
if (codec === 'gz' || name.endsWith('.gz')) {
|
|
53
|
+
const gunzip = new (Gunzip )(onChunk);
|
|
54
|
+
return { push: (chunk, final) => gunzip.push(chunk, final) };
|
|
55
|
+
}
|
|
56
|
+
if (codec === 'zstd' || name.endsWith('.zstd') || name.endsWith('.zst')) {
|
|
57
|
+
const zstd = zstdDecoder ? zstdDecoder(onChunk) : new (ZstdDecompress )(onChunk);
|
|
58
|
+
return { push: (chunk, final) => zstd.push(chunk, final), cancel: () => (zstd ).cancel?.() };
|
|
59
|
+
}
|
|
60
|
+
if (codec === 'snappy' || name.endsWith('.snappy')) {
|
|
57
61
|
const snappy = createSnappyBlockDecoder(onChunk);
|
|
58
|
-
snappy.push(
|
|
59
|
-
snappy.end();
|
|
60
|
-
} else {
|
|
61
|
-
onChunk(raw);
|
|
62
|
+
return { push(chunk, final) { snappy.push(chunk); if (final) snappy.end(); } };
|
|
62
63
|
}
|
|
64
|
+
return { push: (chunk) => onChunk(chunk) };
|
|
65
|
+
}
|
|
66
|
+
|
|
67
|
+
// Wraps openCodecSink for a stream of unknown length: holds back the latest
|
|
68
|
+
// chunk so the last one can be pushed with `final` set (codecs reject a
|
|
69
|
+
// trailing empty push less uniformly than a real last chunk), and buffers the
|
|
70
|
+
// start of the entry until it has the 8 bytes sniffCodec needs.
|
|
71
|
+
function createEntryDecoder(name , onChunk , zstdDecoder ) {
|
|
72
|
+
let head = new Uint8Array(0);
|
|
73
|
+
let held = null;
|
|
74
|
+
let sink = null;
|
|
75
|
+
const open = () => {
|
|
76
|
+
sink = openCodecSink(sniffCodec(head), name, onChunk, zstdDecoder);
|
|
77
|
+
held = head;
|
|
78
|
+
};
|
|
79
|
+
return {
|
|
80
|
+
async push(chunk ) {
|
|
81
|
+
if (!sink) {
|
|
82
|
+
const joined = new Uint8Array(head.length + chunk.length);
|
|
83
|
+
joined.set(head);
|
|
84
|
+
joined.set(chunk, head.length);
|
|
85
|
+
head = joined;
|
|
86
|
+
if (head.length >= 8) open();
|
|
87
|
+
return;
|
|
88
|
+
}
|
|
89
|
+
await sink.push(held , false);
|
|
90
|
+
held = chunk;
|
|
91
|
+
},
|
|
92
|
+
async end() {
|
|
93
|
+
if (!sink) {
|
|
94
|
+
if (!head.length) return;
|
|
95
|
+
open();
|
|
96
|
+
}
|
|
97
|
+
await sink .push(held , true);
|
|
98
|
+
},
|
|
99
|
+
cancel() { sink?.cancel?.(); },
|
|
100
|
+
};
|
|
63
101
|
}
|
|
64
102
|
|
|
65
103
|
|
|
66
104
|
|
|
67
|
-
function emitShsError(emit , code ) {
|
|
68
|
-
emit({ type: 'error', source: 'shs', code });
|
|
105
|
+
function emitShsError(emit , code , message ) {
|
|
106
|
+
emit({ type: 'error', source: 'shs', code, ...(message ? { message } : {}) });
|
|
69
107
|
}
|
|
70
108
|
|
|
71
109
|
export async function runParseFromUrl(
|
|
@@ -141,55 +179,127 @@ export async function runParseFromUrl(
|
|
|
141
179
|
if (total) {
|
|
142
180
|
emit({ type: 'progress', pct: 0.5, linesProcessed: 0 });
|
|
143
181
|
}
|
|
144
|
-
decodeShsArchive(zipBytes, state, emit);
|
|
182
|
+
await decodeShsArchive(zipBytes, state, emit);
|
|
183
|
+
}
|
|
184
|
+
|
|
185
|
+
// Reads the in-memory SHS download through the same random-access interface a
|
|
186
|
+
// dropped File offers, so both paths share parseZipArchive.
|
|
187
|
+
function bytesZipSource(bytes ) {
|
|
188
|
+
return {
|
|
189
|
+
size: bytes.length,
|
|
190
|
+
slice: (start, end) => ({ arrayBuffer: async () => bytes.slice(start, end).buffer }),
|
|
191
|
+
};
|
|
145
192
|
}
|
|
146
193
|
|
|
147
194
|
export function decodeShsArchive(
|
|
148
195
|
zipBytes , state , emit , { zstdDecoder } = {},
|
|
149
|
-
)
|
|
150
|
-
|
|
196
|
+
) {
|
|
197
|
+
return parseZipArchive(bytesZipSource(zipBytes), state, emit, {
|
|
198
|
+
zstdDecoder,
|
|
199
|
+
onInvalid: (detail) => emitShsError(emit, 'invalid-event-log', detail ?? undefined),
|
|
200
|
+
});
|
|
201
|
+
}
|
|
202
|
+
|
|
203
|
+
const baseName = (e ) => e.name.slice(e.name.lastIndexOf('/') + 1);
|
|
204
|
+
const isRollingPart = (e ) => /^events_\d+_/.test(baseName(e));
|
|
205
|
+
|
|
206
|
+
// The application attempts a History Server zip holds. Without an attempt ID
|
|
207
|
+
// the History Server packs every attempt into one download: each single-file
|
|
208
|
+
// log is its own root entry, each rolling log its own `eventlog_v2_` directory.
|
|
209
|
+
// Directory entries and `appstatus` markers are not attempts.
|
|
210
|
+
function countLogAttempts(entries ) {
|
|
211
|
+
const attempts = new Set ();
|
|
212
|
+
for (const e of entries) {
|
|
213
|
+
if (e.name.endsWith('/') || baseName(e).toLowerCase().startsWith('appstatus')) continue;
|
|
214
|
+
const dir = e.name.slice(0, e.name.indexOf('/') + 1);
|
|
215
|
+
attempts.add(dir || (isRollingPart(e) ? '' : e.name));
|
|
216
|
+
}
|
|
217
|
+
return attempts.size;
|
|
218
|
+
}
|
|
219
|
+
|
|
220
|
+
// The event-log entries a single-attempt History Server zip holds, in parse
|
|
221
|
+
// order. A rolling log's parts sit under an `eventlog_v2_<appId>/` directory
|
|
222
|
+
// entry; they are matched and reassembled by base name. A single-file log is
|
|
223
|
+
// one entry at the root. Directory entries and the rolling log's `appstatus`
|
|
224
|
+
// marker are skipped.
|
|
225
|
+
function selectLogEntries(entries ) {
|
|
226
|
+
const files = entries.filter((e) => !e.name.endsWith('/'));
|
|
227
|
+
if (files.some(isRollingPart)) {
|
|
228
|
+
const byBase = new Map(files.map((e) => [baseName(e), e]));
|
|
229
|
+
return reassembleRollingEntries(files.map(baseName)).map((n) => byBase.get(n) );
|
|
230
|
+
}
|
|
231
|
+
return files.filter((e) => baseName(e).toLowerCase() !== 'appstatus');
|
|
232
|
+
}
|
|
233
|
+
|
|
234
|
+
|
|
235
|
+
|
|
236
|
+
|
|
237
|
+
|
|
238
|
+
|
|
239
|
+
|
|
240
|
+
|
|
241
|
+
|
|
242
|
+
|
|
243
|
+
|
|
244
|
+
|
|
245
|
+
|
|
246
|
+
|
|
247
|
+
// Parses a single-entry or rolling Spark History Server zip, streaming each
|
|
248
|
+
// entry out of `source` in `chunkSize` slices. One NDJSON line decoder spans
|
|
249
|
+
// all entries, since a file-roll boundary need not fall on a line boundary.
|
|
250
|
+
export async function parseZipArchive(
|
|
251
|
+
source ,
|
|
252
|
+
state ,
|
|
253
|
+
emit ,
|
|
254
|
+
{ zstdDecoder, chunkSize = 512 * 1024, onInvalid, progressEvery = 2000, reportPct = false } ,
|
|
255
|
+
) {
|
|
256
|
+
let logEntries ;
|
|
257
|
+
let attempts ;
|
|
151
258
|
try {
|
|
152
|
-
entries =
|
|
153
|
-
|
|
154
|
-
|
|
259
|
+
const entries = await listZipEntries(source);
|
|
260
|
+
attempts = countLogAttempts(entries);
|
|
261
|
+
logEntries = attempts > 1 ? [] : selectLogEntries(entries);
|
|
262
|
+
} catch (e) {
|
|
263
|
+
onInvalid(`Could not read the zip archive: ${e instanceof Error ? e.message : String(e)}`);
|
|
155
264
|
return;
|
|
156
265
|
}
|
|
157
|
-
|
|
158
|
-
|
|
159
|
-
|
|
160
|
-
if (isRolling) {
|
|
161
|
-
try {
|
|
162
|
-
names = reassembleRollingEntries(allNames);
|
|
163
|
-
} catch {
|
|
164
|
-
emitShsError(emit, 'invalid-event-log');
|
|
165
|
-
return;
|
|
166
|
-
}
|
|
167
|
-
} else {
|
|
168
|
-
names = allNames.filter(n => n.toLowerCase() !== 'appstatus').sort(naturalCompare);
|
|
266
|
+
if (attempts > 1) {
|
|
267
|
+
onInvalid(`The zip archive holds ${attempts} application attempts. Download a single attempt, for example GET /api/v1/applications/<appId>/<attemptId>/logs.`);
|
|
268
|
+
return;
|
|
169
269
|
}
|
|
170
|
-
if (
|
|
171
|
-
|
|
270
|
+
if (logEntries.length === 0) {
|
|
271
|
+
onInvalid('The zip archive contains no event log.');
|
|
172
272
|
return;
|
|
173
273
|
}
|
|
174
274
|
|
|
275
|
+
const totalBytes = logEntries.reduce((sum, e) => sum + e.compressedSize, 0);
|
|
276
|
+
let bytesRead = 0;
|
|
175
277
|
const decoder = buildChunkDecoder();
|
|
176
278
|
const joined = [];
|
|
177
279
|
let linesProcessed = 0;
|
|
178
|
-
|
|
280
|
+
const feed = (bytes ) => {
|
|
281
|
+
joined.length = 0;
|
|
282
|
+
const lines = decoder.decode(bytes, joined);
|
|
283
|
+
for (let i = 0, j = 0; i < lines.length; i++) {
|
|
284
|
+
dispatchLine(lines[i], state, emit, joined[j]?.index === i ? joined[j++] : undefined);
|
|
285
|
+
linesProcessed++;
|
|
286
|
+
if (linesProcessed % progressEvery === 0) {
|
|
287
|
+
emit({ type: 'progress', pct: reportPct && totalBytes ? bytesRead / totalBytes : null, linesProcessed });
|
|
288
|
+
}
|
|
289
|
+
}
|
|
290
|
+
};
|
|
291
|
+
|
|
292
|
+
for (const entry of logEntries) {
|
|
293
|
+
const entryDecoder = createEntryDecoder(entry.name, feed, zstdDecoder);
|
|
179
294
|
try {
|
|
180
|
-
|
|
181
|
-
|
|
182
|
-
|
|
183
|
-
|
|
184
|
-
|
|
185
|
-
|
|
186
|
-
|
|
187
|
-
|
|
188
|
-
}
|
|
189
|
-
}
|
|
190
|
-
}, zstdDecoder);
|
|
191
|
-
} catch {
|
|
192
|
-
emitShsError(emit, 'invalid-event-log');
|
|
295
|
+
await streamZipEntry(source, entry, (chunk) => entryDecoder.push(chunk), {
|
|
296
|
+
chunkSize,
|
|
297
|
+
onRead: (bytes) => { bytesRead += bytes; },
|
|
298
|
+
});
|
|
299
|
+
await entryDecoder.end();
|
|
300
|
+
} catch (e) {
|
|
301
|
+
entryDecoder.cancel();
|
|
302
|
+
onInvalid(`Could not decompress "${entry.name}" in the zip archive: ${e instanceof Error ? e.message : String(e)}`);
|
|
193
303
|
return;
|
|
194
304
|
}
|
|
195
305
|
}
|
|
@@ -198,7 +308,7 @@ export function decodeShsArchive(
|
|
|
198
308
|
}
|
|
199
309
|
|
|
200
310
|
if (!state.app) {
|
|
201
|
-
|
|
311
|
+
onInvalid(null);
|
|
202
312
|
return;
|
|
203
313
|
}
|
|
204
314
|
emitParseCompletion(state, emit, linesProcessed);
|
package/vendor-core/shs-load.js
CHANGED
|
@@ -19,7 +19,7 @@ export const DEFAULT_IDLE_TIMEOUT_MS = envInt('SPARKFORENSICS_SHS_TIMEOUT_MS', 3
|
|
|
19
19
|
|
|
20
20
|
function collectShsAppModel(zipBytes ) {
|
|
21
21
|
return collectViaDispatch(
|
|
22
|
-
(state, emit) => decodeShsArchive(zipBytes, state, emit, nodeArchiveCodecs),
|
|
22
|
+
(state, emit, reject) => { decodeShsArchive(zipBytes, state, emit, nodeArchiveCodecs).catch(reject); },
|
|
23
23
|
(msg) => {
|
|
24
24
|
const m = msg ;
|
|
25
25
|
return mcpError(m?.code ?? 'invalid-event-log', m?.message ?? 'Failed to decode SHS archive.');
|
|
@@ -1,4 +1,5 @@
|
|
|
1
1
|
|
|
2
|
+
|
|
2
3
|
|
|
3
4
|
// Single source of truth for the packed per-task numeric array: FIELDS (offset
|
|
4
5
|
// constants), TASK_FIELD_NAMES (display labels), and the finalizeStage hot-loop
|
|
@@ -74,6 +75,8 @@ export function finalizeStage(
|
|
|
74
75
|
const executorStats = new Map();
|
|
75
76
|
const failureReasons = new Map();
|
|
76
77
|
const failedTaskSamples = [];
|
|
78
|
+
// Keyed by the interned detail object: accumulateTask shares one per distinct failure.
|
|
79
|
+
const failureGroups = new Map ();
|
|
77
80
|
const localityStats = new Map();
|
|
78
81
|
let peakExecutionMemoryMax = 0;
|
|
79
82
|
|
|
@@ -82,6 +85,8 @@ export function finalizeStage(
|
|
|
82
85
|
if (t.failed) {
|
|
83
86
|
failedTasks++;
|
|
84
87
|
if (t.reason) failureReasons.set(t.reason, (failureReasons.get(t.reason) ?? 0) + 1);
|
|
88
|
+
const failure = (t ).failure;
|
|
89
|
+
if (failure) failureGroups.set(failure, (failureGroups.get(failure) ?? 0) + 1);
|
|
85
90
|
if (failedTaskSamples.length < MAX_TASK_SAMPLES) {
|
|
86
91
|
failedTaskSamples.push({
|
|
87
92
|
taskId: t.taskId, attemptNumber: t.attemptNumber, host: t.host, executorId: t.executorId,
|
|
@@ -125,6 +130,7 @@ export function finalizeStage(
|
|
|
125
130
|
stage.failedTasks = failedTasks;
|
|
126
131
|
stage.speculativeTasks = speculativeTasks;
|
|
127
132
|
stage.taskAttempts = null; // no longer needed after finalize, freeing memory
|
|
133
|
+
stage.failureDetails = null;
|
|
128
134
|
|
|
129
135
|
const arr = new Float64Array(buf);
|
|
130
136
|
state.taskStore.set(stageId, arr);
|
|
@@ -150,6 +156,11 @@ export function finalizeStage(
|
|
|
150
156
|
}
|
|
151
157
|
}
|
|
152
158
|
|
|
159
|
+
const peakConcurrentTasks = computePeakConcurrentTasks(arr);
|
|
160
|
+
const tailReplayRecoveryMs = stragglerCount > 0
|
|
161
|
+
? computeTailReplayRecoveryMs(arr, p50, peakConcurrentTasks)
|
|
162
|
+
: 0; // no task over 4x P50: both replays schedule the same durations
|
|
163
|
+
|
|
153
164
|
const hostStatsArr = [...hostStats.entries()].map(
|
|
154
165
|
([host, s]) => ({ host, taskCount: s.taskCount, totalDuration: s.totalDuration })
|
|
155
166
|
);
|
|
@@ -159,6 +170,10 @@ export function finalizeStage(
|
|
|
159
170
|
const failureReasonsArr = [...failureReasons.entries()].map(
|
|
160
171
|
([reason, count]) => ({ reason, count })
|
|
161
172
|
);
|
|
173
|
+
// Most frequent first; ties keep first-seen order (Array.prototype.sort is stable).
|
|
174
|
+
const failureGroupsArr = [...failureGroups.entries()]
|
|
175
|
+
.map(([detail, count]) => ({ ...detail, count }))
|
|
176
|
+
.sort((a, b) => b.count - a.count);
|
|
162
177
|
const localityStatsArr = [...localityStats.entries()].map(
|
|
163
178
|
([locality, count]) => ({ locality, count })
|
|
164
179
|
);
|
|
@@ -166,10 +181,12 @@ export function finalizeStage(
|
|
|
166
181
|
const data = {
|
|
167
182
|
...stage,
|
|
168
183
|
hostStats: hostStatsArr, executorStats: executorStatsArr, failureReasons: failureReasonsArr, localityStats: localityStatsArr, stragglerCount, stragglerExcessMs, longestNonStragglerMs,
|
|
184
|
+
tailReplayRecoveryMs,
|
|
169
185
|
failedTaskSamples,
|
|
186
|
+
failureGroups: failureGroupsArr,
|
|
170
187
|
peakExecutionMemoryMax,
|
|
171
188
|
taskActiveMs: computeTaskActiveMs(arr),
|
|
172
|
-
peakConcurrentTasks
|
|
189
|
+
peakConcurrentTasks,
|
|
173
190
|
taskDurationP50: p50,
|
|
174
191
|
taskDurationP95: p95,
|
|
175
192
|
taskDurationMax: max,
|
|
@@ -181,6 +198,7 @@ export function finalizeStage(
|
|
|
181
198
|
stageType: acc.shuffleReadBytes > 0 ? 'REDUCE' : 'MAP',
|
|
182
199
|
};
|
|
183
200
|
delete data.taskAttempts; // internal-only field, already nulled above; never part of the public message
|
|
201
|
+
delete data.failureDetails; // internal-only intern table, summarized by failureGroups
|
|
184
202
|
|
|
185
203
|
return { type: 'stage', data };
|
|
186
204
|
}
|
|
@@ -234,6 +252,52 @@ export function computePeakConcurrentTasks(arr ) {
|
|
|
234
252
|
return peak;
|
|
235
253
|
}
|
|
236
254
|
|
|
255
|
+
// Wall-clock a tail fix recovers, replayed from the stage's own tasks: list scheduling (tasks in
|
|
256
|
+
// launch order, each on the slot that frees first) over `slots` slots, once with the real
|
|
257
|
+
// durations and once with every task over 4x P50 (the straggler definition above) capped at P50.
|
|
258
|
+
// The difference is the claim; dev/eval-tail-replay.mjs keeps an independent copy as its ground
|
|
259
|
+
// truth. Equal free times are interchangeable slots, so which one a min-heap picks can't change
|
|
260
|
+
// the result. Equal launch times keep the task array's order.
|
|
261
|
+
export function computeTailReplayRecoveryMs(arr , p50 , slots ) {
|
|
262
|
+
const taskCount = arr.length / FIELDS.STRIDE;
|
|
263
|
+
if (taskCount < 2 || !(p50 > 0)) return 0;
|
|
264
|
+
const order = new Uint32Array(taskCount);
|
|
265
|
+
for (let i = 0; i < taskCount; i++) order[i] = i;
|
|
266
|
+
order.sort((a, b) => arr[a * FIELDS.STRIDE + FIELDS.LAUNCH_TIME] - arr[b * FIELDS.STRIDE + FIELDS.LAUNCH_TIME] || a - b);
|
|
267
|
+
const free = new Float64Array(Math.max(1, Math.min(slots, taskCount)));
|
|
268
|
+
const capAboveMs = 4 * p50;
|
|
269
|
+
const actualEndMs = listScheduleEndMs(arr, order, free, Infinity, p50);
|
|
270
|
+
const fixedEndMs = listScheduleEndMs(arr, order, free, capAboveMs, p50);
|
|
271
|
+
return Math.max(0, actualEndMs - fixedEndMs);
|
|
272
|
+
}
|
|
273
|
+
|
|
274
|
+
// End of a list schedule over `free` (a min-heap of slot free times, reset here), each task's
|
|
275
|
+
// duration replaced by `cappedMs` when over `capAboveMs`.
|
|
276
|
+
function listScheduleEndMs(
|
|
277
|
+
arr , order , free , capAboveMs , cappedMs ,
|
|
278
|
+
) {
|
|
279
|
+
free.fill(0);
|
|
280
|
+
const n = free.length;
|
|
281
|
+
let endMs = 0;
|
|
282
|
+
for (let t = 0; t < order.length; t++) {
|
|
283
|
+
const duration = arr[order[t] * FIELDS.STRIDE + FIELDS.DURATION];
|
|
284
|
+
const finish = free[0] + (duration > capAboveMs ? cappedMs : duration);
|
|
285
|
+
if (finish > endMs) endMs = finish;
|
|
286
|
+
// Replace the root (earliest free slot) and sift it down.
|
|
287
|
+
let i = 0;
|
|
288
|
+
for (;;) {
|
|
289
|
+
const left = 2 * i + 1;
|
|
290
|
+
if (left >= n) break;
|
|
291
|
+
const child = left + 1 < n && free[left + 1] < free[left] ? left + 1 : left;
|
|
292
|
+
if (free[child] >= finish) break;
|
|
293
|
+
free[i] = free[child];
|
|
294
|
+
i = child;
|
|
295
|
+
}
|
|
296
|
+
free[i] = finish;
|
|
297
|
+
}
|
|
298
|
+
return endMs;
|
|
299
|
+
}
|
|
300
|
+
|
|
237
301
|
export function computeFieldQuantiles(arr , fieldIndex ) {
|
|
238
302
|
const taskCount = arr.length / FIELDS.STRIDE;
|
|
239
303
|
if (taskCount === 0) return { p50: 0, p95: 0, max: 0 };
|
|
@@ -0,0 +1,151 @@
|
|
|
1
|
+
// What actually went wrong behind a failed task attempt, read from its TaskEnd "Task End Reason".
|
|
2
|
+
// Spark's `Reason` is only the end-reason tag (ExceptionFailure, ExecutorLostFailure, ...); the
|
|
3
|
+
// real error lives in tag-specific fields (JsonProtocol.taskEndReasonToJson):
|
|
4
|
+
// ExceptionFailure Class Name, Description, Full Stack Trace
|
|
5
|
+
// ExecutorLostFailure Loss Reason (e.g. "Container killed by YARN for exceeding memory limits")
|
|
6
|
+
// FetchFailed Message (often a whole exception string, stack included)
|
|
7
|
+
// TaskKilled Kill Reason
|
|
8
|
+
// Every text field is bounded here, at ingest, so neither parser memory nor a report grows with
|
|
9
|
+
// the size of the traces Spark wrote.
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
const MAX_TEXT_CHARS = 300;
|
|
24
|
+
const MAX_EXCERPT_FRAMES = 8;
|
|
25
|
+
const MAX_EXCERPT_LINE_CHARS = 300;
|
|
26
|
+
const MAX_EXCERPT_CHARS = 2000;
|
|
27
|
+
// Distinct failures kept per stage while parsing, and shown per finding.
|
|
28
|
+
export const MAX_FAILURE_DETAILS_PER_STAGE = 50;
|
|
29
|
+
export const MAX_FAILURE_GROUPS = 5;
|
|
30
|
+
|
|
31
|
+
export const REDACTED_TEXT = '[redacted]';
|
|
32
|
+
|
|
33
|
+
function truncate(s , max ) {
|
|
34
|
+
return s.length > max ? `${s.slice(0, max - 3)}...` : s;
|
|
35
|
+
}
|
|
36
|
+
|
|
37
|
+
function text(v ) {
|
|
38
|
+
return typeof v === 'string' && v.trim().length > 0 ? v : null;
|
|
39
|
+
}
|
|
40
|
+
|
|
41
|
+
// First non-blank line, whitespace collapsed: a message can carry a whole stack trace after it.
|
|
42
|
+
function firstLine(s ) {
|
|
43
|
+
if (s == null) return null;
|
|
44
|
+
const line = s.split('\n').map((l) => l.trim()).find((l) => l.length > 0);
|
|
45
|
+
return line ? truncate(line.replace(/\s+/g, ' '), MAX_TEXT_CHARS) : null;
|
|
46
|
+
}
|
|
47
|
+
|
|
48
|
+
const PYTHON_TRACEBACK_HEAD = /(?:^|: )(?:Traceback \(most recent call last\):|An exception was thrown from the Python worker)/;
|
|
49
|
+
const JAVA_FRAME = /^\s*at \S/;
|
|
50
|
+
|
|
51
|
+
// A PySpark error's text is a whole Python traceback whose own error line (`ValueError: bad row`)
|
|
52
|
+
// comes last, before any JVM frames: index of that line among non-blank `lines`, or -1.
|
|
53
|
+
function pythonErrorLineIndex(lines ) {
|
|
54
|
+
if (lines.length < 2 || !PYTHON_TRACEBACK_HEAD.test(lines[0].trim())) return -1;
|
|
55
|
+
let last = -1;
|
|
56
|
+
for (let i = 1; i < lines.length && !JAVA_FRAME.test(lines[i]); i++) last = i;
|
|
57
|
+
return last;
|
|
58
|
+
}
|
|
59
|
+
|
|
60
|
+
// The message's headline: its first line, or a Python traceback's error line.
|
|
61
|
+
function messageLine(s ) {
|
|
62
|
+
if (s == null) return null;
|
|
63
|
+
const lines = s.split('\n').filter((l) => l.trim().length > 0);
|
|
64
|
+
const pyIdx = pythonErrorLineIndex(lines);
|
|
65
|
+
return pyIdx >= 0 ? firstLine(lines[pyIdx]) : firstLine(s);
|
|
66
|
+
}
|
|
67
|
+
|
|
68
|
+
/** Header line, the first frames, and (when cut off) a Python traceback's error line and the last
|
|
69
|
+
* `Caused by:` line, which usually name the root cause. Bounded by frame count, line length and
|
|
70
|
+
* total length. */
|
|
71
|
+
export function buildStackExcerpt(trace ) {
|
|
72
|
+
if (trace == null) return null;
|
|
73
|
+
const lines = trace.split('\n').map((l) => l.replace(/\s+$/, '')).filter((l) => l.trim().length > 0);
|
|
74
|
+
if (lines.length === 0) return null;
|
|
75
|
+
const headCount = 1 + MAX_EXCERPT_FRAMES;
|
|
76
|
+
const kept = lines.slice(0, headCount);
|
|
77
|
+
if (lines.length > headCount) {
|
|
78
|
+
const pyIdx = pythonErrorLineIndex(lines);
|
|
79
|
+
let causeIdx = -1;
|
|
80
|
+
for (let i = lines.length - 1; i >= headCount; i--) {
|
|
81
|
+
if (lines[i].startsWith('Caused by:')) { causeIdx = i; break; }
|
|
82
|
+
}
|
|
83
|
+
if (pyIdx >= headCount) kept.push('\t...', lines[pyIdx]);
|
|
84
|
+
if (pyIdx < lines.length - 1) kept.push('\t...');
|
|
85
|
+
if (causeIdx >= 0) kept.push(...lines.slice(causeIdx, causeIdx + 2));
|
|
86
|
+
}
|
|
87
|
+
return truncate(kept.map((l) => truncate(l, MAX_EXCERPT_LINE_CHARS)).join('\n'), MAX_EXCERPT_CHARS);
|
|
88
|
+
}
|
|
89
|
+
|
|
90
|
+
/** Bounded failure detail for a failed attempt's end reason, or null when there is no end reason. */
|
|
91
|
+
export function extractTaskFailureDetail(endReason ) {
|
|
92
|
+
if (!endReason) return null;
|
|
93
|
+
const reason = text(endReason['Reason']);
|
|
94
|
+
const rawMessage = text(endReason['Description']) ?? text(endReason['Message']) ?? text(endReason['Kill Reason']);
|
|
95
|
+
// FetchFailed has no Full Stack Trace field, but its Message is a full exception string.
|
|
96
|
+
const trace = text(endReason['Full Stack Trace'])
|
|
97
|
+
?? (rawMessage != null && rawMessage.trim().includes('\n') ? rawMessage : null);
|
|
98
|
+
return {
|
|
99
|
+
reason,
|
|
100
|
+
className: firstLine(text(endReason['Class Name'])),
|
|
101
|
+
message: messageLine(rawMessage),
|
|
102
|
+
lossReason: firstLine(text(endReason['Loss Reason'])),
|
|
103
|
+
stackExcerpt: buildStackExcerpt(trace),
|
|
104
|
+
};
|
|
105
|
+
}
|
|
106
|
+
|
|
107
|
+
/** Grouping key: one group per distinct error. The excerpt is left out, so two traces of the same
|
|
108
|
+
* error share a group and the first one seen is kept. */
|
|
109
|
+
export function taskFailureKey(d ) {
|
|
110
|
+
return JSON.stringify([d.reason, d.className, d.message, d.lossReason]);
|
|
111
|
+
}
|
|
112
|
+
|
|
113
|
+
/** Short name for the error: the exception class, or the end-reason tag with its loss reason. */
|
|
114
|
+
export function describeTaskFailure(d ) {
|
|
115
|
+
if (d.className) return d.className;
|
|
116
|
+
if (d.lossReason) return d.reason ? `${d.reason}: ${d.lossReason}` : d.lossReason;
|
|
117
|
+
return d.reason;
|
|
118
|
+
}
|
|
119
|
+
|
|
120
|
+
const FRAME_LINE = /^\s*(?:at \S.*|\.\.\.(?: \d+ more)?)$/;
|
|
121
|
+
const CLASS_HEADER = /^((?:Caused by: |Suppressed: )?[\w$]+(?:\.[\w$]+)+)(?::.*)?$/;
|
|
122
|
+
|
|
123
|
+
// Keeps stack frames and exception class names only: a message (the text after "Class: ", and any
|
|
124
|
+
// continuation line such as a Python traceback's `File "/path"` lines) is dropped.
|
|
125
|
+
function stripExcerptMessages(excerpt ) {
|
|
126
|
+
const kept = [];
|
|
127
|
+
for (const line of excerpt.split('\n')) {
|
|
128
|
+
if (FRAME_LINE.test(line)) { kept.push(line); continue; }
|
|
129
|
+
const header = CLASS_HEADER.exec(line.trim());
|
|
130
|
+
if (header) kept.push(header[1]);
|
|
131
|
+
}
|
|
132
|
+
return kept.join('\n');
|
|
133
|
+
}
|
|
134
|
+
|
|
135
|
+
/** Redacted copy of a failure group: the message and the message text inside the stack excerpt
|
|
136
|
+
* can carry file paths and data values. Class names, frames and the loss reason are kept (hosts in
|
|
137
|
+
* a loss reason are pseudonymized by the caller's host scan like any other free text). */
|
|
138
|
+
export function redactTaskFailureGroup (g ) {
|
|
139
|
+
return {
|
|
140
|
+
...g,
|
|
141
|
+
message: g.message == null ? null : REDACTED_TEXT,
|
|
142
|
+
stackExcerpt: g.stackExcerpt == null ? null : stripExcerptMessages(g.stackExcerpt),
|
|
143
|
+
};
|
|
144
|
+
}
|
|
145
|
+
|
|
146
|
+
/** One-line headline for a failure group, for reports and the Failed Tasks widget. */
|
|
147
|
+
export function formatTaskFailureHeadline(d ) {
|
|
148
|
+
const name = d.className ?? d.reason ?? 'Unknown failure';
|
|
149
|
+
const detail = [d.message, d.lossReason].filter((s) => s != null).join(' · ');
|
|
150
|
+
return detail ? `${name}: ${detail}` : name;
|
|
151
|
+
}
|
package/vendor-core/types.js
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
// Vendored from fflate@0.8.3 (esm/browser.js), MIT license.
|
|
2
2
|
// https://github.com/101arrowz/fflate — sha256 b7ca4450b19559a1d50eb381adcee94b82449674be4cd17789d9beba7e6122a1
|
|
3
|
-
// Only
|
|
3
|
+
// Only Gunzip/UnzipInflate/gunzipSync/strFromU8 are used by this project's source (see src/shs-fetch.ts, src/zip-archive.ts); tests also use the zip writers and strToU8.
|
|
4
4
|
// DEFLATE is a complex format; to read this code, you should probably check the RFC first:
|
|
5
5
|
// https://tools.ietf.org/html/rfc1951
|
|
6
6
|
// You may also wish to take a look at the guide I made about this program:
|