sparkforensics-mcp 0.2.4 → 0.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (71) hide show
  1. package/LICENSE +21 -0
  2. package/README.md +7 -1
  3. package/bin/sparkforensics-mcp.mjs +41 -10
  4. package/package.json +1 -1
  5. package/vendor-core/analyzer.js +156 -48
  6. package/vendor-core/check-coverage.js +88 -0
  7. package/vendor-core/cli/budgets.js +31 -18
  8. package/vendor-core/cli/collect-run.js +76 -31
  9. package/vendor-core/cli/native-zstd.js +2 -2
  10. package/vendor-core/cli/threshold-config.js +28 -0
  11. package/vendor-core/comparison-verdict.js +177 -0
  12. package/vendor-core/core-source-hash.txt +1 -0
  13. package/vendor-core/core-usage-locality.js +56 -2
  14. package/vendor-core/detector-docs.js +58 -0
  15. package/vendor-core/detectors.js +933 -459
  16. package/vendor-core/docs-config.js +0 -36
  17. package/vendor-core/docs-content/chapters/nav-index.json +31 -0
  18. package/vendor-core/docs-content/detection/cstor.md +9 -0
  19. package/vendor-core/docs-content/detection/fail.md +6 -2
  20. package/vendor-core/docs-site-config.js +3 -0
  21. package/vendor-core/event-handlers.js +191 -6
  22. package/vendor-core/event-schemas.js +29 -0
  23. package/vendor-core/evidence-report.js +440 -112
  24. package/vendor-core/export-data.js +79 -6
  25. package/vendor-core/finding-action-label.js +9 -88
  26. package/vendor-core/finding-filter-predicate.js +9 -0
  27. package/vendor-core/finding-generic-recommendation.js +6 -104
  28. package/vendor-core/finding-names.js +21 -45
  29. package/vendor-core/finding-presentation.js +333 -0
  30. package/vendor-core/finding-tag-help.js +110 -0
  31. package/vendor-core/finding-types.js +361 -0
  32. package/vendor-core/findings-of-type.js +11 -0
  33. package/vendor-core/format-utils.js +92 -27
  34. package/vendor-core/html-export.js +51 -0
  35. package/vendor-core/impact-band.js +21 -8
  36. package/vendor-core/impact-estimator.js +8 -521
  37. package/vendor-core/impact-format.js +114 -0
  38. package/vendor-core/impact-model.js +175 -0
  39. package/vendor-core/ingest.js +2 -0
  40. package/vendor-core/intervals.js +13 -0
  41. package/vendor-core/list-runs.js +2 -3
  42. package/vendor-core/load-vendored.js +70 -5
  43. package/vendor-core/mcp-server-factory.js +14 -10
  44. package/vendor-core/mcp-tools.js +105 -45
  45. package/vendor-core/model-assembler.js +12 -0
  46. package/vendor-core/occupancy.js +1 -1
  47. package/vendor-core/parser-worker.js +22 -5
  48. package/vendor-core/plan-graph-model.js +3 -2
  49. package/vendor-core/plan-node-detail.js +1 -1
  50. package/vendor-core/recommendation-rollup.js +63 -3
  51. package/vendor-core/redact.js +68 -28
  52. package/vendor-core/run-comparison.js +40 -7
  53. package/vendor-core/run-interpretation.js +290 -0
  54. package/vendor-core/run-outcome.js +74 -0
  55. package/vendor-core/run-payload.js +17 -0
  56. package/vendor-core/run-shape.js +40 -0
  57. package/vendor-core/run-verdict.js +353 -0
  58. package/vendor-core/scaling-sim.js +4 -5
  59. package/vendor-core/scorecard-estimates.js +62 -0
  60. package/vendor-core/shs-fetch.js +175 -65
  61. package/vendor-core/shs-load.js +1 -1
  62. package/vendor-core/sql-stages.js +11 -0
  63. package/vendor-core/stage-quantiles.js +14 -0
  64. package/vendor-core/task-failure.js +151 -0
  65. package/vendor-core/threshold-overrides.js +160 -0
  66. package/vendor-core/threshold-summary.js +11 -33
  67. package/vendor-core/types.js +6 -42
  68. package/vendor-core/vendor/fflate.js +1 -1
  69. package/vendor-core/wall-clock.js +1 -12
  70. package/vendor-core/wasted-core-hours.js +2 -2
  71. package/vendor-core/zip-archive.js +167 -0
@@ -1,4 +1,4 @@
1
- import { unzipSync, Gunzip } from './vendor/fflate.js';
1
+ import { Gunzip } from './vendor/fflate.js';
2
2
  import { createLz4BlockDecoder } from './lz4-block.js';
3
3
  import { Decompress as ZstdDecompress } from './vendor/fzstd.js';
4
4
  import { createSnappyBlockDecoder } from './snappy-block.js';
@@ -6,6 +6,8 @@ import { buildProxyRequestUrl, isShsErrorCode } from './shs-request.js';
6
6
  import { dispatchLine, buildChunkDecoder, emitParseCompletion, } from './event-handlers.js';
7
7
  import { ShsProxyErrorBodySchema } from './shs-schemas.js';
8
8
  import { naturalCompare, reassembleRollingEntries } from './rolling-log-reassembly.js';
9
+ import { listZipEntries, streamZipEntry, } from './zip-archive.js';
10
+
9
11
 
10
12
  export { naturalCompare, reassembleRollingEntries };
11
13
 
@@ -24,48 +26,84 @@ export function sniffCodec(bytes )
24
26
  return null;
25
27
  }
26
28
 
27
- // Streams one already-in-memory zip entry through its codec's block-by-block
28
- // decoder, invoking `onChunk` per decompressed piece; never materializes the
29
- // whole decompressed entry as one buffer. A giant single-segment zstd frame
30
- // (or gzip member) declares its full content size in the header; one-shotting
31
- // it (fzstd's `decompress()`, fflate's `gunzipSync()`) allocates that entire
32
- // size as a single ArrayBuffer up front, which can exceed what the browser
33
- // will allocate for a multi-GB event log ("Array buffer allocation failed").
29
+ // Decodes one zip entry's (possibly compressed) bytes as they stream out of
30
+ // the archive, block by block, invoking `onChunk` per decompressed piece;
31
+ // never materializes the whole decompressed entry as one buffer. A giant
32
+ // single-segment zstd frame (or gzip member) declares its full content size in
33
+ // the header; one-shotting it (fzstd's `decompress()`, fflate's `gunzipSync()`)
34
+ // allocates that entire size as a single ArrayBuffer up front, which can exceed
35
+ // what the browser will allocate for a multi-GB event log ("Array buffer
36
+ // allocation failed"). The codec is sniffed from the entry's first 8 bytes,
37
+ // falling back to the entry name's suffix.
34
38
  // fflate's Gunzip and fzstd's Decompress are untyped vendor JS, so TS can't
35
39
  // infer a construct signature for them; this local shape types the call sites
36
40
  // without touching the vendored files.
37
41
 
38
42
 
39
- // Like parser-worker.ts's RunOpts.zstdDecoder, but synchronous: decodeEntry never awaits push(),
40
- // so its result is typed `undefined` (not `void`, which would also accept an async decoder's
41
- // Promise). Node callers pass cli/native-zstd.ts's createNativeZstdDecoder (nodeArchiveCodecs).
42
-
43
-
44
- function decodeEntry(
45
- name , raw , onChunk , zstdDecoder ,
46
- ) {
47
- const codec = sniffCodec(raw);
43
+
44
+
45
+ function openCodecSink(
46
+ codec , name , onChunk , zstdDecoder ,
47
+ ) {
48
48
  if (codec === 'lz4' || name.endsWith('.lz4')) {
49
49
  const lz4 = createLz4BlockDecoder(onChunk);
50
- lz4.push(raw);
51
- lz4.end();
52
- } else if (codec === 'gz' || name.endsWith('.gz')) {
53
- new (Gunzip )(onChunk).push(raw, true);
54
- } else if (codec === 'zstd' || name.endsWith('.zstd') || name.endsWith('.zst')) {
55
- (zstdDecoder ? zstdDecoder(onChunk) : new (ZstdDecompress )(onChunk)).push(raw, true);
56
- } else if (codec === 'snappy' || name.endsWith('.snappy')) {
50
+ return { push(chunk, final) { lz4.push(chunk); if (final) lz4.end(); } };
51
+ }
52
+ if (codec === 'gz' || name.endsWith('.gz')) {
53
+ const gunzip = new (Gunzip )(onChunk);
54
+ return { push: (chunk, final) => gunzip.push(chunk, final) };
55
+ }
56
+ if (codec === 'zstd' || name.endsWith('.zstd') || name.endsWith('.zst')) {
57
+ const zstd = zstdDecoder ? zstdDecoder(onChunk) : new (ZstdDecompress )(onChunk);
58
+ return { push: (chunk, final) => zstd.push(chunk, final), cancel: () => (zstd ).cancel?.() };
59
+ }
60
+ if (codec === 'snappy' || name.endsWith('.snappy')) {
57
61
  const snappy = createSnappyBlockDecoder(onChunk);
58
- snappy.push(raw);
59
- snappy.end();
60
- } else {
61
- onChunk(raw);
62
+ return { push(chunk, final) { snappy.push(chunk); if (final) snappy.end(); } };
62
63
  }
64
+ return { push: (chunk) => onChunk(chunk) };
65
+ }
66
+
67
+ // Wraps openCodecSink for a stream of unknown length: holds back the latest
68
+ // chunk so the last one can be pushed with `final` set (codecs reject a
69
+ // trailing empty push less uniformly than a real last chunk), and buffers the
70
+ // start of the entry until it has the 8 bytes sniffCodec needs.
71
+ function createEntryDecoder(name , onChunk , zstdDecoder ) {
72
+ let head = new Uint8Array(0);
73
+ let held = null;
74
+ let sink = null;
75
+ const open = () => {
76
+ sink = openCodecSink(sniffCodec(head), name, onChunk, zstdDecoder);
77
+ held = head;
78
+ };
79
+ return {
80
+ async push(chunk ) {
81
+ if (!sink) {
82
+ const joined = new Uint8Array(head.length + chunk.length);
83
+ joined.set(head);
84
+ joined.set(chunk, head.length);
85
+ head = joined;
86
+ if (head.length >= 8) open();
87
+ return;
88
+ }
89
+ await sink.push(held , false);
90
+ held = chunk;
91
+ },
92
+ async end() {
93
+ if (!sink) {
94
+ if (!head.length) return;
95
+ open();
96
+ }
97
+ await sink .push(held , true);
98
+ },
99
+ cancel() { sink?.cancel?.(); },
100
+ };
63
101
  }
64
102
 
65
103
 
66
104
 
67
- function emitShsError(emit , code ) {
68
- emit({ type: 'error', source: 'shs', code });
105
+ function emitShsError(emit , code , message ) {
106
+ emit({ type: 'error', source: 'shs', code, ...(message ? { message } : {}) });
69
107
  }
70
108
 
71
109
  export async function runParseFromUrl(
@@ -141,55 +179,127 @@ export async function runParseFromUrl(
141
179
  if (total) {
142
180
  emit({ type: 'progress', pct: 0.5, linesProcessed: 0 });
143
181
  }
144
- decodeShsArchive(zipBytes, state, emit);
182
+ await decodeShsArchive(zipBytes, state, emit);
183
+ }
184
+
185
+ // Reads the in-memory SHS download through the same random-access interface a
186
+ // dropped File offers, so both paths share parseZipArchive.
187
+ function bytesZipSource(bytes ) {
188
+ return {
189
+ size: bytes.length,
190
+ slice: (start, end) => ({ arrayBuffer: async () => bytes.slice(start, end).buffer }),
191
+ };
145
192
  }
146
193
 
147
194
  export function decodeShsArchive(
148
195
  zipBytes , state , emit , { zstdDecoder } = {},
149
- ) {
150
- let entries ;
196
+ ) {
197
+ return parseZipArchive(bytesZipSource(zipBytes), state, emit, {
198
+ zstdDecoder,
199
+ onInvalid: (detail) => emitShsError(emit, 'invalid-event-log', detail ?? undefined),
200
+ });
201
+ }
202
+
203
+ const baseName = (e ) => e.name.slice(e.name.lastIndexOf('/') + 1);
204
+ const isRollingPart = (e ) => /^events_\d+_/.test(baseName(e));
205
+
206
+ // The application attempts a History Server zip holds. Without an attempt ID
207
+ // the History Server packs every attempt into one download: each single-file
208
+ // log is its own root entry, each rolling log its own `eventlog_v2_` directory.
209
+ // Directory entries and `appstatus` markers are not attempts.
210
+ function countLogAttempts(entries ) {
211
+ const attempts = new Set ();
212
+ for (const e of entries) {
213
+ if (e.name.endsWith('/') || baseName(e).toLowerCase().startsWith('appstatus')) continue;
214
+ const dir = e.name.slice(0, e.name.indexOf('/') + 1);
215
+ attempts.add(dir || (isRollingPart(e) ? '' : e.name));
216
+ }
217
+ return attempts.size;
218
+ }
219
+
220
+ // The event-log entries a single-attempt History Server zip holds, in parse
221
+ // order. A rolling log's parts sit under an `eventlog_v2_<appId>/` directory
222
+ // entry; they are matched and reassembled by base name. A single-file log is
223
+ // one entry at the root. Directory entries and the rolling log's `appstatus`
224
+ // marker are skipped.
225
+ function selectLogEntries(entries ) {
226
+ const files = entries.filter((e) => !e.name.endsWith('/'));
227
+ if (files.some(isRollingPart)) {
228
+ const byBase = new Map(files.map((e) => [baseName(e), e]));
229
+ return reassembleRollingEntries(files.map(baseName)).map((n) => byBase.get(n) );
230
+ }
231
+ return files.filter((e) => baseName(e).toLowerCase() !== 'appstatus');
232
+ }
233
+
234
+
235
+
236
+
237
+
238
+
239
+
240
+
241
+
242
+
243
+
244
+
245
+
246
+
247
+ // Parses a single-entry or rolling Spark History Server zip, streaming each
248
+ // entry out of `source` in `chunkSize` slices. One NDJSON line decoder spans
249
+ // all entries, since a file-roll boundary need not fall on a line boundary.
250
+ export async function parseZipArchive(
251
+ source ,
252
+ state ,
253
+ emit ,
254
+ { zstdDecoder, chunkSize = 512 * 1024, onInvalid, progressEvery = 2000, reportPct = false } ,
255
+ ) {
256
+ let logEntries ;
257
+ let attempts ;
151
258
  try {
152
- entries = unzipSync(zipBytes);
153
- } catch {
154
- emitShsError(emit, 'invalid-event-log');
259
+ const entries = await listZipEntries(source);
260
+ attempts = countLogAttempts(entries);
261
+ logEntries = attempts > 1 ? [] : selectLogEntries(entries);
262
+ } catch (e) {
263
+ onInvalid(`Could not read the zip archive: ${e instanceof Error ? e.message : String(e)}`);
155
264
  return;
156
265
  }
157
- const allNames = Object.keys(entries);
158
- const isRolling = allNames.some(n => /^events_\d+_/.test(n));
159
- let names;
160
- if (isRolling) {
161
- try {
162
- names = reassembleRollingEntries(allNames);
163
- } catch {
164
- emitShsError(emit, 'invalid-event-log');
165
- return;
166
- }
167
- } else {
168
- names = allNames.filter(n => n.toLowerCase() !== 'appstatus').sort(naturalCompare);
266
+ if (attempts > 1) {
267
+ onInvalid(`The zip archive holds ${attempts} application attempts. Download a single attempt, for example GET /api/v1/applications/<appId>/<attemptId>/logs.`);
268
+ return;
169
269
  }
170
- if (names.length === 0) {
171
- emitShsError(emit, 'invalid-event-log');
270
+ if (logEntries.length === 0) {
271
+ onInvalid('The zip archive contains no event log.');
172
272
  return;
173
273
  }
174
274
 
275
+ const totalBytes = logEntries.reduce((sum, e) => sum + e.compressedSize, 0);
276
+ let bytesRead = 0;
175
277
  const decoder = buildChunkDecoder();
176
278
  const joined = [];
177
279
  let linesProcessed = 0;
178
- for (const name of names) {
280
+ const feed = (bytes ) => {
281
+ joined.length = 0;
282
+ const lines = decoder.decode(bytes, joined);
283
+ for (let i = 0, j = 0; i < lines.length; i++) {
284
+ dispatchLine(lines[i], state, emit, joined[j]?.index === i ? joined[j++] : undefined);
285
+ linesProcessed++;
286
+ if (linesProcessed % progressEvery === 0) {
287
+ emit({ type: 'progress', pct: reportPct && totalBytes ? bytesRead / totalBytes : null, linesProcessed });
288
+ }
289
+ }
290
+ };
291
+
292
+ for (const entry of logEntries) {
293
+ const entryDecoder = createEntryDecoder(entry.name, feed, zstdDecoder);
179
294
  try {
180
- decodeEntry(name, entries[name], (bytes) => {
181
- joined.length = 0;
182
- const lines = decoder.decode(bytes, joined);
183
- for (let i = 0, j = 0; i < lines.length; i++) {
184
- dispatchLine(lines[i], state, emit, joined[j]?.index === i ? joined[j++] : undefined);
185
- linesProcessed++;
186
- if (linesProcessed % 2000 === 0) {
187
- emit({ type: 'progress', pct: null, linesProcessed });
188
- }
189
- }
190
- }, zstdDecoder);
191
- } catch {
192
- emitShsError(emit, 'invalid-event-log');
295
+ await streamZipEntry(source, entry, (chunk) => entryDecoder.push(chunk), {
296
+ chunkSize,
297
+ onRead: (bytes) => { bytesRead += bytes; },
298
+ });
299
+ await entryDecoder.end();
300
+ } catch (e) {
301
+ entryDecoder.cancel();
302
+ onInvalid(`Could not decompress "${entry.name}" in the zip archive: ${e instanceof Error ? e.message : String(e)}`);
193
303
  return;
194
304
  }
195
305
  }
@@ -198,7 +308,7 @@ export function decodeShsArchive(
198
308
  }
199
309
 
200
310
  if (!state.app) {
201
- emitShsError(emit, 'invalid-event-log');
311
+ onInvalid(null);
202
312
  return;
203
313
  }
204
314
  emitParseCompletion(state, emit, linesProcessed);
@@ -19,7 +19,7 @@ export const DEFAULT_IDLE_TIMEOUT_MS = envInt('SPARKFORENSICS_SHS_TIMEOUT_MS', 3
19
19
 
20
20
  function collectShsAppModel(zipBytes ) {
21
21
  return collectViaDispatch(
22
- (state, emit) => decodeShsArchive(zipBytes, state, emit, nodeArchiveCodecs),
22
+ (state, emit, reject) => { decodeShsArchive(zipBytes, state, emit, nodeArchiveCodecs).catch(reject); },
23
23
  (msg) => {
24
24
  const m = msg ;
25
25
  return mcpError(m?.code ?? 'invalid-event-log', m?.message ?? 'Failed to decode SHS archive.');
@@ -0,0 +1,11 @@
1
+ // Shared by every scope:'sql' detector and the plan views. sql.get(id).stageIds is always empty (parser-worker
2
+ // never populates it), so stage linkage is derived from each stage's own sqlExecutionId.
3
+ // Minimal param shape (not DetectorStage): external callers pass Map<StageId, Stage>.
4
+ export function stageIdsForSqlExec(
5
+ executionId ,
6
+ stages ,
7
+ ) {
8
+ const out = [];
9
+ for (const s of stages.values()) if (s.sqlExecutionId === executionId) out.push(s.id);
10
+ return out;
11
+ }
@@ -1,4 +1,5 @@
1
1
 
2
+
2
3
 
3
4
  // Single source of truth for the packed per-task numeric array: FIELDS (offset
4
5
  // constants), TASK_FIELD_NAMES (display labels), and the finalizeStage hot-loop
@@ -74,6 +75,8 @@ export function finalizeStage(
74
75
  const executorStats = new Map();
75
76
  const failureReasons = new Map();
76
77
  const failedTaskSamples = [];
78
+ // Keyed by the interned detail object: accumulateTask shares one per distinct failure.
79
+ const failureGroups = new Map ();
77
80
  const localityStats = new Map();
78
81
  let peakExecutionMemoryMax = 0;
79
82
 
@@ -82,6 +85,8 @@ export function finalizeStage(
82
85
  if (t.failed) {
83
86
  failedTasks++;
84
87
  if (t.reason) failureReasons.set(t.reason, (failureReasons.get(t.reason) ?? 0) + 1);
88
+ const failure = (t ).failure;
89
+ if (failure) failureGroups.set(failure, (failureGroups.get(failure) ?? 0) + 1);
85
90
  if (failedTaskSamples.length < MAX_TASK_SAMPLES) {
86
91
  failedTaskSamples.push({
87
92
  taskId: t.taskId, attemptNumber: t.attemptNumber, host: t.host, executorId: t.executorId,
@@ -125,6 +130,7 @@ export function finalizeStage(
125
130
  stage.failedTasks = failedTasks;
126
131
  stage.speculativeTasks = speculativeTasks;
127
132
  stage.taskAttempts = null; // no longer needed after finalize, freeing memory
133
+ stage.failureDetails = null;
128
134
 
129
135
  const arr = new Float64Array(buf);
130
136
  state.taskStore.set(stageId, arr);
@@ -164,6 +170,10 @@ export function finalizeStage(
164
170
  const failureReasonsArr = [...failureReasons.entries()].map(
165
171
  ([reason, count]) => ({ reason, count })
166
172
  );
173
+ // Most frequent first; ties keep first-seen order (Array.prototype.sort is stable).
174
+ const failureGroupsArr = [...failureGroups.entries()]
175
+ .map(([detail, count]) => ({ ...detail, count }))
176
+ .sort((a, b) => b.count - a.count);
167
177
  const localityStatsArr = [...localityStats.entries()].map(
168
178
  ([locality, count]) => ({ locality, count })
169
179
  );
@@ -173,6 +183,7 @@ export function finalizeStage(
173
183
  hostStats: hostStatsArr, executorStats: executorStatsArr, failureReasons: failureReasonsArr, localityStats: localityStatsArr, stragglerCount, stragglerExcessMs, longestNonStragglerMs,
174
184
  tailReplayRecoveryMs,
175
185
  failedTaskSamples,
186
+ failureGroups: failureGroupsArr,
176
187
  peakExecutionMemoryMax,
177
188
  taskActiveMs: computeTaskActiveMs(arr),
178
189
  peakConcurrentTasks,
@@ -187,6 +198,9 @@ export function finalizeStage(
187
198
  stageType: acc.shuffleReadBytes > 0 ? 'REDUCE' : 'MAP',
188
199
  };
189
200
  delete data.taskAttempts; // internal-only field, already nulled above; never part of the public message
201
+ delete data.failureDetails; // internal-only intern table, summarized by failureGroups
202
+ delete data.speculativeWinners; // internal-only late-TaskEnd pairing state, kept worker-side
203
+ delete data.lateSpeculationWaste;
190
204
 
191
205
  return { type: 'stage', data };
192
206
  }
@@ -0,0 +1,151 @@
1
+ // What actually went wrong behind a failed task attempt, read from its TaskEnd "Task End Reason".
2
+ // Spark's `Reason` is only the end-reason tag (ExceptionFailure, ExecutorLostFailure, ...); the
3
+ // real error lives in tag-specific fields (JsonProtocol.taskEndReasonToJson):
4
+ // ExceptionFailure Class Name, Description, Full Stack Trace
5
+ // ExecutorLostFailure Loss Reason (e.g. "Container killed by YARN for exceeding memory limits")
6
+ // FetchFailed Message (often a whole exception string, stack included)
7
+ // TaskKilled Kill Reason
8
+ // Every text field is bounded here, at ingest, so neither parser memory nor a report grows with
9
+ // the size of the traces Spark wrote.
10
+
11
+
12
+
13
+
14
+
15
+
16
+
17
+
18
+
19
+
20
+
21
+
22
+
23
+ const MAX_TEXT_CHARS = 300;
24
+ const MAX_EXCERPT_FRAMES = 8;
25
+ const MAX_EXCERPT_LINE_CHARS = 300;
26
+ const MAX_EXCERPT_CHARS = 2000;
27
+ // Distinct failures kept per stage while parsing, and shown per finding.
28
+ export const MAX_FAILURE_DETAILS_PER_STAGE = 50;
29
+ export const MAX_FAILURE_GROUPS = 5;
30
+
31
+ export const REDACTED_TEXT = '[redacted]';
32
+
33
+ function truncate(s , max ) {
34
+ return s.length > max ? `${s.slice(0, max - 3)}...` : s;
35
+ }
36
+
37
+ function text(v ) {
38
+ return typeof v === 'string' && v.trim().length > 0 ? v : null;
39
+ }
40
+
41
+ // First non-blank line, whitespace collapsed: a message can carry a whole stack trace after it.
42
+ function firstLine(s ) {
43
+ if (s == null) return null;
44
+ const line = s.split('\n').map((l) => l.trim()).find((l) => l.length > 0);
45
+ return line ? truncate(line.replace(/\s+/g, ' '), MAX_TEXT_CHARS) : null;
46
+ }
47
+
48
+ const PYTHON_TRACEBACK_HEAD = /(?:^|: )(?:Traceback \(most recent call last\):|An exception was thrown from the Python worker)/;
49
+ const JAVA_FRAME = /^\s*at \S/;
50
+
51
+ // A PySpark error's text is a whole Python traceback whose own error line (`ValueError: bad row`)
52
+ // comes last, before any JVM frames: index of that line among non-blank `lines`, or -1.
53
+ function pythonErrorLineIndex(lines ) {
54
+ if (lines.length < 2 || !PYTHON_TRACEBACK_HEAD.test(lines[0].trim())) return -1;
55
+ let last = -1;
56
+ for (let i = 1; i < lines.length && !JAVA_FRAME.test(lines[i]); i++) last = i;
57
+ return last;
58
+ }
59
+
60
+ // The message's headline: its first line, or a Python traceback's error line.
61
+ function messageLine(s ) {
62
+ if (s == null) return null;
63
+ const lines = s.split('\n').filter((l) => l.trim().length > 0);
64
+ const pyIdx = pythonErrorLineIndex(lines);
65
+ return pyIdx >= 0 ? firstLine(lines[pyIdx]) : firstLine(s);
66
+ }
67
+
68
+ /** Header line, the first frames, and (when cut off) a Python traceback's error line and the last
69
+ * `Caused by:` line, which usually name the root cause. Bounded by frame count, line length and
70
+ * total length. */
71
+ export function buildStackExcerpt(trace ) {
72
+ if (trace == null) return null;
73
+ const lines = trace.split('\n').map((l) => l.replace(/\s+$/, '')).filter((l) => l.trim().length > 0);
74
+ if (lines.length === 0) return null;
75
+ const headCount = 1 + MAX_EXCERPT_FRAMES;
76
+ const kept = lines.slice(0, headCount);
77
+ if (lines.length > headCount) {
78
+ const pyIdx = pythonErrorLineIndex(lines);
79
+ let causeIdx = -1;
80
+ for (let i = lines.length - 1; i >= headCount; i--) {
81
+ if (lines[i].startsWith('Caused by:')) { causeIdx = i; break; }
82
+ }
83
+ if (pyIdx >= headCount) kept.push('\t...', lines[pyIdx]);
84
+ if (pyIdx < lines.length - 1) kept.push('\t...');
85
+ if (causeIdx >= 0) kept.push(...lines.slice(causeIdx, causeIdx + 2));
86
+ }
87
+ return truncate(kept.map((l) => truncate(l, MAX_EXCERPT_LINE_CHARS)).join('\n'), MAX_EXCERPT_CHARS);
88
+ }
89
+
90
+ /** Bounded failure detail for a failed attempt's end reason, or null when there is no end reason. */
91
+ export function extractTaskFailureDetail(endReason ) {
92
+ if (!endReason) return null;
93
+ const reason = text(endReason['Reason']);
94
+ const rawMessage = text(endReason['Description']) ?? text(endReason['Message']) ?? text(endReason['Kill Reason']);
95
+ // FetchFailed has no Full Stack Trace field, but its Message is a full exception string.
96
+ const trace = text(endReason['Full Stack Trace'])
97
+ ?? (rawMessage != null && rawMessage.trim().includes('\n') ? rawMessage : null);
98
+ return {
99
+ reason,
100
+ className: firstLine(text(endReason['Class Name'])),
101
+ message: messageLine(rawMessage),
102
+ lossReason: firstLine(text(endReason['Loss Reason'])),
103
+ stackExcerpt: buildStackExcerpt(trace),
104
+ };
105
+ }
106
+
107
+ /** Grouping key: one group per distinct error. The excerpt is left out, so two traces of the same
108
+ * error share a group and the first one seen is kept. */
109
+ export function taskFailureKey(d ) {
110
+ return JSON.stringify([d.reason, d.className, d.message, d.lossReason]);
111
+ }
112
+
113
+ /** Short name for the error: the exception class, or the end-reason tag with its loss reason. */
114
+ export function describeTaskFailure(d ) {
115
+ if (d.className) return d.className;
116
+ if (d.lossReason) return d.reason ? `${d.reason}: ${d.lossReason}` : d.lossReason;
117
+ return d.reason;
118
+ }
119
+
120
+ const FRAME_LINE = /^\s*(?:at \S.*|\.\.\.(?: \d+ more)?)$/;
121
+ const CLASS_HEADER = /^((?:Caused by: |Suppressed: )?[\w$]+(?:\.[\w$]+)+)(?::.*)?$/;
122
+
123
+ // Keeps stack frames and exception class names only: a message (the text after "Class: ", and any
124
+ // continuation line such as a Python traceback's `File "/path"` lines) is dropped.
125
+ function stripExcerptMessages(excerpt ) {
126
+ const kept = [];
127
+ for (const line of excerpt.split('\n')) {
128
+ if (FRAME_LINE.test(line)) { kept.push(line); continue; }
129
+ const header = CLASS_HEADER.exec(line.trim());
130
+ if (header) kept.push(header[1]);
131
+ }
132
+ return kept.join('\n');
133
+ }
134
+
135
+ /** Redacted copy of a failure group: the message and the message text inside the stack excerpt
136
+ * can carry file paths and data values. Class names, frames and the loss reason are kept (hosts in
137
+ * a loss reason are pseudonymized by the caller's host scan like any other free text). */
138
+ export function redactTaskFailureGroup (g ) {
139
+ return {
140
+ ...g,
141
+ message: g.message == null ? null : REDACTED_TEXT,
142
+ stackExcerpt: g.stackExcerpt == null ? null : stripExcerptMessages(g.stackExcerpt),
143
+ };
144
+ }
145
+
146
+ /** One-line headline for a failure group, for reports and the Failed Tasks widget. */
147
+ export function formatTaskFailureHeadline(d ) {
148
+ const name = d.className ?? d.reason ?? 'Unknown failure';
149
+ const detail = [d.message, d.lossReason].filter((s) => s != null).join(' · ');
150
+ return detail ? `${name}: ${detail}` : name;
151
+ }