sparkforensics-mcp 0.2.0 → 0.2.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (44) hide show
  1. package/package.json +5 -4
  2. package/vendor-core/cli/collect-run.js +3 -2
  3. package/vendor-core/cli/native-zstd.js +351 -0
  4. package/vendor-core/detectors.js +390 -75
  5. package/vendor-core/docs-config.js +34 -8
  6. package/vendor-core/docs-content/chapters/03-memory-model.md +39 -0
  7. package/vendor-core/docs-content/chapters/11-cluster-config.md +40 -0
  8. package/vendor-core/docs-content/detection/cache.md +4 -3
  9. package/vendor-core/docs-content/detection/chrn.md +4 -2
  10. package/vendor-core/docs-content/detection/gc.md +2 -0
  11. package/vendor-core/docs-content/detection/host.md +2 -1
  12. package/vendor-core/docs-content/detection/local.md +2 -3
  13. package/vendor-core/docs-content/detection/mem.md +3 -3
  14. package/vendor-core/docs-content/detection/plan.md +3 -1
  15. package/vendor-core/docs-content/detection/shape.md +2 -1
  16. package/vendor-core/docs-content/detection/shfl.md +2 -1
  17. package/vendor-core/docs-content/detection/spec.md +4 -3
  18. package/vendor-core/docs-content/detection/spill.md +1 -1
  19. package/vendor-core/docs-content/detection/strag.md +2 -1
  20. package/vendor-core/docs-content/detection/tiny.md +2 -1
  21. package/vendor-core/docs-content/tuning/failures.md +1 -1
  22. package/vendor-core/docs-content/tuning/gc.md +11 -4
  23. package/vendor-core/docs-content/tuning/shuffle.md +25 -5
  24. package/vendor-core/docs-content/tuning/skew.md +14 -6
  25. package/vendor-core/docs-content/tuning/small-files.md +12 -7
  26. package/vendor-core/docs-content/tuning/straggler.md +34 -0
  27. package/vendor-core/docs-content/tuning/tiny-tasks.md +1 -1
  28. package/vendor-core/docs-content/tuning/utilization.md +57 -7
  29. package/vendor-core/docs-content/upstream.json +4 -0
  30. package/vendor-core/event-handlers.js +321 -69
  31. package/vendor-core/event-schemas.js +8 -6
  32. package/vendor-core/evidence-report.js +4 -2
  33. package/vendor-core/impact-estimator.js +150 -34
  34. package/vendor-core/mcp-tools.js +20 -7
  35. package/vendor-core/occupancy.js +70 -2
  36. package/vendor-core/parser-worker.js +30 -14
  37. package/vendor-core/plan-summary.js +4 -0
  38. package/vendor-core/run-comparison.js +22 -17
  39. package/vendor-core/shs-fetch.js +18 -7
  40. package/vendor-core/shs-load.js +2 -1
  41. package/vendor-core/stage-quantiles.js +59 -3
  42. package/vendor-core/string-hash.js +15 -0
  43. package/vendor-core/types.js +9 -1
  44. package/vendor-core/vendor/fzstd.js +94 -18
@@ -150,7 +150,7 @@ const MAX_TASK_SAMPLES = 20;
150
150
 
151
151
 
152
152
 
153
-
153
+
154
154
 
155
155
 
156
156
 
@@ -174,81 +174,166 @@ const MAX_TASK_SAMPLES = 20;
174
174
 
175
175
 
176
176
 
177
+
178
+
179
+
180
+
181
+
177
182
 
178
183
 
179
- // Splits decompressed byte chunks into NDJSON lines. Each chunk is decoded whole in one
180
- // TextDecoder.decode call, newlines located by scanning raw bytes (indexOf(0x0A)), and each line
181
- // is a substring of the `pending + text` concatenation.
184
+ // Splits decompressed byte chunks into NDJSON lines. Each chunk is decoded whole in one streaming
185
+ // TextDecoder.decode call (per-line decode was ~8x slower: 227k calls vs ~36k on a 138MB / 3.5GB
186
+ // log), then newlines are found with String.prototype.indexOf on that fresh, flat decoded text.
187
+ // A '\n' char is exactly the 0x0A byte (0x0A never occurs inside a UTF-8 multibyte sequence), and
188
+ // `{ stream: true }` reassembles a character split across a chunk boundary, so no byte-to-UTF-16
189
+ // offset mapping is needed. Only the first line of a chunk joins the carried-over `pending` text.
182
190
  //
183
- // Why (profiled on a 138MB / 3.5GB-decompressed log, decode phase, each line JSON.parsed to force
184
- // materialization): per-line decode (227k calls) ~57s vs whole-chunk decode (~36k calls) ~7s;
185
- // scanning the decoded UTF-16 string for newlines costs 130-450s while scanning raw bytes ~7s;
186
- // the line MUST be a substring of the freshly concatenated string (V8 flattens once per chunk),
187
- // substrings of raw decode() output cost ~4x more. Combined ~14s, byte-identical output.
191
+ // Measured 2026-09-23 against the previous raw-byte-scan version (identical output): 3.5s -> 2.8s
192
+ // on that 3.5GB all-ASCII log, 0.37s -> 0.18s on 93MB of synthetic 2/3/4-byte-heavy NDJSON.
188
193
  //
189
- // 0x0A can't appear inside a UTF-8 multibyte sequence, so byte-scanning for it is UTF-8 safe. A
190
- // newline's byte offset equals its char offset only when the chunk is 1:1 byte<->char, detected by
191
- // text.length === buffer.length (the all-ASCII fast path); otherwise a byte walk recovers char
192
- // offsets. `{ stream: true }` reassembles a character split across the chunk boundary.
193
- const NEWLINE = 0x0a;
194
+ // A SQL UI event's physicalPlanDescription value (see stripPlanDescription) that is still open
195
+ // when a chunk ends is cut off the pending line, and the following bytes are dropped up to its
196
+ // closing quote without being decoded. On the largest real log that value is 1.9 GB of the
197
+ // 3.5 GB stream, so it is never decoded, joined or scanned as text. A chunk longer than
198
+ // MAX_DECODE_SLICE is decoded in slices of that size, so the skip also applies within one chunk:
199
+ // Node's native zstd emits whole frames, up to 39 MB on that log, which held all but 51 MB of the
200
+ // value inside a single chunk.
201
+ const MAX_DECODE_SLICE = 512 * 1024;
202
+
203
+ // A line joined from text decoded in more than one slice is a V8 cons string, which the first
204
+ // character read (startsWith, charCodeAt, endsWith) copies into one flat string. `head` and `tail`
205
+ // are its flat first and last pieces, so dispatchLine can check a prefix or suffix without that
206
+ // copy. `index` is the line's position in the array decode() returned.
207
+
208
+
209
+
210
+
211
+
194
212
 
195
213
  export function buildChunkDecoder() {
196
214
  const decoder = new TextDecoder('utf-8');
197
- // Decoded partial line after the last newline, carried to the next chunk. A character split
198
- // across the boundary is reassembled by the streaming decoder, so this is already-decoded.
215
+ // Decoded partial line after the last newline, carried to the next chunk, and its flat first
216
+ // piece (pending itself until later text is appended to it).
199
217
  let pending = '';
218
+ let pendingHead = '';
219
+ // True once the pending line is known to need no plan-description skip.
220
+ let pendingSettled = false;
221
+ // While it isn't, how far its plan-key search got, so each slice searches only its own text:
222
+ // whether the line starts with the SQL UI event prefix, and the end of the text already searched
223
+ // (one char short of the key, so a key split across slices still matches). Searching the whole
224
+ // line again after every slice scanned about 10 GB on a 100 MB line.
225
+ let pendingIsSqlEvent = false;
226
+ let keyTail = '';
227
+ let skippingPlanDescription = false;
228
+ // Length of the backslash run the skipped bytes ended with, which escapes a quote at the start
229
+ // of the next chunk when odd.
230
+ let carriedBackslashes = 0;
231
+
232
+ // Where the plan key starts in `pending`, searching only `appended` (the text just added to its
233
+ // end) and the seam before it, or -1.
234
+ function findPlanKey(appended ) {
235
+ const seamLength = PLAN_DESCRIPTION_KEY.length - 1;
236
+ const before = pending.length - appended.length;
237
+ if (keyTail !== '') {
238
+ const inSeam = (keyTail + appended.slice(0, seamLength)).indexOf(PLAN_DESCRIPTION_KEY);
239
+ if (inSeam !== -1) return before - keyTail.length + inSeam;
240
+ }
241
+ const inText = appended.indexOf(PLAN_DESCRIPTION_KEY);
242
+ if (inText !== -1) return before + inText;
243
+ keyTail = appended.length >= seamLength ? appended.slice(-seamLength) : (keyTail + appended).slice(-seamLength);
244
+ return -1;
245
+ }
200
246
 
201
- return {
202
- decode(buffer ) {
203
- const lines = [];
204
- const text = decoder.decode(buffer, { stream: true });
205
- const full = pending === '' ? text : pending + text;
206
- const base = pending.length;
207
- const len = buffer.length;
208
- // Char offset in `full` of the current line's start. Zero-length lines
209
- // (`cut === start`) are dropped to match a `.filter(l => l.length)`.
210
- let start = 0;
211
-
212
- if (text.length === len) {
213
- // 1:1 byte<->char: a newline's byte offset is its char offset (+ base).
214
- let nl = buffer.indexOf(NEWLINE, 0);
215
- while (nl !== -1) {
216
- const cut = base + nl;
217
- if (cut > start) lines.push(full.substring(start, cut));
218
- start = cut + 1;
219
- nl = buffer.indexOf(NEWLINE, nl + 1);
220
- }
247
+ function settlePending(lastByte , appended ) {
248
+ if (!pendingIsSqlEvent) {
249
+ if (pending.length < SQL_UI_EVENT_PREFIX.length) {
250
+ pendingSettled = !SQL_UI_EVENT_PREFIX.startsWith(pending);
251
+ return;
252
+ }
253
+ // The flat first piece, when it is long enough: startsWith on the joined line would copy it.
254
+ const head = pendingHead.length >= SQL_UI_EVENT_PREFIX.length ? pendingHead : pending;
255
+ if (!head.startsWith(SQL_UI_EVENT_PREFIX)) { pendingSettled = true; return; }
256
+ pendingIsSqlEvent = true;
257
+ keyTail = '';
258
+ appended = pending; // nothing of it was searched while it was shorter than the prefix
259
+ }
260
+ const keyAt = findPlanKey(appended);
261
+ if (keyAt === -1) return; // the key may still arrive in a later chunk
262
+ pendingSettled = true;
263
+ const valueStart = keyAt + PLAN_DESCRIPTION_KEY.length;
264
+ if (closingQuoteIndex(pending, valueStart) !== -1) return; // complete: stripPlanDescription empties it
265
+ // Only a chunk whose last byte is a backslash carries a run over; any other last byte (such
266
+ // as part of a split multibyte char) ends it.
267
+ let run = 0;
268
+ if (lastByte === BACKSLASH) {
269
+ for (let i = pending.length - 1; i >= valueStart && pending.charCodeAt(i) === BACKSLASH; i--) run++;
270
+ }
271
+ carriedBackslashes = run;
272
+ pending = pending.substring(0, valueStart);
273
+ pendingHead = pending;
274
+ decoder.decode(); // discard a split multibyte char held from the dropped value
275
+ skippingPlanDescription = true;
276
+ }
277
+
278
+ function decodeSlice(buffer , lines , joined ) {
279
+ let from = 0;
280
+ if (skippingPlanDescription) {
281
+ const lineEnd = buffer.indexOf(NEWLINE);
282
+ const close = closingQuoteAt(buffer, lineEnd === -1 ? buffer.length : lineEnd, carriedBackslashes);
283
+ if (close === -1 && lineEnd === -1) {
284
+ let i = buffer.length - 1;
285
+ while (i >= 0 && buffer[i] === BACKSLASH) i--;
286
+ carriedBackslashes = buffer.length - 1 - i + (i < 0 ? carriedBackslashes : 0);
287
+ return;
288
+ }
289
+ // Resume at the closing quote, or at the newline of an unterminated value: that line then
290
+ // ends in an open string and JSON.parse rejects it, as it would have the whole line.
291
+ skippingPlanDescription = false;
292
+ from = close !== -1 ? close : lineEnd;
293
+ }
294
+ const text = decoder.decode(from === 0 ? buffer : buffer.subarray(from), { stream: true });
295
+ let nl = text.indexOf('\n');
296
+ // The text this slice added to the end of `pending`.
297
+ let appended = text;
298
+ if (nl === -1) {
299
+ if (pending === '') {
300
+ pending = pendingHead = text;
301
+ pendingIsSqlEvent = false;
302
+ } else {
303
+ pending = pending + text;
304
+ }
305
+ } else {
306
+ // Zero-length lines are dropped to match a `.filter(l => l.length)`.
307
+ let first ;
308
+ if (pending === '') {
309
+ first = text.substring(0, nl);
221
310
  } else {
222
- // Some byte does not map 1:1. Walk the bytes counting UTF-16 units to
223
- // turn each newline's byte offset into a char offset in `full`.
224
- let unit = base;
225
- let bpos = 0;
226
- // A char whose lead byte was in the previous chunk arrives here as
227
- // leading continuation bytes; the streaming decoder emits it as text[0].
228
- // Count it once (a surrogate pair is two units), then skip its bytes.
229
- if (len > 0 && (buffer[0] & 0xc0) === 0x80) {
230
- const c0 = text.charCodeAt(0);
231
- unit += c0 >= 0xd800 && c0 <= 0xdbff ? 2 : 1;
232
- while (bpos < len && (buffer[bpos] & 0xc0) === 0x80) bpos++;
233
- }
234
- let nl = buffer.indexOf(NEWLINE, bpos);
235
- while (nl !== -1) {
236
- for (let i = bpos; i < nl; i++) {
237
- const b = buffer[i];
238
- if (b < 0x80) unit++; // ASCII
239
- else if (b >= 0xf0) unit += 2; // 4-byte lead -> surrogate pair
240
- else if (b >= 0xc0) unit++; // 2/3-byte lead -> one unit
241
- // continuation byte (0x80..0xBF) -> zero units
242
- }
243
- if (unit > start) lines.push(full.substring(start, unit));
244
- unit++; // the '\n' itself
245
- start = unit;
246
- bpos = nl + 1;
247
- nl = buffer.indexOf(NEWLINE, bpos);
248
- }
311
+ const tail = text.substring(0, nl);
312
+ first = pending + tail;
313
+ joined?.push({ index: lines.length, head: pendingHead, tail });
314
+ }
315
+ if (first.length > 0) lines.push(first);
316
+ let start = nl + 1;
317
+ nl = text.indexOf('\n', start);
318
+ while (nl !== -1) {
319
+ if (nl > start) lines.push(text.substring(start, nl));
320
+ start = nl + 1;
321
+ nl = text.indexOf('\n', start);
249
322
  }
323
+ pending = pendingHead = appended = start < text.length ? text.substring(start) : '';
324
+ pendingSettled = false;
325
+ pendingIsSqlEvent = false;
326
+ }
327
+ if (!pendingSettled && pending !== '') settlePending(buffer[buffer.length - 1], appended);
328
+ }
250
329
 
251
- pending = start < full.length ? full.substring(start) : '';
330
+ return {
331
+ // `joined`, when given, receives a JoinedLine for each returned line that was joined across
332
+ // slices, in line order.
333
+ decode(buffer , joined ) {
334
+ const lines = [];
335
+ if (buffer.length <= MAX_DECODE_SLICE) decodeSlice(buffer, lines, joined);
336
+ else for (let at = 0; at < buffer.length; at += MAX_DECODE_SLICE) decodeSlice(buffer.subarray(at, at + MAX_DECODE_SLICE), lines, joined);
252
337
  return lines;
253
338
  },
254
339
  flush() {
@@ -277,6 +362,8 @@ export function createState() {
277
362
  accumState: new Map(),
278
363
  rddInfo: new Map(),
279
364
  taskAccumStages: new Map(),
365
+ pendingAdaptiveUpdates: new Map(),
366
+ resolvedPlanExecutions: new Set(),
280
367
  evidenceInputs: {
281
368
  environmentUpdates: 0,
282
369
  applicationEnds: 0,
@@ -785,11 +872,12 @@ export function startSqlExecution(event
785
872
  const exec = {
786
873
  id: event.executionId, description: event.description ?? '',
787
874
  startTime: event.time, endTime: null, stageIds: [],
788
- physicalPlanDescription: event.physicalPlanDescription ?? '',
789
875
  sparkPlanInfo,
790
876
  hadAdaptiveUpdate: false,
791
877
  };
792
878
  state.sqlExecutions.set(exec.id, exec);
879
+ // A restarted execution carries a new plan: its next end must resolve it again.
880
+ state.resolvedPlanExecutions.delete(exec.id);
793
881
  if (sparkPlanInfo !== null) {
794
882
  state.accumState.set(exec.id, new Map());
795
883
  }
@@ -802,6 +890,9 @@ export function endSqlExecution(
802
890
  ) {
803
891
  const exec = state.sqlExecutions.get(event.executionId);
804
892
  if (exec) exec.endTime = event.time;
893
+ // A repeated end for an execution whose plan was already posted: its raw plan is gone, and the
894
+ // plain 'sql' copy below would replace the model entry that holds the planTree (onSql overwrites).
895
+ if (state.resolvedPlanExecutions.has(event.executionId)) return null;
805
896
 
806
897
  const planInfo = exec?.sparkPlanInfo ?? null;
807
898
  if (!planInfo || !planInfo.nodeName) {
@@ -814,6 +905,10 @@ export function endSqlExecution(
814
905
  planInfo, accumMap, state.taskAccumStages, state.sqlExecStages.get(event.executionId), event.executionId,
815
906
  );
816
907
  state.accumState.delete(event.executionId);
908
+ // The resolved tree is all anything downstream reads; the raw plan (often megabytes per
909
+ // execution under AQE) would otherwise stay live for the rest of the parse.
910
+ exec .sparkPlanInfo = null;
911
+ state.resolvedPlanExecutions.add(event.executionId);
817
912
 
818
913
  state.evidenceInputs.resolvedSqlPlans++;
819
914
  return { type: 'sqlPlan', data: { executionId: event.executionId, planTree } };
@@ -836,12 +931,13 @@ export function applyAdaptiveExecutionUpdate(
836
931
  const exec = state.sqlExecutions.get(event.executionId);
837
932
  if (!exec) return null; // late update for an unseen execution, discard (same pattern as applyDriverAccumUpdates)
838
933
  if (event.sparkPlanInfo != null) exec.sparkPlanInfo = event.sparkPlanInfo;
839
- if (event.physicalPlanDescription != null) exec.physicalPlanDescription = event.physicalPlanDescription;
840
934
  exec.hadAdaptiveUpdate = true;
841
935
  // Re-emit a 'sql' message so the browser's structured-cloned appModel.sql copy sees the flip:
842
936
  // otherwise hadAdaptiveUpdate only reads true via collectRun's Node-path object aliasing, never
843
937
  // in the shipping worker. Shallow copy so the posted object isn't the mutable reference the worker keeps mutating.
844
- return { type: 'sql', data: { ...exec } };
938
+ // The raw plan stays worker-side: the main thread only ever reads the resolved `sqlPlan` tree,
939
+ // and a copy here would keep a superseded plan alive after endSqlExecution releases it.
940
+ return { type: 'sql', data: { ...exec, sparkPlanInfo: null } };
845
941
  }
846
942
 
847
943
  export function addExecutor(event , state ) {
@@ -903,6 +999,10 @@ export function processEvent(event , state ) {
903
999
  const stage = state.stages.get(id);
904
1000
  if (!stage) return null;
905
1001
  stage.completedAt = info['Completion Time'] ?? 0;
1002
+ // Older Spark (seen on 1.x-2.0 logs) posts StageSubmitted before the stage's submission
1003
+ // time is set; StageCompleted carries it. Without this backfill submittedAt stays 0 and
1004
+ // every stage-duration figure becomes the epoch timestamp itself (a "47-year" stage).
1005
+ if (!stage.submittedAt && info['Submission Time'] != null) stage.submittedAt = info['Submission Time'];
906
1006
  stage.stageFailureReason = info['Failure Reason'] ?? null;
907
1007
  // finalizeStage keeps its `stage` parameter typed as a loose Record (see that module); bridge
908
1008
  // StageRecord's more precise shape across that boundary with an explicit cast.
@@ -947,10 +1047,157 @@ const KNOWN_EVENT_TYPES = new Set(
947
1047
  SparkEventSchema.options.map((option) => option.shape.Event.value)
948
1048
  );
949
1049
 
950
- export function dispatchLine(line , state , emit ) {
1050
+ // SQLExecutionStart and SQLAdaptiveExecutionUpdate carry `physicalPlanDescription`, Spark's text
1051
+ // rendering of the plan. Nothing reads it (the plan tree comes from sparkPlanInfo), yet on a real
1052
+ // 3.5 GB log it was 72% of the AQE-update bytes, which were themselves 73% of the log. Cutting its
1053
+ // string value out before JSON.parse halves the parse cost of those lines.
1054
+ const SQL_UI_EVENT_PREFIX = '{"Event":"org.apache.spark.sql.execution.ui.SparkListenerSQL';
1055
+ const PLAN_DESCRIPTION_KEY = '"physicalPlanDescription":"';
1056
+
1057
+ const QUOTE = 0x22, BACKSLASH = 0x5c, NEWLINE = 0x0a;
1058
+
1059
+ // Index of the closing quote of the JSON string whose content starts at `valueStart`: the first
1060
+ // quote preceded by an even number of backslashes. -1 when the string is unterminated.
1061
+ function closingQuoteIndex(line , valueStart ) {
1062
+ for (let quote = line.indexOf('"', valueStart); quote !== -1; quote = line.indexOf('"', quote + 1)) {
1063
+ let backslashes = 0;
1064
+ for (let i = quote - 1; i >= valueStart && line.charCodeAt(i) === BACKSLASH; i--) backslashes++;
1065
+ if (backslashes % 2 === 0) return quote;
1066
+ }
1067
+ return -1;
1068
+ }
1069
+
1070
+ // Byte-level closingQuoteIndex over a chunk that starts inside the string, stopping at `limit`.
1071
+ // `carried` is the backslash run the previous chunk ended with, which continues into this one.
1072
+ // A quote byte never occurs inside a UTF-8 multibyte sequence.
1073
+ function closingQuoteAt(buf , limit , carried ) {
1074
+ for (let q = buf.indexOf(QUOTE); q !== -1 && q < limit; q = buf.indexOf(QUOTE, q + 1)) {
1075
+ let i = q - 1;
1076
+ while (i >= 0 && buf[i] === BACKSLASH) i--;
1077
+ if ((q - 1 - i + (i < 0 ? carried : 0)) % 2 === 0) return q;
1078
+ }
1079
+ return -1;
1080
+ }
1081
+
1082
+ // Returns `line` with the physicalPlanDescription string value emptied, or `line` unchanged when
1083
+ // the key isn't found in Spark's compact form. The key pattern can't match inside another JSON
1084
+ // string: there its quotes would be backslash-escaped. buildChunkDecoder already empties a value
1085
+ // that spans chunks, so here the value is empty or within one chunk.
1086
+ export function stripPlanDescription(line ) {
1087
+ if (!line.startsWith(SQL_UI_EVENT_PREFIX)) return line;
1088
+ const keyAt = line.indexOf(PLAN_DESCRIPTION_KEY);
1089
+ if (keyAt === -1) return line;
1090
+ const valueStart = keyAt + PLAN_DESCRIPTION_KEY.length;
1091
+ if (line.charCodeAt(valueStart) === QUOTE) return line; // already empty
1092
+ const quote = closingQuoteIndex(line, valueStart);
1093
+ // Unterminated string (a truncated line): leave it for JSON.parse to reject.
1094
+ if (quote === -1) return line;
1095
+ return line.slice(0, valueStart) + line.slice(quote);
1096
+ }
1097
+
1098
+ const TASK_END_PREFIX = '{"Event":"SparkListenerTaskEnd",';
1099
+ const ACCUMULABLES_KEY = '"Accumulables":[';
1100
+ const ACCUMULABLE_ID_KEY = '"ID":';
1101
+ // Beyond 15 digits the digit loop below could round differently from JSON.parse.
1102
+ const MAX_ACCUMULABLE_ID_DIGITS = 15;
1103
+
1104
+ // Parses a TaskEnd line with its Task Info Accumulables array reduced to the `{ID}` entries
1105
+ // accumulateTask reads, or returns null for the caller to parse the line whole. That array is 71%
1106
+ // of TaskEnd bytes on the largest real log, where parsing its TaskEnd lines took 1.3s (0.5s here).
1107
+ // The IDs are read with a string scan and the array is cut out before JSON.parse, but only when
1108
+ // every entry has Spark's flat form (`{"ID":n,...}`, ID first, no nested array): any `{` that
1109
+ // doesn't open `{"ID":n` or any `[` in the array (updatedBlockStatuses) falls back. A `]` inside a
1110
+ // Name string cuts the array short and leaves invalid JSON, which falls back too.
1111
+ export function parseTaskEnd(line ) {
1112
+ if (!line.startsWith(TASK_END_PREFIX)) return null;
1113
+ const keyAt = line.indexOf(ACCUMULABLES_KEY);
1114
+ if (keyAt === -1) return null;
1115
+ const from = keyAt + ACCUMULABLES_KEY.length;
1116
+ const close = line.indexOf(']', from);
1117
+ if (close === -1) return null;
1118
+ const nested = line.indexOf('[', from);
1119
+ if (nested !== -1 && nested < close) return null;
1120
+ const ids = [];
1121
+ for (let brace = line.indexOf('{', from); brace !== -1 && brace < close; brace = line.indexOf('{', brace + 1)) {
1122
+ if (!line.startsWith(ACCUMULABLE_ID_KEY, brace + 1)) return null;
1123
+ const digitsFrom = brace + 1 + ACCUMULABLE_ID_KEY.length;
1124
+ let i = digitsFrom, id = 0;
1125
+ for (let c = line.charCodeAt(i); c >= 48 && c <= 57; c = line.charCodeAt(++i)) id = id * 10 + c - 48;
1126
+ if (i === digitsFrom || i - digitsFrom > MAX_ACCUMULABLE_ID_DIGITS) return null;
1127
+ const after = line.charCodeAt(i);
1128
+ if (after !== 0x2c && after !== 0x7d) return null; // `,` or `}`
1129
+ ids.push(id);
1130
+ }
1131
+ let parsed ;
1132
+ try {
1133
+ parsed = JSON.parse(line.slice(0, from) + line.slice(close));
1134
+ } catch {
1135
+ return null;
1136
+ }
1137
+ const info = parsed?.['Task Info'];
1138
+ if (!info || !Array.isArray(info.Accumulables) || info.Accumulables.length !== 0) return null;
1139
+ info.Accumulables = ids.map((ID) => ({ ID }));
1140
+ return parsed;
1141
+ }
1142
+
1143
+ const ADAPTIVE_UPDATE_PREFIX =
1144
+ '{"Event":"org.apache.spark.sql.execution.ui.SparkListenerSQLAdaptiveExecutionUpdate","executionId":';
1145
+ // More digits than an executionId (a JVM long) has, so a head this long past the prefix holds the
1146
+ // id and the comma after it.
1147
+ const MAX_EXECUTION_ID_DIGITS = 20;
1148
+
1149
+ // An AQE update replaces its execution's whole plan, and nothing reads a plan an update replaced:
1150
+ // only the last one before SQLExecutionEnd is resolved. So while the execution is open, only its
1151
+ // latest update is kept, as unparsed text, and superseded ones are never parsed. On the largest
1152
+ // real log 541 of 664 updates were superseded, 1.1 MB of plan JSON each after the plan text is
1153
+ // cut. Returns false (parse it now, as any other line) unless the line is Spark's compact form
1154
+ // ending in an object-valued `sparkPlanInfo`, its last field: a null-plan update keeps the
1155
+ // previous plan, so it can't supersede one. Trade-off: a malformed superseded update is never
1156
+ // seen, so it no longer counts as a skipped line.
1157
+ //
1158
+ // These updates span decode slices, so `line` is a cons string: the prefix and suffix checks read
1159
+ // `joined`'s flat pieces instead, or a superseded update is copied flat only to be dropped (541
1160
+ // copies of 1.1 MB on that log). A piece too short to hold what is checked falls back to `line`.
1161
+ function deferAdaptiveUpdate(
1162
+ line , state , emit , joined ,
1163
+ ) {
1164
+ const head = joined && joined.head.length > ADAPTIVE_UPDATE_PREFIX.length + MAX_EXECUTION_ID_DIGITS
1165
+ ? joined.head : line;
1166
+ if (!head.startsWith(ADAPTIVE_UPDATE_PREFIX)) return false;
1167
+ let executionId = 0, i = ADAPTIVE_UPDATE_PREFIX.length;
1168
+ for (; i < head.length && head.charCodeAt(i) >= 48 && head.charCodeAt(i) <= 57; i++) {
1169
+ executionId = executionId * 10 + head.charCodeAt(i) - 48;
1170
+ }
1171
+ if (i === ADAPTIVE_UPDATE_PREFIX.length || head.charCodeAt(i) !== 0x2c) return false; // no `N,`
1172
+ const exec = state.sqlExecutions.get(executionId);
1173
+ if (!exec || exec.endTime != null) return false;
1174
+ const tail = joined && joined.tail.length >= 2 ? joined.tail : line;
1175
+ if (!tail.endsWith('}}')) {
1176
+ flushAdaptiveUpdate(executionId, state, emit); // keep this line's order after the pending one
1177
+ return false;
1178
+ }
1179
+ state.pendingAdaptiveUpdates.set(executionId, line);
1180
+ return true;
1181
+ }
1182
+
1183
+ function flushAdaptiveUpdate(executionId , state , emit ) {
1184
+ const line = state.pendingAdaptiveUpdates.get(executionId);
1185
+ if (line === undefined) return;
1186
+ state.pendingAdaptiveUpdates.delete(executionId);
1187
+ parseAndDispatch(line, state, emit);
1188
+ }
1189
+
1190
+ export function dispatchLine(
1191
+ line , state , emit , joined ,
1192
+ ) {
1193
+ if (deferAdaptiveUpdate(line, state, emit, joined)) return;
1194
+ parseAndDispatch(line, state, emit);
1195
+ }
1196
+
1197
+ function parseAndDispatch(line , state , emit ) {
951
1198
  let parsed ;
952
1199
  try {
953
- parsed = JSON.parse(line);
1200
+ parsed = parseTaskEnd(line) ?? JSON.parse(stripPlanDescription(line));
954
1201
  } catch {
955
1202
  state.skippedLines++;
956
1203
  return;
@@ -967,6 +1214,9 @@ export function dispatchLine(line , state , emit
967
1214
  state.skippedLines++;
968
1215
  return;
969
1216
  }
1217
+ if (result.data.Event === 'org.apache.spark.sql.execution.ui.SparkListenerSQLExecutionEnd') {
1218
+ flushAdaptiveUpdate(result.data.executionId, state, emit);
1219
+ }
970
1220
  let msg ;
971
1221
  try {
972
1222
  msg = processEvent(result.data, state);
@@ -995,6 +1245,8 @@ export function collectStageExecutorMetrics(state )
995
1245
  }
996
1246
 
997
1247
  export function emitParseCompletion(state , emit , linesProcessed ) {
1248
+ // Executions that never ended keep their latest AQE update, as they did before it was deferred.
1249
+ for (const executionId of [...state.pendingAdaptiveUpdates.keys()]) flushAdaptiveUpdate(executionId, state, emit);
998
1250
  emit({ type: 'progress', pct: 1, linesProcessed });
999
1251
  emit({ type: 'runAggregates', data: computeRunAggregates(state.taskStore) });
1000
1252
  emit({ type: 'stageExecutorMetrics', data: collectStageExecutorMetrics(state) });
@@ -214,6 +214,8 @@ export const StageCompletedEventSchema = z.object({
214
214
  Event: z.literal('SparkListenerStageCompleted'),
215
215
  'Stage Info': z.object({
216
216
  'Stage ID': z.number(),
217
+ // Backfills submittedAt when StageSubmitted lacked it (older Spark).
218
+ 'Submission Time': z.number().optional(),
217
219
  'Completion Time': z.number().optional(),
218
220
  'Failure Reason': z.string().optional(),
219
221
  }),
@@ -253,11 +255,12 @@ export const TaskEndEventSchema = z.object({
253
255
  'Task ID': z.number().optional(),
254
256
  'Attempt Number': z.number().optional(),
255
257
  // Bounded well above any real per-task metric count: caps how much a crafted TaskEnd can grow
256
- // taskAccumStages, which is never pruned for the life of the parse.
258
+ // taskAccumStages, which is never pruned for the life of the parse. Only ID is read. Update and
259
+ // Value stay undeclared: Spark writes some as JSON arrays (internal.metrics.updatedBlockStatuses),
260
+ // and a string|number check on them rejected the whole TaskEnd, dropping the task from its
261
+ // stage's stats. They were also ~15% of TaskEnd validation time.
257
262
  Accumulables: z.array(z.object({
258
263
  ID: z.number(),
259
- Update: z.union([z.string(), z.number()]).optional(),
260
- Value: z.union([z.string(), z.number()]).optional(),
261
264
  })).max(MAX_ACCUMULABLES_PER_TASK).optional(),
262
265
  }).optional(),
263
266
  'Task Metrics': z.object({
@@ -290,16 +293,15 @@ export const SqlExecutionStartEventSchema = z.object({
290
293
  executionId: z.number(),
291
294
  description: z.string().optional(),
292
295
  time: z.number(),
293
- physicalPlanDescription: z.string().optional(),
294
296
  sparkPlanInfo: SparkPlanInfoFieldSchema,
295
297
  });
296
298
 
297
299
  // applyAdaptiveExecutionUpdate: AQE re-plans mid-query and re-emits sparkPlanInfo for the same
298
- // executionId; last write wins.
300
+ // executionId; last write wins. Neither SQL schema declares physicalPlanDescription: nothing reads
301
+ // it, so zod strips it (dispatchLine already empties it before JSON.parse, see stripPlanDescription).
299
302
  export const SqlAdaptiveExecutionUpdateEventSchema = z.object({
300
303
  Event: z.literal('org.apache.spark.sql.execution.ui.SparkListenerSQLAdaptiveExecutionUpdate'),
301
304
  executionId: z.number(),
302
- physicalPlanDescription: z.string().optional(),
303
305
  sparkPlanInfo: SparkPlanInfoFieldSchema,
304
306
  });
305
307
 
@@ -93,7 +93,9 @@ const NON_EVIDENCE_KEYS = new Set(['planNodeIds']);
93
93
  function findingRow(f ) {
94
94
  const evidence = {};
95
95
  for (const k of Object.keys(f).sort()) {
96
- if (!CORE_KEYS.has(k) && !NON_EVIDENCE_KEYS.has(k)) evidence[k] = (f )[k];
96
+ // An undefined field (spill's spillMagnitude without a magnitude) is absent, as in the JSON.
97
+ const v = (f )[k];
98
+ if (!CORE_KEYS.has(k) && !NON_EVIDENCE_KEYS.has(k) && v !== undefined) evidence[k] = v;
97
99
  }
98
100
  const row = {
99
101
  id: f.id ?? null,
@@ -269,7 +271,7 @@ function renderEvidenceValue(key , value ) {
269
271
 
270
272
  function formatWallClockRange(low , high ) {
271
273
  const fmtMs = (ms ) => (ms === 0 ? '0s' : formatDuration(ms));
272
- return low === high ? `Est. ${fmtMs(high)}` : `Est. ${fmtMs(low)}-${fmtMs(high)}`;
274
+ return low === high ? `Estimated ${fmtMs(high)}` : `Estimated ${fmtMs(low)}-${fmtMs(high)}`;
273
275
  }
274
276
 
275
277
  function formatRawWaste(rawWaste ) {