sparkforensics-mcp 0.2.2 → 0.2.4

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (42) hide show
  1. package/bin/sparkforensics-mcp.mjs +11 -5
  2. package/package.json +3 -3
  3. package/vendor-core/cli/collect-run.js +3 -2
  4. package/vendor-core/cli/native-zstd.js +351 -0
  5. package/vendor-core/detectors.js +243 -53
  6. package/vendor-core/docs-config.js +34 -8
  7. package/vendor-core/docs-content/chapters/03-memory-model.md +39 -0
  8. package/vendor-core/docs-content/chapters/11-cluster-config.md +40 -0
  9. package/vendor-core/docs-content/detection/gc.md +2 -0
  10. package/vendor-core/docs-content/detection/host.md +2 -1
  11. package/vendor-core/docs-content/detection/plan.md +3 -1
  12. package/vendor-core/docs-content/detection/shape.md +2 -1
  13. package/vendor-core/docs-content/detection/shfl.md +2 -1
  14. package/vendor-core/docs-content/detection/spill.md +1 -1
  15. package/vendor-core/docs-content/detection/strag.md +2 -1
  16. package/vendor-core/docs-content/detection/tiny.md +2 -1
  17. package/vendor-core/docs-content/tuning/failures.md +1 -1
  18. package/vendor-core/docs-content/tuning/gc.md +11 -4
  19. package/vendor-core/docs-content/tuning/shuffle.md +25 -5
  20. package/vendor-core/docs-content/tuning/skew.md +14 -6
  21. package/vendor-core/docs-content/tuning/small-files.md +12 -7
  22. package/vendor-core/docs-content/tuning/straggler.md +34 -0
  23. package/vendor-core/docs-content/tuning/tiny-tasks.md +1 -1
  24. package/vendor-core/docs-content/tuning/utilization.md +57 -7
  25. package/vendor-core/docs-content/upstream.json +4 -0
  26. package/vendor-core/event-handlers.js +321 -69
  27. package/vendor-core/event-schemas.js +8 -6
  28. package/vendor-core/evidence-report.js +3 -1
  29. package/vendor-core/impact-estimator.js +170 -34
  30. package/vendor-core/mcp-tools.js +20 -7
  31. package/vendor-core/occupancy.js +71 -2
  32. package/vendor-core/parser-worker.js +56 -24
  33. package/vendor-core/plan-summary.js +5 -1
  34. package/vendor-core/run-comparison.js +1 -15
  35. package/vendor-core/shs-fetch.js +18 -7
  36. package/vendor-core/shs-load.js +2 -1
  37. package/vendor-core/stage-quantiles.js +111 -3
  38. package/vendor-core/string-hash.js +15 -0
  39. package/vendor-core/types.js +14 -1
  40. package/vendor-core/vendor/fzstd.js +94 -18
  41. package/vendor-core/zstd-worker-client.js +180 -0
  42. package/vendor-core/zstd-worker.js +103 -0
@@ -4,6 +4,17 @@
4
4
  // Only the streaming Decompress class is used by this project (see
5
5
  // src/parser-worker.js) to inflate Spark event logs written with
6
6
  // spark.io.compression.codec=zstd, one block at a time.
7
+ // Local patches, each marked "Local patch" below: Decompress.push loops over
8
+ // frame boundaries instead of recursing; streaming decodes every block into one
9
+ // reused [window | block] buffer (see the note above Decompress) instead of a
10
+ // fresh window per frame and a fresh buffer per block; the sequence loop moves
11
+ // literal runs, non-overlapping matches and window reads longer than CPW_MIN
12
+ // bytes with copyWithin/set instead of a byte loop. On the largest real log
13
+ // they took fzstd from 11.5s to 2.9s in Chrome (5.7s to 2.8s in Node), output
14
+ // byte-identical on all 14 real logs. On 2400 randomly corrupted streams the
15
+ // output and error matched upstream on every one but a corrupt window size
16
+ // that overflows negative: both throw "Invalid typed array length" at the same
17
+ // byte, with a different number in the message (the dev/fuzz-fzstd.mjs check).
7
18
  // Some numerical data is initialized as -1 even when it doesn't need initialization to help the JIT infer types
8
19
  // aliases for shorter compressed code (most minifers don't do this)
9
20
  var ab = ArrayBuffer, u8 = Uint8Array, u16 = Uint16Array, i16 = Int16Array, u32 = Uint32Array, i32 = Int32Array;
@@ -106,7 +117,9 @@ var rzfh = function (dat, w) {
106
117
  }
107
118
  if (ws > 2145386496)
108
119
  err(1);
109
- var buf = new u8((w == 1 ? (fss || ws) : w ? 0 : ws) + 12);
120
+ // Local patch: the streaming Decompress (no `w`) keeps its window in its own buffer (see
121
+ // the note above Decompress), so none is allocated here.
122
+ var buf = new u8((w == 1 ? (fss || ws) : 0) + 12);
110
123
  buf[0] = 1, buf[4] = 4, buf[8] = 8;
111
124
  return {
112
125
  b: bt + fsb,
@@ -394,6 +407,13 @@ var dhu4 = function (dat, out, hu) {
394
407
  dhu(dat.subarray(bt, bt += dat[4] | (dat[5] << 8)), out.subarray(sz2, sz3), hu);
395
408
  dhu(dat.subarray(bt), out.subarray(sz3), hu);
396
409
  };
410
+ // Local patch: runs longer than this are copied natively, where a forward byte copy and
411
+ // copyWithin (memmove) agree: a match whose source ends before its destination starts (offset >=
412
+ // length), and a literal run whose destination doesn't pass its source (corrupt input can push
413
+ // the output past the literals). An overlapping match repeats its own output, which only the byte
414
+ // loop does. A source range past the end of its buffer (corrupt input) stays a loop, which reads
415
+ // undefined and so writes 0 there, as upstream does. 16 and 32 measured the same, 8 and 64 slower.
416
+ var CPW_MIN = 16;
397
417
  // read Zstandard block
398
418
  var rzb = function (dat, st, out) {
399
419
  var _a;
@@ -410,7 +430,7 @@ var rzb = function (dat, st, out) {
410
430
  st.b = bt + 1;
411
431
  if (out) {
412
432
  fill(out, dat[bt], st.y, st.y += sz);
413
- return out;
433
+ return st.hv == null ? out : out.subarray(st.y - sz, st.y);
414
434
  }
415
435
  return fill(new u8(sz), dat[bt]);
416
436
  }
@@ -421,7 +441,7 @@ var rzb = function (dat, st, out) {
421
441
  if (out) {
422
442
  out.set(dat.subarray(bt, ebt), st.y);
423
443
  st.y += sz;
424
- return out;
444
+ return st.hv == null ? out : out.subarray(st.y - sz, st.y);
425
445
  }
426
446
  return slc(dat, bt, ebt);
427
447
  }
@@ -548,9 +568,12 @@ var rzb = function (dat, st, out) {
548
568
  else
549
569
  off = st.o[0];
550
570
  }
551
- for (var i = 0; i < ll; ++i) {
552
- buf[oubt + i] = buf[spl + i];
553
- }
571
+ if (ll > CPW_MIN && spl + ll <= buf.length && oubt <= spl) // see CPW_MIN
572
+ buf.copyWithin(oubt, spl, spl + ll);
573
+ else
574
+ for (var i = 0; i < ll; ++i) {
575
+ buf[oubt + i] = buf[spl + i];
576
+ }
554
577
  oubt += ll, spl += ll;
555
578
  var stin = oubt - off;
556
579
  if (stin < 0) {
@@ -558,14 +581,34 @@ var rzb = function (dat, st, out) {
558
581
  var bs = st.e + stin;
559
582
  if (len > ml)
560
583
  len = ml;
561
- for (var i = 0; i < len; ++i) {
562
- buf[oubt + i] = st.w[bs + i];
584
+ // Local patch: streaming keeps one window buffer for the whole stream (see
585
+ // Decompress.push), so history the frame hasn't written yet reads as the 0s
586
+ // of upstream's fresh zeroed window.
587
+ if (st.hv != null && stin < -st.hv) {
588
+ var z = Math.min(len, -st.hv - stin);
589
+ fill(buf, 0, oubt, oubt + z);
590
+ oubt += z, ml -= z, len -= z, bs += z;
563
591
  }
592
+ // Local patch: in bounds only; an out-of-range read stays a loop so it
593
+ // still yields 0 (subarray would wrap a negative start).
594
+ if (len > CPW_MIN && bs >= 0 && bs + len <= st.w.length && oubt + len <= buf.length) {
595
+ if (st.w.buffer === buf.buffer)
596
+ st.w.copyWithin(st.e + oubt, bs, bs + len);
597
+ else
598
+ buf.set(st.w.subarray(bs, bs + len), oubt);
599
+ }
600
+ else
601
+ for (var i = 0; i < len; ++i) {
602
+ buf[oubt + i] = st.w[bs + i];
603
+ }
564
604
  oubt += len, ml -= len, stin = 0;
565
605
  }
566
- for (var i = 0; i < ml; ++i) {
567
- buf[oubt + i] = buf[stin + i];
568
- }
606
+ if (ml > CPW_MIN && oubt - stin >= ml)
607
+ buf.copyWithin(oubt, stin, stin + ml);
608
+ else
609
+ for (var i = 0; i < ml; ++i) {
610
+ buf[oubt + i] = buf[stin + i];
611
+ }
569
612
  oubt += ml;
570
613
  }
571
614
  if (oubt != spl) {
@@ -575,12 +618,12 @@ var rzb = function (dat, st, out) {
575
618
  }
576
619
  else
577
620
  oubt = buf.length;
578
- if (out)
621
+ if (out && st.hv == null)
579
622
  st.y += oubt;
580
623
  else
581
- buf = slc(buf, 0, oubt);
624
+ buf = buf.subarray(0, oubt); // local patch: buf is this block's own, no copy needed
582
625
  }
583
- else if (out) {
626
+ else if (out && st.hv == null) {
584
627
  st.y += lss;
585
628
  if (spl) {
586
629
  for (var i = 0; i < lss; ++i) {
@@ -589,7 +632,7 @@ var rzb = function (dat, st, out) {
589
632
  }
590
633
  }
591
634
  else if (spl)
592
- buf = slc(buf, spl);
635
+ buf = buf.subarray(spl); // local patch: as above
593
636
  st.b = ebt;
594
637
  return buf;
595
638
  }
@@ -654,6 +697,15 @@ export function decompress(dat, buf) {
654
697
  }
655
698
  return cct(bufs, ol);
656
699
  }
700
+ // Local patch: the streaming Decompress keeps one buffer, this.p, laid out as [window | block]:
701
+ // its first st.e (window size) bytes hold the frame's latest output, and each block is decoded
702
+ // right after them (st.hv tracks how much of the window this frame has written; older positions
703
+ // read as 0, as upstream's fresh zeroed window). A back-reference into an earlier block is then a
704
+ // copy within that one buffer. The block part is zeroed before each block, so a block sees
705
+ // exactly what upstream's fresh block buffer held. Upstream allocated and zeroed a window per
706
+ // frame, a buffer per block, then copied each block out and shifted the window. A chunk passed
707
+ // to ondata is a view of this.p and is only valid until ondata returns: every caller in this
708
+ // project decodes it at once.
657
709
  /**
658
710
  * Decompressor for Zstandard streamed data
659
711
  */
@@ -737,7 +789,22 @@ var Decompress = /*#__PURE__*/ (function () {
737
789
  else
738
790
  this.z = 0;
739
791
  for (;;) {
740
- var blk = rzb(chunk, this.s);
792
+ // Local patch: decode into this.p (see the note above Decompress), grown to fit
793
+ // the block's declared size, keeping the window.
794
+ var st = this.s;
795
+ if (st.hv == null)
796
+ st.hv = 0;
797
+ var hb = st.b, need = Math.max(st.m, (chunk[hb] >> 3) | (chunk[hb + 1] << 5) | (chunk[hb + 2] << 13));
798
+ var P = this.p;
799
+ if (!P || this.pe != st.e || P.length < st.e + need) {
800
+ var np = new u8(st.e + need);
801
+ if (P && this.pe == st.e)
802
+ np.set(P.subarray(0, st.e));
803
+ P = this.p = np, this.pe = st.e;
804
+ }
805
+ st.w = P, st.y = st.e;
806
+ P.fill(0, st.e, st.e + need);
807
+ var blk = rzb(chunk, st, P);
741
808
  if (!blk) {
742
809
  if (final)
743
810
  err(5);
@@ -748,8 +815,17 @@ var Decompress = /*#__PURE__*/ (function () {
748
815
  }
749
816
  else {
750
817
  this.ondata(blk, false);
751
- cpw(this.s.w, 0, blk.length);
752
- this.s.w.set(blk, this.s.w.length - blk.length);
818
+ // Local patch: upstream's window update, on the window part of this.p. It
819
+ // is skipped after a frame's last block, which nothing reads, except
820
+ // that a block longer than the window (corrupt input only) still throws
821
+ // the RangeError upstream's update does.
822
+ if (!st.l) {
823
+ P.copyWithin(0, blk.length, st.e);
824
+ P.set(blk, st.e - blk.length);
825
+ st.hv = Math.min(st.e, st.hv + blk.length);
826
+ }
827
+ else if (blk.length > st.e)
828
+ P.set(blk, st.e - blk.length);
753
829
  }
754
830
  if (this.s.l) {
755
831
  chunk = chunk.subarray(this.s.b);
@@ -0,0 +1,180 @@
1
+ // Parse-worker side of the zstd decompress worker (zstd-worker.ts). Builds a ZstdDecoderFactory
2
+ // for streamFile whose push() ships each compressed read slice to the decompress worker and
3
+ // returns while the worker decodes it, so the parse worker parses slice N's output while the
4
+ // decompress worker decodes slice N+1.
5
+ //
6
+ // Flow control is a window of input slices: push() resolves once fewer than `maxInFlight`
7
+ // slices are unacknowledged, and a slice is acknowledged only when its `consumed` reply is
8
+ // handled, which comes after all of its decoded chunks (each fed to onChunk from the message
9
+ // handler). So at most `maxInFlight` slices' output is ever queued on the parse worker, and a
10
+ // fast decompressor on a slow parse cannot grow memory without bound. The final push resolves
11
+ // only after every chunk was fed, which is what streamFile's callers need before they flush.
12
+ //
13
+ // Buffers move by transfer both ways: push() takes ownership of the slice it is given.
14
+
15
+
16
+
17
+ // Three 512 KiB slices of a ~20x-compressed log keep about 30 MB of output queued at most,
18
+ // and give the decompress worker a slice of slack while the parse worker reads the next one.
19
+ export const MAX_IN_FLIGHT_SLICES = 3;
20
+
21
+ // The slice of the Worker API this client uses; tests pass a MessagePort adapter.
22
+
23
+
24
+
25
+
26
+
27
+
28
+
29
+
30
+
31
+
32
+
33
+
34
+
35
+
36
+
37
+
38
+
39
+
40
+
41
+ // `spawn` starts the decompress worker; it runs at most once, on the first zstd stream, and the
42
+ // worker then serves every later stream (the files of a rolling log) one at a time. When it
43
+ // cannot start (no nested workers, a blocked script, a crash), each stream is decoded by
44
+ // `fallback` on the calling thread instead, the path used before this worker existed.
45
+ export function createWorkerZstdDecoders(
46
+ spawn ,
47
+ fallback ,
48
+ { maxInFlight = MAX_IN_FLIGHT_SLICES, onFallback } = {},
49
+ ) {
50
+ let port = null;
51
+ let ready = null;
52
+ let active = null;
53
+ let nextId = 0;
54
+
55
+ const giveUp = (reason ) => {
56
+ port?.terminate?.();
57
+ port = null;
58
+ ready = Promise.resolve(false);
59
+ onFallback?.(reason);
60
+ };
61
+
62
+ const startWorker = () => new Promise((resolve) => {
63
+ let candidate ;
64
+ try {
65
+ candidate = spawn();
66
+ } catch (e) {
67
+ giveUp(`could not start the decompress worker: ${e instanceof Error ? e.message : String(e)}`);
68
+ resolve(false);
69
+ return;
70
+ }
71
+ let settled = false;
72
+ const settle = (ok , reason = '') => {
73
+ if (settled) return;
74
+ settled = true;
75
+ if (ok) port = candidate;
76
+ else {
77
+ candidate.terminate?.();
78
+ giveUp(reason);
79
+ }
80
+ resolve(ok);
81
+ };
82
+ candidate.onmessage = ({ data }) => {
83
+ if (data.type === 'ready') settle(true);
84
+ else if (active && 'id' in data && data.id === active.id) active.onReply(data);
85
+ };
86
+ candidate.onerror = (ev) => {
87
+ // Handled here: left alone, a nested worker's error also reaches the parse worker's own
88
+ // global handler and from there the page's "Worker crashed" path.
89
+ ev.preventDefault();
90
+ const message = ev.message || 'unknown error';
91
+ if (!settled) {
92
+ settle(false, `the decompress worker failed to load: ${message}`);
93
+ return;
94
+ }
95
+ const stream = active;
96
+ giveUp(`the decompress worker crashed: ${message}`);
97
+ stream?.fail(new Error(`Decompress worker crashed: ${message}`));
98
+ };
99
+ });
100
+
101
+ return (onChunk) => {
102
+ const id = nextId++;
103
+ let started = false;
104
+ let local = null;
105
+ let seq = 0;
106
+ let inFlight = 0;
107
+ let failure = null;
108
+ let wake = null;
109
+
110
+ const until = (done ) => new Promise ((resolve) => {
111
+ const check = () => {
112
+ if (!done()) return;
113
+ wake = null;
114
+ resolve();
115
+ };
116
+ wake = check;
117
+ check();
118
+ });
119
+
120
+ const stream = {
121
+ id,
122
+ onReply(msg) {
123
+ if (failure) return;
124
+ if (msg.type === 'chunk') {
125
+ try {
126
+ onChunk(new Uint8Array(msg.bytes, 0, msg.length));
127
+ } catch (e) {
128
+ port?.postMessage({ type: 'cancel', id }, []);
129
+ stream.fail(e instanceof Error ? e : new Error(String(e)));
130
+ }
131
+ } else if (msg.type === 'consumed') {
132
+ inFlight--;
133
+ wake?.();
134
+ } else if (msg.type === 'error') {
135
+ stream.fail(new Error(msg.message));
136
+ }
137
+ },
138
+ fail(err) {
139
+ if (failure) return;
140
+ failure = err;
141
+ if (active === stream) active = null;
142
+ wake?.();
143
+ },
144
+ };
145
+
146
+ const begin = async () => {
147
+ started = true;
148
+ ready ??= startWorker();
149
+ if (!(await ready) || !port) {
150
+ local = fallback(onChunk);
151
+ return;
152
+ }
153
+ active = stream;
154
+ port.postMessage({ type: 'start', id }, []);
155
+ };
156
+
157
+ return {
158
+ async push(chunk, final = false) {
159
+ if (!started) await begin();
160
+ if (local) return local.push(chunk, final);
161
+ if (failure) throw failure;
162
+ // Transfer the slice's own buffer when it spans all of it; copy a view of a larger one.
163
+ const bytes = chunk.byteOffset === 0 && chunk.byteLength === chunk.buffer.byteLength
164
+ ? chunk.buffer
165
+ : chunk.slice().buffer;
166
+ if (!port) throw new Error('Decompress worker is gone');
167
+ port.postMessage({ type: 'data', id, seq: seq++, bytes, final }, [bytes]);
168
+ inFlight++;
169
+ await until(() => failure !== null || inFlight < (final ? 1 : maxInFlight));
170
+ if (failure) throw failure;
171
+ if (final && active === stream) active = null;
172
+ },
173
+ cancel() {
174
+ if (local || !started || failure) return;
175
+ port?.postMessage({ type: 'cancel', id }, []);
176
+ stream.fail(new Error('Decompression cancelled'));
177
+ },
178
+ };
179
+ };
180
+ }
@@ -0,0 +1,103 @@
1
+ // Decompress worker: runs the vendored fzstd off the parse worker's thread, so decompression and
2
+ // NDJSON parsing of a zstd log overlap instead of taking turns. The parse worker spawns it and
3
+ // drives it through zstd-worker-client.ts; both ends of the message protocol live in the types
4
+ // below, and zstd-worker-client.ts documents the flow control.
5
+ import { Decompress as ZstdDecompress } from './vendor/fzstd.js';
6
+
7
+ // Parse worker -> decompress worker. `seq` numbers a stream's input slices from 0.
8
+
9
+
10
+
11
+
12
+
13
+ // Decompress worker -> parse worker. Every `chunk` for input slice `seq` is posted before that
14
+ // slice's `consumed`, and a stream's `error` is the last message it gets.
15
+
16
+
17
+
18
+
19
+
20
+
21
+
22
+
23
+ // fzstd emits one chunk per zstd block (128 KiB at most). Batching them into 1 MiB messages cuts
24
+ // the per-message cost on both threads about eight-fold.
25
+ export const OUTPUT_BATCH_BYTES = 1024 * 1024;
26
+
27
+
28
+
29
+
30
+ // The decompress side of the protocol, transport-free so tests can drive it over a
31
+ // MessageChannel. Holds at most one live stream: `start` replaces it, and a message for any
32
+ // other stream id (one already cancelled or failed) is dropped.
33
+ export function createZstdWorkerHandler(
34
+ post ,
35
+ batchBytes = OUTPUT_BATCH_BYTES,
36
+ ) {
37
+ let streamId = -1;
38
+ let decoder = null;
39
+ let batch = new Uint8Array(batchBytes);
40
+ let batchUsed = 0;
41
+
42
+ // Hand `batch` over without copying it again: transfer its whole buffer with the filled
43
+ // length and start a fresh one. fzstd's chunk is a view of a buffer it reuses, so the copy
44
+ // into `batch` is the one copy this path cannot avoid.
45
+ const flush = () => {
46
+ if (batchUsed === 0) return;
47
+ const bytes = batch.buffer ;
48
+ post({ type: 'chunk', id: streamId, bytes, length: batchUsed }, [bytes]);
49
+ batch = new Uint8Array(batchBytes);
50
+ batchUsed = 0;
51
+ };
52
+
53
+ const collect = (chunk ) => {
54
+ let offset = 0;
55
+ while (offset < chunk.length) {
56
+ const n = Math.min(chunk.length - offset, batchBytes - batchUsed);
57
+ batch.set(chunk.subarray(offset, offset + n), batchUsed);
58
+ batchUsed += n;
59
+ offset += n;
60
+ if (batchUsed === batchBytes) flush();
61
+ }
62
+ };
63
+
64
+ const reset = () => {
65
+ decoder = null;
66
+ batchUsed = 0;
67
+ };
68
+
69
+ return (msg) => {
70
+ if (msg.type === 'start') {
71
+ streamId = msg.id;
72
+ batchUsed = 0;
73
+ decoder = new (ZstdDecompress )(collect);
74
+ return;
75
+ }
76
+ if (msg.id !== streamId || !decoder) return;
77
+ if (msg.type === 'cancel') {
78
+ reset();
79
+ return;
80
+ }
81
+ try {
82
+ decoder.push(new Uint8Array(msg.bytes), msg.final);
83
+ flush();
84
+ } catch (e) {
85
+ reset();
86
+ post({ type: 'error', id: msg.id, message: e instanceof Error ? e.message : String(e) });
87
+ return;
88
+ }
89
+ post({ type: 'consumed', id: msg.id, seq: msg.seq });
90
+ if (msg.final) reset();
91
+ };
92
+ }
93
+
94
+ // ─── Worker message bus (only active when running as a Web Worker) ──────────────
95
+
96
+ const isWorker = typeof WorkerGlobalScope !== 'undefined' && self instanceof WorkerGlobalScope;
97
+
98
+ if (isWorker) {
99
+ const scope = self ;
100
+ const handle = createZstdWorkerHandler((msg, transfer) => scope.postMessage(msg, transfer ?? []));
101
+ scope.onmessage = ({ data } ) => handle(data);
102
+ scope.postMessage({ type: 'ready' } );
103
+ }