sparkforensics-mcp 0.2.2 → 0.2.4
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/bin/sparkforensics-mcp.mjs +11 -5
- package/package.json +3 -3
- package/vendor-core/cli/collect-run.js +3 -2
- package/vendor-core/cli/native-zstd.js +351 -0
- package/vendor-core/detectors.js +243 -53
- package/vendor-core/docs-config.js +34 -8
- package/vendor-core/docs-content/chapters/03-memory-model.md +39 -0
- package/vendor-core/docs-content/chapters/11-cluster-config.md +40 -0
- package/vendor-core/docs-content/detection/gc.md +2 -0
- package/vendor-core/docs-content/detection/host.md +2 -1
- package/vendor-core/docs-content/detection/plan.md +3 -1
- package/vendor-core/docs-content/detection/shape.md +2 -1
- package/vendor-core/docs-content/detection/shfl.md +2 -1
- package/vendor-core/docs-content/detection/spill.md +1 -1
- package/vendor-core/docs-content/detection/strag.md +2 -1
- package/vendor-core/docs-content/detection/tiny.md +2 -1
- package/vendor-core/docs-content/tuning/failures.md +1 -1
- package/vendor-core/docs-content/tuning/gc.md +11 -4
- package/vendor-core/docs-content/tuning/shuffle.md +25 -5
- package/vendor-core/docs-content/tuning/skew.md +14 -6
- package/vendor-core/docs-content/tuning/small-files.md +12 -7
- package/vendor-core/docs-content/tuning/straggler.md +34 -0
- package/vendor-core/docs-content/tuning/tiny-tasks.md +1 -1
- package/vendor-core/docs-content/tuning/utilization.md +57 -7
- package/vendor-core/docs-content/upstream.json +4 -0
- package/vendor-core/event-handlers.js +321 -69
- package/vendor-core/event-schemas.js +8 -6
- package/vendor-core/evidence-report.js +3 -1
- package/vendor-core/impact-estimator.js +170 -34
- package/vendor-core/mcp-tools.js +20 -7
- package/vendor-core/occupancy.js +71 -2
- package/vendor-core/parser-worker.js +56 -24
- package/vendor-core/plan-summary.js +5 -1
- package/vendor-core/run-comparison.js +1 -15
- package/vendor-core/shs-fetch.js +18 -7
- package/vendor-core/shs-load.js +2 -1
- package/vendor-core/stage-quantiles.js +111 -3
- package/vendor-core/string-hash.js +15 -0
- package/vendor-core/types.js +14 -1
- package/vendor-core/vendor/fzstd.js +94 -18
- package/vendor-core/zstd-worker-client.js +180 -0
- package/vendor-core/zstd-worker.js +103 -0
|
@@ -4,6 +4,17 @@
|
|
|
4
4
|
// Only the streaming Decompress class is used by this project (see
|
|
5
5
|
// src/parser-worker.js) to inflate Spark event logs written with
|
|
6
6
|
// spark.io.compression.codec=zstd, one block at a time.
|
|
7
|
+
// Local patches, each marked "Local patch" below: Decompress.push loops over
|
|
8
|
+
// frame boundaries instead of recursing; streaming decodes every block into one
|
|
9
|
+
// reused [window | block] buffer (see the note above Decompress) instead of a
|
|
10
|
+
// fresh window per frame and a fresh buffer per block; the sequence loop moves
|
|
11
|
+
// literal runs, non-overlapping matches and window reads longer than CPW_MIN
|
|
12
|
+
// bytes with copyWithin/set instead of a byte loop. On the largest real log
|
|
13
|
+
// they took fzstd from 11.5s to 2.9s in Chrome (5.7s to 2.8s in Node), output
|
|
14
|
+
// byte-identical on all 14 real logs. On 2400 randomly corrupted streams the
|
|
15
|
+
// output and error matched upstream on every one but a corrupt window size
|
|
16
|
+
// that overflows negative: both throw "Invalid typed array length" at the same
|
|
17
|
+
// byte, with a different number in the message (the dev/fuzz-fzstd.mjs check).
|
|
7
18
|
// Some numerical data is initialized as -1 even when it doesn't need initialization to help the JIT infer types
|
|
8
19
|
// aliases for shorter compressed code (most minifers don't do this)
|
|
9
20
|
var ab = ArrayBuffer, u8 = Uint8Array, u16 = Uint16Array, i16 = Int16Array, u32 = Uint32Array, i32 = Int32Array;
|
|
@@ -106,7 +117,9 @@ var rzfh = function (dat, w) {
|
|
|
106
117
|
}
|
|
107
118
|
if (ws > 2145386496)
|
|
108
119
|
err(1);
|
|
109
|
-
|
|
120
|
+
// Local patch: the streaming Decompress (no `w`) keeps its window in its own buffer (see
|
|
121
|
+
// the note above Decompress), so none is allocated here.
|
|
122
|
+
var buf = new u8((w == 1 ? (fss || ws) : 0) + 12);
|
|
110
123
|
buf[0] = 1, buf[4] = 4, buf[8] = 8;
|
|
111
124
|
return {
|
|
112
125
|
b: bt + fsb,
|
|
@@ -394,6 +407,13 @@ var dhu4 = function (dat, out, hu) {
|
|
|
394
407
|
dhu(dat.subarray(bt, bt += dat[4] | (dat[5] << 8)), out.subarray(sz2, sz3), hu);
|
|
395
408
|
dhu(dat.subarray(bt), out.subarray(sz3), hu);
|
|
396
409
|
};
|
|
410
|
+
// Local patch: runs longer than this are copied natively, where a forward byte copy and
|
|
411
|
+
// copyWithin (memmove) agree: a match whose source ends before its destination starts (offset >=
|
|
412
|
+
// length), and a literal run whose destination doesn't pass its source (corrupt input can push
|
|
413
|
+
// the output past the literals). An overlapping match repeats its own output, which only the byte
|
|
414
|
+
// loop does. A source range past the end of its buffer (corrupt input) stays a loop, which reads
|
|
415
|
+
// undefined and so writes 0 there, as upstream does. 16 and 32 measured the same, 8 and 64 slower.
|
|
416
|
+
var CPW_MIN = 16;
|
|
397
417
|
// read Zstandard block
|
|
398
418
|
var rzb = function (dat, st, out) {
|
|
399
419
|
var _a;
|
|
@@ -410,7 +430,7 @@ var rzb = function (dat, st, out) {
|
|
|
410
430
|
st.b = bt + 1;
|
|
411
431
|
if (out) {
|
|
412
432
|
fill(out, dat[bt], st.y, st.y += sz);
|
|
413
|
-
return out;
|
|
433
|
+
return st.hv == null ? out : out.subarray(st.y - sz, st.y);
|
|
414
434
|
}
|
|
415
435
|
return fill(new u8(sz), dat[bt]);
|
|
416
436
|
}
|
|
@@ -421,7 +441,7 @@ var rzb = function (dat, st, out) {
|
|
|
421
441
|
if (out) {
|
|
422
442
|
out.set(dat.subarray(bt, ebt), st.y);
|
|
423
443
|
st.y += sz;
|
|
424
|
-
return out;
|
|
444
|
+
return st.hv == null ? out : out.subarray(st.y - sz, st.y);
|
|
425
445
|
}
|
|
426
446
|
return slc(dat, bt, ebt);
|
|
427
447
|
}
|
|
@@ -548,9 +568,12 @@ var rzb = function (dat, st, out) {
|
|
|
548
568
|
else
|
|
549
569
|
off = st.o[0];
|
|
550
570
|
}
|
|
551
|
-
|
|
552
|
-
buf
|
|
553
|
-
|
|
571
|
+
if (ll > CPW_MIN && spl + ll <= buf.length && oubt <= spl) // see CPW_MIN
|
|
572
|
+
buf.copyWithin(oubt, spl, spl + ll);
|
|
573
|
+
else
|
|
574
|
+
for (var i = 0; i < ll; ++i) {
|
|
575
|
+
buf[oubt + i] = buf[spl + i];
|
|
576
|
+
}
|
|
554
577
|
oubt += ll, spl += ll;
|
|
555
578
|
var stin = oubt - off;
|
|
556
579
|
if (stin < 0) {
|
|
@@ -558,14 +581,34 @@ var rzb = function (dat, st, out) {
|
|
|
558
581
|
var bs = st.e + stin;
|
|
559
582
|
if (len > ml)
|
|
560
583
|
len = ml;
|
|
561
|
-
|
|
562
|
-
|
|
584
|
+
// Local patch: streaming keeps one window buffer for the whole stream (see
|
|
585
|
+
// Decompress.push), so history the frame hasn't written yet reads as the 0s
|
|
586
|
+
// of upstream's fresh zeroed window.
|
|
587
|
+
if (st.hv != null && stin < -st.hv) {
|
|
588
|
+
var z = Math.min(len, -st.hv - stin);
|
|
589
|
+
fill(buf, 0, oubt, oubt + z);
|
|
590
|
+
oubt += z, ml -= z, len -= z, bs += z;
|
|
563
591
|
}
|
|
592
|
+
// Local patch: in bounds only; an out-of-range read stays a loop so it
|
|
593
|
+
// still yields 0 (subarray would wrap a negative start).
|
|
594
|
+
if (len > CPW_MIN && bs >= 0 && bs + len <= st.w.length && oubt + len <= buf.length) {
|
|
595
|
+
if (st.w.buffer === buf.buffer)
|
|
596
|
+
st.w.copyWithin(st.e + oubt, bs, bs + len);
|
|
597
|
+
else
|
|
598
|
+
buf.set(st.w.subarray(bs, bs + len), oubt);
|
|
599
|
+
}
|
|
600
|
+
else
|
|
601
|
+
for (var i = 0; i < len; ++i) {
|
|
602
|
+
buf[oubt + i] = st.w[bs + i];
|
|
603
|
+
}
|
|
564
604
|
oubt += len, ml -= len, stin = 0;
|
|
565
605
|
}
|
|
566
|
-
|
|
567
|
-
buf
|
|
568
|
-
|
|
606
|
+
if (ml > CPW_MIN && oubt - stin >= ml)
|
|
607
|
+
buf.copyWithin(oubt, stin, stin + ml);
|
|
608
|
+
else
|
|
609
|
+
for (var i = 0; i < ml; ++i) {
|
|
610
|
+
buf[oubt + i] = buf[stin + i];
|
|
611
|
+
}
|
|
569
612
|
oubt += ml;
|
|
570
613
|
}
|
|
571
614
|
if (oubt != spl) {
|
|
@@ -575,12 +618,12 @@ var rzb = function (dat, st, out) {
|
|
|
575
618
|
}
|
|
576
619
|
else
|
|
577
620
|
oubt = buf.length;
|
|
578
|
-
if (out)
|
|
621
|
+
if (out && st.hv == null)
|
|
579
622
|
st.y += oubt;
|
|
580
623
|
else
|
|
581
|
-
buf =
|
|
624
|
+
buf = buf.subarray(0, oubt); // local patch: buf is this block's own, no copy needed
|
|
582
625
|
}
|
|
583
|
-
else if (out) {
|
|
626
|
+
else if (out && st.hv == null) {
|
|
584
627
|
st.y += lss;
|
|
585
628
|
if (spl) {
|
|
586
629
|
for (var i = 0; i < lss; ++i) {
|
|
@@ -589,7 +632,7 @@ var rzb = function (dat, st, out) {
|
|
|
589
632
|
}
|
|
590
633
|
}
|
|
591
634
|
else if (spl)
|
|
592
|
-
buf =
|
|
635
|
+
buf = buf.subarray(spl); // local patch: as above
|
|
593
636
|
st.b = ebt;
|
|
594
637
|
return buf;
|
|
595
638
|
}
|
|
@@ -654,6 +697,15 @@ export function decompress(dat, buf) {
|
|
|
654
697
|
}
|
|
655
698
|
return cct(bufs, ol);
|
|
656
699
|
}
|
|
700
|
+
// Local patch: the streaming Decompress keeps one buffer, this.p, laid out as [window | block]:
|
|
701
|
+
// its first st.e (window size) bytes hold the frame's latest output, and each block is decoded
|
|
702
|
+
// right after them (st.hv tracks how much of the window this frame has written; older positions
|
|
703
|
+
// read as 0, as upstream's fresh zeroed window). A back-reference into an earlier block is then a
|
|
704
|
+
// copy within that one buffer. The block part is zeroed before each block, so a block sees
|
|
705
|
+
// exactly what upstream's fresh block buffer held. Upstream allocated and zeroed a window per
|
|
706
|
+
// frame, a buffer per block, then copied each block out and shifted the window. A chunk passed
|
|
707
|
+
// to ondata is a view of this.p and is only valid until ondata returns: every caller in this
|
|
708
|
+
// project decodes it at once.
|
|
657
709
|
/**
|
|
658
710
|
* Decompressor for Zstandard streamed data
|
|
659
711
|
*/
|
|
@@ -737,7 +789,22 @@ var Decompress = /*#__PURE__*/ (function () {
|
|
|
737
789
|
else
|
|
738
790
|
this.z = 0;
|
|
739
791
|
for (;;) {
|
|
740
|
-
|
|
792
|
+
// Local patch: decode into this.p (see the note above Decompress), grown to fit
|
|
793
|
+
// the block's declared size, keeping the window.
|
|
794
|
+
var st = this.s;
|
|
795
|
+
if (st.hv == null)
|
|
796
|
+
st.hv = 0;
|
|
797
|
+
var hb = st.b, need = Math.max(st.m, (chunk[hb] >> 3) | (chunk[hb + 1] << 5) | (chunk[hb + 2] << 13));
|
|
798
|
+
var P = this.p;
|
|
799
|
+
if (!P || this.pe != st.e || P.length < st.e + need) {
|
|
800
|
+
var np = new u8(st.e + need);
|
|
801
|
+
if (P && this.pe == st.e)
|
|
802
|
+
np.set(P.subarray(0, st.e));
|
|
803
|
+
P = this.p = np, this.pe = st.e;
|
|
804
|
+
}
|
|
805
|
+
st.w = P, st.y = st.e;
|
|
806
|
+
P.fill(0, st.e, st.e + need);
|
|
807
|
+
var blk = rzb(chunk, st, P);
|
|
741
808
|
if (!blk) {
|
|
742
809
|
if (final)
|
|
743
810
|
err(5);
|
|
@@ -748,8 +815,17 @@ var Decompress = /*#__PURE__*/ (function () {
|
|
|
748
815
|
}
|
|
749
816
|
else {
|
|
750
817
|
this.ondata(blk, false);
|
|
751
|
-
|
|
752
|
-
|
|
818
|
+
// Local patch: upstream's window update, on the window part of this.p. It
|
|
819
|
+
// is skipped after a frame's last block, which nothing reads, except
|
|
820
|
+
// that a block longer than the window (corrupt input only) still throws
|
|
821
|
+
// the RangeError upstream's update does.
|
|
822
|
+
if (!st.l) {
|
|
823
|
+
P.copyWithin(0, blk.length, st.e);
|
|
824
|
+
P.set(blk, st.e - blk.length);
|
|
825
|
+
st.hv = Math.min(st.e, st.hv + blk.length);
|
|
826
|
+
}
|
|
827
|
+
else if (blk.length > st.e)
|
|
828
|
+
P.set(blk, st.e - blk.length);
|
|
753
829
|
}
|
|
754
830
|
if (this.s.l) {
|
|
755
831
|
chunk = chunk.subarray(this.s.b);
|
|
@@ -0,0 +1,180 @@
|
|
|
1
|
+
// Parse-worker side of the zstd decompress worker (zstd-worker.ts). Builds a ZstdDecoderFactory
|
|
2
|
+
// for streamFile whose push() ships each compressed read slice to the decompress worker and
|
|
3
|
+
// returns while the worker decodes it, so the parse worker parses slice N's output while the
|
|
4
|
+
// decompress worker decodes slice N+1.
|
|
5
|
+
//
|
|
6
|
+
// Flow control is a window of input slices: push() resolves once fewer than `maxInFlight`
|
|
7
|
+
// slices are unacknowledged, and a slice is acknowledged only when its `consumed` reply is
|
|
8
|
+
// handled, which comes after all of its decoded chunks (each fed to onChunk from the message
|
|
9
|
+
// handler). So at most `maxInFlight` slices' output is ever queued on the parse worker, and a
|
|
10
|
+
// fast decompressor on a slow parse cannot grow memory without bound. The final push resolves
|
|
11
|
+
// only after every chunk was fed, which is what streamFile's callers need before they flush.
|
|
12
|
+
//
|
|
13
|
+
// Buffers move by transfer both ways: push() takes ownership of the slice it is given.
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
// Three 512 KiB slices of a ~20x-compressed log keep about 30 MB of output queued at most,
|
|
18
|
+
// and give the decompress worker a slice of slack while the parse worker reads the next one.
|
|
19
|
+
export const MAX_IN_FLIGHT_SLICES = 3;
|
|
20
|
+
|
|
21
|
+
// The slice of the Worker API this client uses; tests pass a MessagePort adapter.
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
// `spawn` starts the decompress worker; it runs at most once, on the first zstd stream, and the
|
|
42
|
+
// worker then serves every later stream (the files of a rolling log) one at a time. When it
|
|
43
|
+
// cannot start (no nested workers, a blocked script, a crash), each stream is decoded by
|
|
44
|
+
// `fallback` on the calling thread instead, the path used before this worker existed.
|
|
45
|
+
export function createWorkerZstdDecoders(
|
|
46
|
+
spawn ,
|
|
47
|
+
fallback ,
|
|
48
|
+
{ maxInFlight = MAX_IN_FLIGHT_SLICES, onFallback } = {},
|
|
49
|
+
) {
|
|
50
|
+
let port = null;
|
|
51
|
+
let ready = null;
|
|
52
|
+
let active = null;
|
|
53
|
+
let nextId = 0;
|
|
54
|
+
|
|
55
|
+
const giveUp = (reason ) => {
|
|
56
|
+
port?.terminate?.();
|
|
57
|
+
port = null;
|
|
58
|
+
ready = Promise.resolve(false);
|
|
59
|
+
onFallback?.(reason);
|
|
60
|
+
};
|
|
61
|
+
|
|
62
|
+
const startWorker = () => new Promise((resolve) => {
|
|
63
|
+
let candidate ;
|
|
64
|
+
try {
|
|
65
|
+
candidate = spawn();
|
|
66
|
+
} catch (e) {
|
|
67
|
+
giveUp(`could not start the decompress worker: ${e instanceof Error ? e.message : String(e)}`);
|
|
68
|
+
resolve(false);
|
|
69
|
+
return;
|
|
70
|
+
}
|
|
71
|
+
let settled = false;
|
|
72
|
+
const settle = (ok , reason = '') => {
|
|
73
|
+
if (settled) return;
|
|
74
|
+
settled = true;
|
|
75
|
+
if (ok) port = candidate;
|
|
76
|
+
else {
|
|
77
|
+
candidate.terminate?.();
|
|
78
|
+
giveUp(reason);
|
|
79
|
+
}
|
|
80
|
+
resolve(ok);
|
|
81
|
+
};
|
|
82
|
+
candidate.onmessage = ({ data }) => {
|
|
83
|
+
if (data.type === 'ready') settle(true);
|
|
84
|
+
else if (active && 'id' in data && data.id === active.id) active.onReply(data);
|
|
85
|
+
};
|
|
86
|
+
candidate.onerror = (ev) => {
|
|
87
|
+
// Handled here: left alone, a nested worker's error also reaches the parse worker's own
|
|
88
|
+
// global handler and from there the page's "Worker crashed" path.
|
|
89
|
+
ev.preventDefault();
|
|
90
|
+
const message = ev.message || 'unknown error';
|
|
91
|
+
if (!settled) {
|
|
92
|
+
settle(false, `the decompress worker failed to load: ${message}`);
|
|
93
|
+
return;
|
|
94
|
+
}
|
|
95
|
+
const stream = active;
|
|
96
|
+
giveUp(`the decompress worker crashed: ${message}`);
|
|
97
|
+
stream?.fail(new Error(`Decompress worker crashed: ${message}`));
|
|
98
|
+
};
|
|
99
|
+
});
|
|
100
|
+
|
|
101
|
+
return (onChunk) => {
|
|
102
|
+
const id = nextId++;
|
|
103
|
+
let started = false;
|
|
104
|
+
let local = null;
|
|
105
|
+
let seq = 0;
|
|
106
|
+
let inFlight = 0;
|
|
107
|
+
let failure = null;
|
|
108
|
+
let wake = null;
|
|
109
|
+
|
|
110
|
+
const until = (done ) => new Promise ((resolve) => {
|
|
111
|
+
const check = () => {
|
|
112
|
+
if (!done()) return;
|
|
113
|
+
wake = null;
|
|
114
|
+
resolve();
|
|
115
|
+
};
|
|
116
|
+
wake = check;
|
|
117
|
+
check();
|
|
118
|
+
});
|
|
119
|
+
|
|
120
|
+
const stream = {
|
|
121
|
+
id,
|
|
122
|
+
onReply(msg) {
|
|
123
|
+
if (failure) return;
|
|
124
|
+
if (msg.type === 'chunk') {
|
|
125
|
+
try {
|
|
126
|
+
onChunk(new Uint8Array(msg.bytes, 0, msg.length));
|
|
127
|
+
} catch (e) {
|
|
128
|
+
port?.postMessage({ type: 'cancel', id }, []);
|
|
129
|
+
stream.fail(e instanceof Error ? e : new Error(String(e)));
|
|
130
|
+
}
|
|
131
|
+
} else if (msg.type === 'consumed') {
|
|
132
|
+
inFlight--;
|
|
133
|
+
wake?.();
|
|
134
|
+
} else if (msg.type === 'error') {
|
|
135
|
+
stream.fail(new Error(msg.message));
|
|
136
|
+
}
|
|
137
|
+
},
|
|
138
|
+
fail(err) {
|
|
139
|
+
if (failure) return;
|
|
140
|
+
failure = err;
|
|
141
|
+
if (active === stream) active = null;
|
|
142
|
+
wake?.();
|
|
143
|
+
},
|
|
144
|
+
};
|
|
145
|
+
|
|
146
|
+
const begin = async () => {
|
|
147
|
+
started = true;
|
|
148
|
+
ready ??= startWorker();
|
|
149
|
+
if (!(await ready) || !port) {
|
|
150
|
+
local = fallback(onChunk);
|
|
151
|
+
return;
|
|
152
|
+
}
|
|
153
|
+
active = stream;
|
|
154
|
+
port.postMessage({ type: 'start', id }, []);
|
|
155
|
+
};
|
|
156
|
+
|
|
157
|
+
return {
|
|
158
|
+
async push(chunk, final = false) {
|
|
159
|
+
if (!started) await begin();
|
|
160
|
+
if (local) return local.push(chunk, final);
|
|
161
|
+
if (failure) throw failure;
|
|
162
|
+
// Transfer the slice's own buffer when it spans all of it; copy a view of a larger one.
|
|
163
|
+
const bytes = chunk.byteOffset === 0 && chunk.byteLength === chunk.buffer.byteLength
|
|
164
|
+
? chunk.buffer
|
|
165
|
+
: chunk.slice().buffer;
|
|
166
|
+
if (!port) throw new Error('Decompress worker is gone');
|
|
167
|
+
port.postMessage({ type: 'data', id, seq: seq++, bytes, final }, [bytes]);
|
|
168
|
+
inFlight++;
|
|
169
|
+
await until(() => failure !== null || inFlight < (final ? 1 : maxInFlight));
|
|
170
|
+
if (failure) throw failure;
|
|
171
|
+
if (final && active === stream) active = null;
|
|
172
|
+
},
|
|
173
|
+
cancel() {
|
|
174
|
+
if (local || !started || failure) return;
|
|
175
|
+
port?.postMessage({ type: 'cancel', id }, []);
|
|
176
|
+
stream.fail(new Error('Decompression cancelled'));
|
|
177
|
+
},
|
|
178
|
+
};
|
|
179
|
+
};
|
|
180
|
+
}
|
|
@@ -0,0 +1,103 @@
|
|
|
1
|
+
// Decompress worker: runs the vendored fzstd off the parse worker's thread, so decompression and
|
|
2
|
+
// NDJSON parsing of a zstd log overlap instead of taking turns. The parse worker spawns it and
|
|
3
|
+
// drives it through zstd-worker-client.ts; both ends of the message protocol live in the types
|
|
4
|
+
// below, and zstd-worker-client.ts documents the flow control.
|
|
5
|
+
import { Decompress as ZstdDecompress } from './vendor/fzstd.js';
|
|
6
|
+
|
|
7
|
+
// Parse worker -> decompress worker. `seq` numbers a stream's input slices from 0.
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
// Decompress worker -> parse worker. Every `chunk` for input slice `seq` is posted before that
|
|
14
|
+
// slice's `consumed`, and a stream's `error` is the last message it gets.
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
// fzstd emits one chunk per zstd block (128 KiB at most). Batching them into 1 MiB messages cuts
|
|
24
|
+
// the per-message cost on both threads about eight-fold.
|
|
25
|
+
export const OUTPUT_BATCH_BYTES = 1024 * 1024;
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
// The decompress side of the protocol, transport-free so tests can drive it over a
|
|
31
|
+
// MessageChannel. Holds at most one live stream: `start` replaces it, and a message for any
|
|
32
|
+
// other stream id (one already cancelled or failed) is dropped.
|
|
33
|
+
export function createZstdWorkerHandler(
|
|
34
|
+
post ,
|
|
35
|
+
batchBytes = OUTPUT_BATCH_BYTES,
|
|
36
|
+
) {
|
|
37
|
+
let streamId = -1;
|
|
38
|
+
let decoder = null;
|
|
39
|
+
let batch = new Uint8Array(batchBytes);
|
|
40
|
+
let batchUsed = 0;
|
|
41
|
+
|
|
42
|
+
// Hand `batch` over without copying it again: transfer its whole buffer with the filled
|
|
43
|
+
// length and start a fresh one. fzstd's chunk is a view of a buffer it reuses, so the copy
|
|
44
|
+
// into `batch` is the one copy this path cannot avoid.
|
|
45
|
+
const flush = () => {
|
|
46
|
+
if (batchUsed === 0) return;
|
|
47
|
+
const bytes = batch.buffer ;
|
|
48
|
+
post({ type: 'chunk', id: streamId, bytes, length: batchUsed }, [bytes]);
|
|
49
|
+
batch = new Uint8Array(batchBytes);
|
|
50
|
+
batchUsed = 0;
|
|
51
|
+
};
|
|
52
|
+
|
|
53
|
+
const collect = (chunk ) => {
|
|
54
|
+
let offset = 0;
|
|
55
|
+
while (offset < chunk.length) {
|
|
56
|
+
const n = Math.min(chunk.length - offset, batchBytes - batchUsed);
|
|
57
|
+
batch.set(chunk.subarray(offset, offset + n), batchUsed);
|
|
58
|
+
batchUsed += n;
|
|
59
|
+
offset += n;
|
|
60
|
+
if (batchUsed === batchBytes) flush();
|
|
61
|
+
}
|
|
62
|
+
};
|
|
63
|
+
|
|
64
|
+
const reset = () => {
|
|
65
|
+
decoder = null;
|
|
66
|
+
batchUsed = 0;
|
|
67
|
+
};
|
|
68
|
+
|
|
69
|
+
return (msg) => {
|
|
70
|
+
if (msg.type === 'start') {
|
|
71
|
+
streamId = msg.id;
|
|
72
|
+
batchUsed = 0;
|
|
73
|
+
decoder = new (ZstdDecompress )(collect);
|
|
74
|
+
return;
|
|
75
|
+
}
|
|
76
|
+
if (msg.id !== streamId || !decoder) return;
|
|
77
|
+
if (msg.type === 'cancel') {
|
|
78
|
+
reset();
|
|
79
|
+
return;
|
|
80
|
+
}
|
|
81
|
+
try {
|
|
82
|
+
decoder.push(new Uint8Array(msg.bytes), msg.final);
|
|
83
|
+
flush();
|
|
84
|
+
} catch (e) {
|
|
85
|
+
reset();
|
|
86
|
+
post({ type: 'error', id: msg.id, message: e instanceof Error ? e.message : String(e) });
|
|
87
|
+
return;
|
|
88
|
+
}
|
|
89
|
+
post({ type: 'consumed', id: msg.id, seq: msg.seq });
|
|
90
|
+
if (msg.final) reset();
|
|
91
|
+
};
|
|
92
|
+
}
|
|
93
|
+
|
|
94
|
+
// ─── Worker message bus (only active when running as a Web Worker) ──────────────
|
|
95
|
+
|
|
96
|
+
const isWorker = typeof WorkerGlobalScope !== 'undefined' && self instanceof WorkerGlobalScope;
|
|
97
|
+
|
|
98
|
+
if (isWorker) {
|
|
99
|
+
const scope = self ;
|
|
100
|
+
const handle = createZstdWorkerHandler((msg, transfer) => scope.postMessage(msg, transfer ?? []));
|
|
101
|
+
scope.onmessage = ({ data } ) => handle(data);
|
|
102
|
+
scope.postMessage({ type: 'ready' } );
|
|
103
|
+
}
|