sparkforensics-mcp 0.2.3 → 0.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -0
- package/README.md +7 -1
- package/bin/sparkforensics-mcp.mjs +11 -5
- package/package.json +3 -3
- package/vendor-core/cli/native-zstd.js +2 -2
- package/vendor-core/detectors.js +16 -2
- package/vendor-core/docs-content/detection/fail.md +6 -2
- package/vendor-core/event-handlers.js +21 -0
- package/vendor-core/event-schemas.js +8 -0
- package/vendor-core/evidence-report.js +20 -1
- package/vendor-core/impact-estimator.js +23 -3
- package/vendor-core/occupancy.js +8 -7
- package/vendor-core/parser-worker.js +51 -18
- package/vendor-core/plan-summary.js +1 -1
- package/vendor-core/redact.js +24 -2
- package/vendor-core/run-comparison.js +8 -0
- package/vendor-core/shs-fetch.js +175 -65
- package/vendor-core/shs-load.js +1 -1
- package/vendor-core/stage-quantiles.js +65 -1
- package/vendor-core/task-failure.js +151 -0
- package/vendor-core/types.js +5 -0
- package/vendor-core/vendor/fflate.js +1 -1
- package/vendor-core/zip-archive.js +167 -0
- package/vendor-core/zstd-worker-client.js +180 -0
- package/vendor-core/zstd-worker.js +103 -0
|
@@ -0,0 +1,167 @@
|
|
|
1
|
+
// Random-access zip reader for Spark History Server log archives. Reads the
|
|
2
|
+
// central directory from the archive's tail, then inflates one entry at a time
|
|
3
|
+
// in bounded slices through fflate's streaming UnzipInflate: the source is read
|
|
4
|
+
// a slice at a time and no decompressed entry is ever held whole in memory.
|
|
5
|
+
//
|
|
6
|
+
// Why the central directory and not fflate's forward-scanning `Unzip`: the
|
|
7
|
+
// History Server writes its zip with Java's ZipOutputStream, which leaves
|
|
8
|
+
// every local header's sizes blank and puts them in a data descriptor after
|
|
9
|
+
// the entry. `Unzip` then has to find each entry's end by scanning the
|
|
10
|
+
// compressed bytes for the descriptor signature, and it hands rolling-log
|
|
11
|
+
// parts over in archive order, not the order they must be parsed in. The
|
|
12
|
+
// central directory has the real sizes and lets the caller pick the order.
|
|
13
|
+
import { UnzipInflate, strFromU8 } from './vendor/fflate.js';
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
const LOCAL_HEADER_SIG = 0x04034b50;
|
|
29
|
+
const CENTRAL_HEADER_SIG = 0x02014b50;
|
|
30
|
+
const EOCD_SIG = 0x06054b50;
|
|
31
|
+
const ZIP64_EOCD_LOCATOR_SIG = 0x07064b50;
|
|
32
|
+
const ZIP64_EOCD_SIG = 0x06064b50;
|
|
33
|
+
const ZIP64_EXTRA_ID = 0x0001;
|
|
34
|
+
const EOCD_MIN_SIZE = 22;
|
|
35
|
+
// The end-of-central-directory record ends with a comment of at most 65535 bytes.
|
|
36
|
+
const EOCD_MAX_SEARCH = EOCD_MIN_SIZE + 0xffff;
|
|
37
|
+
const UINT32_MAX = 0xffffffff;
|
|
38
|
+
|
|
39
|
+
// fflate's UnzipInflate is untyped vendor JS; this local shape types the call site.
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
const u16 = (d , i ) => d[i] | (d[i + 1] << 8);
|
|
47
|
+
const u32 = (d , i ) => (d[i] | (d[i + 1] << 8) | (d[i + 2] << 16) | (d[i + 3] << 24)) >>> 0;
|
|
48
|
+
const u64 = (d , i ) => u32(d, i) + u32(d, i + 4) * 2 ** 32;
|
|
49
|
+
|
|
50
|
+
async function readRange(source , start , end ) {
|
|
51
|
+
if (start < 0 || end > source.size || start > end) throw new Error('zip structure points outside the file');
|
|
52
|
+
return new Uint8Array(await source.slice(start, end).arrayBuffer());
|
|
53
|
+
}
|
|
54
|
+
|
|
55
|
+
/** True when `header` starts with a zip local-file header or an empty archive's EOCD record. */
|
|
56
|
+
export function isZip(header ) {
|
|
57
|
+
if (header.length < 4) return false;
|
|
58
|
+
const sig = u32(header, 0);
|
|
59
|
+
return sig === LOCAL_HEADER_SIG || sig === EOCD_SIG;
|
|
60
|
+
}
|
|
61
|
+
|
|
62
|
+
// Locates the central directory via the EOCD record (or its zip64 variant).
|
|
63
|
+
async function readDirectoryLocation(source ) {
|
|
64
|
+
const tailStart = Math.max(0, source.size - EOCD_MAX_SEARCH);
|
|
65
|
+
const tail = await readRange(source, tailStart, source.size);
|
|
66
|
+
let eocd = tail.length - EOCD_MIN_SIZE;
|
|
67
|
+
while (eocd >= 0 && u32(tail, eocd) !== EOCD_SIG) eocd--;
|
|
68
|
+
if (eocd < 0) throw new Error('no end-of-central-directory record');
|
|
69
|
+
|
|
70
|
+
const location = { count: u16(tail, eocd + 10), size: u32(tail, eocd + 12), offset: u32(tail, eocd + 16) };
|
|
71
|
+
const locator = tailStart + eocd - 20;
|
|
72
|
+
if (locator >= 0) {
|
|
73
|
+
const loc = await readRange(source, locator, locator + 20);
|
|
74
|
+
if (u32(loc, 0) === ZIP64_EOCD_LOCATOR_SIG) {
|
|
75
|
+
const zip64Offset = u64(loc, 8);
|
|
76
|
+
const z = await readRange(source, zip64Offset, zip64Offset + 56);
|
|
77
|
+
if (u32(z, 0) !== ZIP64_EOCD_SIG) throw new Error('bad zip64 end-of-central-directory record');
|
|
78
|
+
return { count: u64(z, 32), size: u64(z, 40), offset: u64(z, 48) };
|
|
79
|
+
}
|
|
80
|
+
}
|
|
81
|
+
return location;
|
|
82
|
+
}
|
|
83
|
+
|
|
84
|
+
// Reads the zip64 extended-information extra field, which carries (in this
|
|
85
|
+
// order) only the sizes/offset whose 32-bit central-directory field is 0xFFFFFFFF.
|
|
86
|
+
function applyZip64Extra(dir , extraStart , extraEnd , fields ) {
|
|
87
|
+
for (let p = extraStart; p + 4 <= extraEnd;) {
|
|
88
|
+
const id = u16(dir, p), len = u16(dir, p + 2);
|
|
89
|
+
if (id === ZIP64_EXTRA_ID) {
|
|
90
|
+
let q = p + 4;
|
|
91
|
+
for (const key of ['uncompressed', 'compressed', 'offset'] ) {
|
|
92
|
+
if (fields[key] === UINT32_MAX && q + 8 <= p + 4 + len) { fields[key] = u64(dir, q); q += 8; }
|
|
93
|
+
}
|
|
94
|
+
return;
|
|
95
|
+
}
|
|
96
|
+
p += 4 + len;
|
|
97
|
+
}
|
|
98
|
+
}
|
|
99
|
+
|
|
100
|
+
/** Lists every entry in the archive's central directory, in directory order. */
|
|
101
|
+
export async function listZipEntries(source ) {
|
|
102
|
+
const { count, offset, size } = await readDirectoryLocation(source);
|
|
103
|
+
const dir = await readRange(source, offset, offset + size);
|
|
104
|
+
const entries = [];
|
|
105
|
+
let p = 0;
|
|
106
|
+
for (let i = 0; i < count; i++) {
|
|
107
|
+
if (p + 46 > dir.length || u32(dir, p) !== CENTRAL_HEADER_SIG) throw new Error('bad central-directory header');
|
|
108
|
+
const nameLength = u16(dir, p + 28), extraLength = u16(dir, p + 30), commentLength = u16(dir, p + 32);
|
|
109
|
+
const utf8 = (u16(dir, p + 8) & 0x800) !== 0;
|
|
110
|
+
const nameStart = p + 46, extraStart = nameStart + nameLength;
|
|
111
|
+
const fields = { uncompressed: u32(dir, p + 24), compressed: u32(dir, p + 20), offset: u32(dir, p + 42) };
|
|
112
|
+
applyZip64Extra(dir, extraStart, extraStart + extraLength, fields);
|
|
113
|
+
entries.push({
|
|
114
|
+
name: strFromU8(dir.subarray(nameStart, extraStart), !utf8),
|
|
115
|
+
compression: u16(dir, p + 10),
|
|
116
|
+
compressedSize: fields.compressed,
|
|
117
|
+
localHeaderOffset: fields.offset,
|
|
118
|
+
});
|
|
119
|
+
p = extraStart + extraLength + commentLength;
|
|
120
|
+
}
|
|
121
|
+
return entries;
|
|
122
|
+
}
|
|
123
|
+
|
|
124
|
+
/**
|
|
125
|
+
* Streams one entry's decompressed bytes to `onChunk`, reading `chunkSize`
|
|
126
|
+
* compressed bytes at a time and awaiting `onChunk` before reading more, so an
|
|
127
|
+
* async consumer (an off-thread zstd decoder) applies backpressure. `onRead`
|
|
128
|
+
* reports compressed bytes consumed, for progress.
|
|
129
|
+
*/
|
|
130
|
+
export async function streamZipEntry(
|
|
131
|
+
source ,
|
|
132
|
+
entry ,
|
|
133
|
+
onChunk ,
|
|
134
|
+
{ chunkSize, onRead } ,
|
|
135
|
+
) {
|
|
136
|
+
const header = await readRange(source, entry.localHeaderOffset, entry.localHeaderOffset + 30);
|
|
137
|
+
if (u32(header, 0) !== LOCAL_HEADER_SIG) throw new Error(`bad local header for "${entry.name}"`);
|
|
138
|
+
const dataStart = entry.localHeaderOffset + 30 + u16(header, 26) + u16(header, 28);
|
|
139
|
+
const dataEnd = dataStart + entry.compressedSize;
|
|
140
|
+
if (dataEnd > source.size) throw new Error(`"${entry.name}" is truncated`);
|
|
141
|
+
if (entry.compression !== 0 && entry.compression !== 8) {
|
|
142
|
+
throw new Error(`"${entry.name}" uses unsupported zip compression method ${entry.compression}`);
|
|
143
|
+
}
|
|
144
|
+
|
|
145
|
+
const pending = [];
|
|
146
|
+
let inflateError = null;
|
|
147
|
+
const inflater = entry.compression === 8 ? new (UnzipInflate )() : null;
|
|
148
|
+
if (inflater) {
|
|
149
|
+
inflater.ondata = (err, data) => {
|
|
150
|
+
if (err) inflateError = err;
|
|
151
|
+
else if (data.length) pending.push(data);
|
|
152
|
+
};
|
|
153
|
+
}
|
|
154
|
+
|
|
155
|
+
for (let offset = dataStart; offset < dataEnd;) {
|
|
156
|
+
const end = Math.min(dataEnd, offset + chunkSize);
|
|
157
|
+
const slice = await readRange(source, offset, end);
|
|
158
|
+
offset = end;
|
|
159
|
+
if (inflater) inflater.push(slice, offset >= dataEnd);
|
|
160
|
+
else pending.push(slice);
|
|
161
|
+
if (inflateError) throw inflateError;
|
|
162
|
+
onRead?.(slice.length);
|
|
163
|
+
// Inflate output chunks are views that fflate may reuse on the next push,
|
|
164
|
+
// so each is fully consumed here before the loop reads further.
|
|
165
|
+
while (pending.length) await onChunk(pending.shift() );
|
|
166
|
+
}
|
|
167
|
+
}
|
|
@@ -0,0 +1,180 @@
|
|
|
1
|
+
// Parse-worker side of the zstd decompress worker (zstd-worker.ts). Builds a ZstdDecoderFactory
|
|
2
|
+
// for streamFile whose push() ships each compressed read slice to the decompress worker and
|
|
3
|
+
// returns while the worker decodes it, so the parse worker parses slice N's output while the
|
|
4
|
+
// decompress worker decodes slice N+1.
|
|
5
|
+
//
|
|
6
|
+
// Flow control is a window of input slices: push() resolves once fewer than `maxInFlight`
|
|
7
|
+
// slices are unacknowledged, and a slice is acknowledged only when its `consumed` reply is
|
|
8
|
+
// handled, which comes after all of its decoded chunks (each fed to onChunk from the message
|
|
9
|
+
// handler). So at most `maxInFlight` slices' output is ever queued on the parse worker, and a
|
|
10
|
+
// fast decompressor on a slow parse cannot grow memory without bound. The final push resolves
|
|
11
|
+
// only after every chunk was fed, which is what streamFile's callers need before they flush.
|
|
12
|
+
//
|
|
13
|
+
// Buffers move by transfer both ways: push() takes ownership of the slice it is given.
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
// Three 512 KiB slices of a ~20x-compressed log keep about 30 MB of output queued at most,
|
|
18
|
+
// and give the decompress worker a slice of slack while the parse worker reads the next one.
|
|
19
|
+
export const MAX_IN_FLIGHT_SLICES = 3;
|
|
20
|
+
|
|
21
|
+
// The slice of the Worker API this client uses; tests pass a MessagePort adapter.
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
// `spawn` starts the decompress worker; it runs at most once, on the first zstd stream, and the
|
|
42
|
+
// worker then serves every later stream (the files of a rolling log) one at a time. When it
|
|
43
|
+
// cannot start (no nested workers, a blocked script, a crash), each stream is decoded by
|
|
44
|
+
// `fallback` on the calling thread instead, the path used before this worker existed.
|
|
45
|
+
export function createWorkerZstdDecoders(
|
|
46
|
+
spawn ,
|
|
47
|
+
fallback ,
|
|
48
|
+
{ maxInFlight = MAX_IN_FLIGHT_SLICES, onFallback } = {},
|
|
49
|
+
) {
|
|
50
|
+
let port = null;
|
|
51
|
+
let ready = null;
|
|
52
|
+
let active = null;
|
|
53
|
+
let nextId = 0;
|
|
54
|
+
|
|
55
|
+
const giveUp = (reason ) => {
|
|
56
|
+
port?.terminate?.();
|
|
57
|
+
port = null;
|
|
58
|
+
ready = Promise.resolve(false);
|
|
59
|
+
onFallback?.(reason);
|
|
60
|
+
};
|
|
61
|
+
|
|
62
|
+
const startWorker = () => new Promise((resolve) => {
|
|
63
|
+
let candidate ;
|
|
64
|
+
try {
|
|
65
|
+
candidate = spawn();
|
|
66
|
+
} catch (e) {
|
|
67
|
+
giveUp(`could not start the decompress worker: ${e instanceof Error ? e.message : String(e)}`);
|
|
68
|
+
resolve(false);
|
|
69
|
+
return;
|
|
70
|
+
}
|
|
71
|
+
let settled = false;
|
|
72
|
+
const settle = (ok , reason = '') => {
|
|
73
|
+
if (settled) return;
|
|
74
|
+
settled = true;
|
|
75
|
+
if (ok) port = candidate;
|
|
76
|
+
else {
|
|
77
|
+
candidate.terminate?.();
|
|
78
|
+
giveUp(reason);
|
|
79
|
+
}
|
|
80
|
+
resolve(ok);
|
|
81
|
+
};
|
|
82
|
+
candidate.onmessage = ({ data }) => {
|
|
83
|
+
if (data.type === 'ready') settle(true);
|
|
84
|
+
else if (active && 'id' in data && data.id === active.id) active.onReply(data);
|
|
85
|
+
};
|
|
86
|
+
candidate.onerror = (ev) => {
|
|
87
|
+
// Handled here: left alone, a nested worker's error also reaches the parse worker's own
|
|
88
|
+
// global handler and from there the page's "Worker crashed" path.
|
|
89
|
+
ev.preventDefault();
|
|
90
|
+
const message = ev.message || 'unknown error';
|
|
91
|
+
if (!settled) {
|
|
92
|
+
settle(false, `the decompress worker failed to load: ${message}`);
|
|
93
|
+
return;
|
|
94
|
+
}
|
|
95
|
+
const stream = active;
|
|
96
|
+
giveUp(`the decompress worker crashed: ${message}`);
|
|
97
|
+
stream?.fail(new Error(`Decompress worker crashed: ${message}`));
|
|
98
|
+
};
|
|
99
|
+
});
|
|
100
|
+
|
|
101
|
+
return (onChunk) => {
|
|
102
|
+
const id = nextId++;
|
|
103
|
+
let started = false;
|
|
104
|
+
let local = null;
|
|
105
|
+
let seq = 0;
|
|
106
|
+
let inFlight = 0;
|
|
107
|
+
let failure = null;
|
|
108
|
+
let wake = null;
|
|
109
|
+
|
|
110
|
+
const until = (done ) => new Promise ((resolve) => {
|
|
111
|
+
const check = () => {
|
|
112
|
+
if (!done()) return;
|
|
113
|
+
wake = null;
|
|
114
|
+
resolve();
|
|
115
|
+
};
|
|
116
|
+
wake = check;
|
|
117
|
+
check();
|
|
118
|
+
});
|
|
119
|
+
|
|
120
|
+
const stream = {
|
|
121
|
+
id,
|
|
122
|
+
onReply(msg) {
|
|
123
|
+
if (failure) return;
|
|
124
|
+
if (msg.type === 'chunk') {
|
|
125
|
+
try {
|
|
126
|
+
onChunk(new Uint8Array(msg.bytes, 0, msg.length));
|
|
127
|
+
} catch (e) {
|
|
128
|
+
port?.postMessage({ type: 'cancel', id }, []);
|
|
129
|
+
stream.fail(e instanceof Error ? e : new Error(String(e)));
|
|
130
|
+
}
|
|
131
|
+
} else if (msg.type === 'consumed') {
|
|
132
|
+
inFlight--;
|
|
133
|
+
wake?.();
|
|
134
|
+
} else if (msg.type === 'error') {
|
|
135
|
+
stream.fail(new Error(msg.message));
|
|
136
|
+
}
|
|
137
|
+
},
|
|
138
|
+
fail(err) {
|
|
139
|
+
if (failure) return;
|
|
140
|
+
failure = err;
|
|
141
|
+
if (active === stream) active = null;
|
|
142
|
+
wake?.();
|
|
143
|
+
},
|
|
144
|
+
};
|
|
145
|
+
|
|
146
|
+
const begin = async () => {
|
|
147
|
+
started = true;
|
|
148
|
+
ready ??= startWorker();
|
|
149
|
+
if (!(await ready) || !port) {
|
|
150
|
+
local = fallback(onChunk);
|
|
151
|
+
return;
|
|
152
|
+
}
|
|
153
|
+
active = stream;
|
|
154
|
+
port.postMessage({ type: 'start', id }, []);
|
|
155
|
+
};
|
|
156
|
+
|
|
157
|
+
return {
|
|
158
|
+
async push(chunk, final = false) {
|
|
159
|
+
if (!started) await begin();
|
|
160
|
+
if (local) return local.push(chunk, final);
|
|
161
|
+
if (failure) throw failure;
|
|
162
|
+
// Transfer the slice's own buffer when it spans all of it; copy a view of a larger one.
|
|
163
|
+
const bytes = chunk.byteOffset === 0 && chunk.byteLength === chunk.buffer.byteLength
|
|
164
|
+
? chunk.buffer
|
|
165
|
+
: chunk.slice().buffer;
|
|
166
|
+
if (!port) throw new Error('Decompress worker is gone');
|
|
167
|
+
port.postMessage({ type: 'data', id, seq: seq++, bytes, final }, [bytes]);
|
|
168
|
+
inFlight++;
|
|
169
|
+
await until(() => failure !== null || inFlight < (final ? 1 : maxInFlight));
|
|
170
|
+
if (failure) throw failure;
|
|
171
|
+
if (final && active === stream) active = null;
|
|
172
|
+
},
|
|
173
|
+
cancel() {
|
|
174
|
+
if (local || !started || failure) return;
|
|
175
|
+
port?.postMessage({ type: 'cancel', id }, []);
|
|
176
|
+
stream.fail(new Error('Decompression cancelled'));
|
|
177
|
+
},
|
|
178
|
+
};
|
|
179
|
+
};
|
|
180
|
+
}
|
|
@@ -0,0 +1,103 @@
|
|
|
1
|
+
// Decompress worker: runs the vendored fzstd off the parse worker's thread, so decompression and
|
|
2
|
+
// NDJSON parsing of a zstd log overlap instead of taking turns. The parse worker spawns it and
|
|
3
|
+
// drives it through zstd-worker-client.ts; both ends of the message protocol live in the types
|
|
4
|
+
// below, and zstd-worker-client.ts documents the flow control.
|
|
5
|
+
import { Decompress as ZstdDecompress } from './vendor/fzstd.js';
|
|
6
|
+
|
|
7
|
+
// Parse worker -> decompress worker. `seq` numbers a stream's input slices from 0.
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
// Decompress worker -> parse worker. Every `chunk` for input slice `seq` is posted before that
|
|
14
|
+
// slice's `consumed`, and a stream's `error` is the last message it gets.
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
// fzstd emits one chunk per zstd block (128 KiB at most). Batching them into 1 MiB messages cuts
|
|
24
|
+
// the per-message cost on both threads about eight-fold.
|
|
25
|
+
export const OUTPUT_BATCH_BYTES = 1024 * 1024;
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
// The decompress side of the protocol, transport-free so tests can drive it over a
|
|
31
|
+
// MessageChannel. Holds at most one live stream: `start` replaces it, and a message for any
|
|
32
|
+
// other stream id (one already cancelled or failed) is dropped.
|
|
33
|
+
export function createZstdWorkerHandler(
|
|
34
|
+
post ,
|
|
35
|
+
batchBytes = OUTPUT_BATCH_BYTES,
|
|
36
|
+
) {
|
|
37
|
+
let streamId = -1;
|
|
38
|
+
let decoder = null;
|
|
39
|
+
let batch = new Uint8Array(batchBytes);
|
|
40
|
+
let batchUsed = 0;
|
|
41
|
+
|
|
42
|
+
// Hand `batch` over without copying it again: transfer its whole buffer with the filled
|
|
43
|
+
// length and start a fresh one. fzstd's chunk is a view of a buffer it reuses, so the copy
|
|
44
|
+
// into `batch` is the one copy this path cannot avoid.
|
|
45
|
+
const flush = () => {
|
|
46
|
+
if (batchUsed === 0) return;
|
|
47
|
+
const bytes = batch.buffer ;
|
|
48
|
+
post({ type: 'chunk', id: streamId, bytes, length: batchUsed }, [bytes]);
|
|
49
|
+
batch = new Uint8Array(batchBytes);
|
|
50
|
+
batchUsed = 0;
|
|
51
|
+
};
|
|
52
|
+
|
|
53
|
+
const collect = (chunk ) => {
|
|
54
|
+
let offset = 0;
|
|
55
|
+
while (offset < chunk.length) {
|
|
56
|
+
const n = Math.min(chunk.length - offset, batchBytes - batchUsed);
|
|
57
|
+
batch.set(chunk.subarray(offset, offset + n), batchUsed);
|
|
58
|
+
batchUsed += n;
|
|
59
|
+
offset += n;
|
|
60
|
+
if (batchUsed === batchBytes) flush();
|
|
61
|
+
}
|
|
62
|
+
};
|
|
63
|
+
|
|
64
|
+
const reset = () => {
|
|
65
|
+
decoder = null;
|
|
66
|
+
batchUsed = 0;
|
|
67
|
+
};
|
|
68
|
+
|
|
69
|
+
return (msg) => {
|
|
70
|
+
if (msg.type === 'start') {
|
|
71
|
+
streamId = msg.id;
|
|
72
|
+
batchUsed = 0;
|
|
73
|
+
decoder = new (ZstdDecompress )(collect);
|
|
74
|
+
return;
|
|
75
|
+
}
|
|
76
|
+
if (msg.id !== streamId || !decoder) return;
|
|
77
|
+
if (msg.type === 'cancel') {
|
|
78
|
+
reset();
|
|
79
|
+
return;
|
|
80
|
+
}
|
|
81
|
+
try {
|
|
82
|
+
decoder.push(new Uint8Array(msg.bytes), msg.final);
|
|
83
|
+
flush();
|
|
84
|
+
} catch (e) {
|
|
85
|
+
reset();
|
|
86
|
+
post({ type: 'error', id: msg.id, message: e instanceof Error ? e.message : String(e) });
|
|
87
|
+
return;
|
|
88
|
+
}
|
|
89
|
+
post({ type: 'consumed', id: msg.id, seq: msg.seq });
|
|
90
|
+
if (msg.final) reset();
|
|
91
|
+
};
|
|
92
|
+
}
|
|
93
|
+
|
|
94
|
+
// ─── Worker message bus (only active when running as a Web Worker) ──────────────
|
|
95
|
+
|
|
96
|
+
const isWorker = typeof WorkerGlobalScope !== 'undefined' && self instanceof WorkerGlobalScope;
|
|
97
|
+
|
|
98
|
+
if (isWorker) {
|
|
99
|
+
const scope = self ;
|
|
100
|
+
const handle = createZstdWorkerHandler((msg, transfer) => scope.postMessage(msg, transfer ?? []));
|
|
101
|
+
scope.onmessage = ({ data } ) => handle(data);
|
|
102
|
+
scope.postMessage({ type: 'ready' } );
|
|
103
|
+
}
|