sparkforensics-mcp 0.2.0 → 0.2.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (44) hide show
  1. package/package.json +5 -4
  2. package/vendor-core/cli/collect-run.js +3 -2
  3. package/vendor-core/cli/native-zstd.js +351 -0
  4. package/vendor-core/detectors.js +390 -75
  5. package/vendor-core/docs-config.js +34 -8
  6. package/vendor-core/docs-content/chapters/03-memory-model.md +39 -0
  7. package/vendor-core/docs-content/chapters/11-cluster-config.md +40 -0
  8. package/vendor-core/docs-content/detection/cache.md +4 -3
  9. package/vendor-core/docs-content/detection/chrn.md +4 -2
  10. package/vendor-core/docs-content/detection/gc.md +2 -0
  11. package/vendor-core/docs-content/detection/host.md +2 -1
  12. package/vendor-core/docs-content/detection/local.md +2 -3
  13. package/vendor-core/docs-content/detection/mem.md +3 -3
  14. package/vendor-core/docs-content/detection/plan.md +3 -1
  15. package/vendor-core/docs-content/detection/shape.md +2 -1
  16. package/vendor-core/docs-content/detection/shfl.md +2 -1
  17. package/vendor-core/docs-content/detection/spec.md +4 -3
  18. package/vendor-core/docs-content/detection/spill.md +1 -1
  19. package/vendor-core/docs-content/detection/strag.md +2 -1
  20. package/vendor-core/docs-content/detection/tiny.md +2 -1
  21. package/vendor-core/docs-content/tuning/failures.md +1 -1
  22. package/vendor-core/docs-content/tuning/gc.md +11 -4
  23. package/vendor-core/docs-content/tuning/shuffle.md +25 -5
  24. package/vendor-core/docs-content/tuning/skew.md +14 -6
  25. package/vendor-core/docs-content/tuning/small-files.md +12 -7
  26. package/vendor-core/docs-content/tuning/straggler.md +34 -0
  27. package/vendor-core/docs-content/tuning/tiny-tasks.md +1 -1
  28. package/vendor-core/docs-content/tuning/utilization.md +57 -7
  29. package/vendor-core/docs-content/upstream.json +4 -0
  30. package/vendor-core/event-handlers.js +321 -69
  31. package/vendor-core/event-schemas.js +8 -6
  32. package/vendor-core/evidence-report.js +4 -2
  33. package/vendor-core/impact-estimator.js +150 -34
  34. package/vendor-core/mcp-tools.js +20 -7
  35. package/vendor-core/occupancy.js +70 -2
  36. package/vendor-core/parser-worker.js +30 -14
  37. package/vendor-core/plan-summary.js +4 -0
  38. package/vendor-core/run-comparison.js +22 -17
  39. package/vendor-core/shs-fetch.js +18 -7
  40. package/vendor-core/shs-load.js +2 -1
  41. package/vendor-core/stage-quantiles.js +59 -3
  42. package/vendor-core/string-hash.js +15 -0
  43. package/vendor-core/types.js +9 -1
  44. package/vendor-core/vendor/fzstd.js +94 -18
@@ -1,5 +1,6 @@
1
1
  import { computeWallClock } from './wall-clock.js';
2
2
  import { normalizeDetail } from './detectors.js';
3
+ import { cyrb53 } from './string-hash.js';
3
4
  import { captureSnapshot } from './session-snapshot.js';
4
5
 
5
6
 
@@ -35,21 +36,6 @@ export function normalizeStageName(name ) {
35
36
  .trim();
36
37
  }
37
38
 
38
- // cyrb53: fast, non-cryptographic, deterministic 53-bit string hash. Not a
39
- // security boundary; 53 bits keeps collision risk negligible at the per-node,
40
- // per-stage call volume planTreeIdentity/sqlNodeIdentity put through it.
41
- function cyrb53(str , seed = 0) {
42
- let h1 = 0xdeadbeef ^ seed, h2 = 0x41c6ce57 ^ seed;
43
- for (let i = 0; i < str.length; i++) {
44
- const ch = str.charCodeAt(i);
45
- h1 = Math.imul(h1 ^ ch, 2654435761);
46
- h2 = Math.imul(h2 ^ ch, 1597334677);
47
- }
48
- h1 = Math.imul(h1 ^ (h1 >>> 16), 2246822507) ^ Math.imul(h2 ^ (h2 >>> 13), 3266489909);
49
- h2 = Math.imul(h2 ^ (h2 >>> 16), 2246822507) ^ Math.imul(h1 ^ (h1 >>> 13), 3266489909);
50
- return (4294967296 * (2097151 & h2) + (h1 >>> 0)).toString(16);
51
- }
52
-
53
39
  // Bottom-up, order-independent structural identity of a resolved plan tree:
54
40
  // each node folds its normalized name/detail with its children's digests
55
41
  // (children sorted, so AQE picking a different broadcast side still matches),
@@ -395,6 +381,14 @@ export function buildComparison(
395
381
  );
396
382
  }
397
383
 
384
+ // matchStages' coverage is a Dice coefficient: (2 * pairs.length) / (baseCount
385
+ // + candCount). Below 0.5, more than half of each run's stages went unpaired,
386
+ // so the stage-level rows (stageSkew, baseStages/candStages) mostly show
387
+ // unrelated work side by side rather than the same stage before/after -- the
388
+ // comparison is dominated by guesswork, not genuine pairing. 0.5 is thus the
389
+ // natural midpoint for "more matched than not," not an arbitrary tuning knob.
390
+ const LOW_COVERAGE_THRESHOLD = 0.5;
391
+
398
392
  export function compareRuns(
399
393
  baseline ,
400
394
  candidate ,
@@ -405,10 +399,21 @@ export function compareRuns(
405
399
  // is the normal way to label an A/B experiment, so a mismatch must not block
406
400
  // the (matching-free, name-independent) deltas. Surface it as `low` instead.
407
401
  const namesDiffer = namesConflict(baseSnap.app, candSnap.app);
402
+ // Coverage is the other half of the signal: identical names on two runs that
403
+ // barely share any stages are just as misleading as differing names on two
404
+ // runs that match well, so either condition alone drops confidence to `low`.
405
+ const lowCoverage = match.coverage < LOW_COVERAGE_THRESHOLD;
406
+ const reason = namesDiffer && lowCoverage
407
+ ? `Run names differ and only ${(match.coverage * 100).toFixed(0)}% of stages matched, so deltas may compare different work.`
408
+ : namesDiffer
409
+ ? 'Run names differ, so deltas may compare different work.'
410
+ : lowCoverage
411
+ ? `Only ${(match.coverage * 100).toFixed(0)}% of stages matched between runs, so per-stage rows mostly compare unrelated work.`
412
+ : null;
408
413
  return {
409
414
  baselineLabel: baseline.label, candidateLabel: candidate.label,
410
- confidence: namesDiffer ? 'low' : 'ok',
411
- reason: namesDiffer ? 'Run names differ, so deltas may compare different work.' : null,
415
+ confidence: namesDiffer || lowCoverage ? 'low' : 'ok',
416
+ reason,
412
417
  matchedCoverage: match.coverage,
413
418
  metrics: metricDeltas(baseSnap, candSnap),
414
419
  findings: findingsDelta(baseSnap, candSnap),
@@ -3,7 +3,7 @@ import { createLz4BlockDecoder } from './lz4-block.js';
3
3
  import { Decompress as ZstdDecompress } from './vendor/fzstd.js';
4
4
  import { createSnappyBlockDecoder } from './snappy-block.js';
5
5
  import { buildProxyRequestUrl, isShsErrorCode } from './shs-request.js';
6
- import { dispatchLine, buildChunkDecoder, emitParseCompletion, } from './event-handlers.js';
6
+ import { dispatchLine, buildChunkDecoder, emitParseCompletion, } from './event-handlers.js';
7
7
  import { ShsProxyErrorBodySchema } from './shs-schemas.js';
8
8
  import { naturalCompare, reassembleRollingEntries } from './rolling-log-reassembly.js';
9
9
 
@@ -36,8 +36,14 @@ export function sniffCodec(bytes )
36
36
  // without touching the vendored files.
37
37
 
38
38
 
39
+ // Like parser-worker.ts's RunOpts.zstdDecoder, but synchronous: decodeEntry never awaits push(),
40
+ // so its result is typed `undefined` (not `void`, which would also accept an async decoder's
41
+ // Promise). Node callers pass cli/native-zstd.ts's createNativeZstdDecoder (nodeArchiveCodecs).
42
+
39
43
 
40
- function decodeEntry(name , raw , onChunk ) {
44
+ function decodeEntry(
45
+ name , raw , onChunk , zstdDecoder ,
46
+ ) {
41
47
  const codec = sniffCodec(raw);
42
48
  if (codec === 'lz4' || name.endsWith('.lz4')) {
43
49
  const lz4 = createLz4BlockDecoder(onChunk);
@@ -46,7 +52,7 @@ function decodeEntry(name , raw , onChunk
46
52
  } else if (codec === 'gz' || name.endsWith('.gz')) {
47
53
  new (Gunzip )(onChunk).push(raw, true);
48
54
  } else if (codec === 'zstd' || name.endsWith('.zstd') || name.endsWith('.zst')) {
49
- new (ZstdDecompress )(onChunk).push(raw, true);
55
+ (zstdDecoder ? zstdDecoder(onChunk) : new (ZstdDecompress )(onChunk)).push(raw, true);
50
56
  } else if (codec === 'snappy' || name.endsWith('.snappy')) {
51
57
  const snappy = createSnappyBlockDecoder(onChunk);
52
58
  snappy.push(raw);
@@ -138,7 +144,9 @@ export async function runParseFromUrl(
138
144
  decodeShsArchive(zipBytes, state, emit);
139
145
  }
140
146
 
141
- export function decodeShsArchive(zipBytes , state , emit ) {
147
+ export function decodeShsArchive(
148
+ zipBytes , state , emit , { zstdDecoder } = {},
149
+ ) {
142
150
  let entries ;
143
151
  try {
144
152
  entries = unzipSync(zipBytes);
@@ -165,18 +173,21 @@ export function decodeShsArchive(zipBytes , state , emit
165
173
  }
166
174
 
167
175
  const decoder = buildChunkDecoder();
176
+ const joined = [];
168
177
  let linesProcessed = 0;
169
178
  for (const name of names) {
170
179
  try {
171
180
  decodeEntry(name, entries[name], (bytes) => {
172
- for (const line of decoder.decode(bytes)) {
173
- dispatchLine(line, state, emit);
181
+ joined.length = 0;
182
+ const lines = decoder.decode(bytes, joined);
183
+ for (let i = 0, j = 0; i < lines.length; i++) {
184
+ dispatchLine(lines[i], state, emit, joined[j]?.index === i ? joined[j++] : undefined);
174
185
  linesProcessed++;
175
186
  if (linesProcessed % 2000 === 0) {
176
187
  emit({ type: 'progress', pct: null, linesProcessed });
177
188
  }
178
189
  }
179
- });
190
+ }, zstdDecoder);
180
191
  } catch {
181
192
  emitShsError(emit, 'invalid-event-log');
182
193
  return;
@@ -2,6 +2,7 @@ import { validateShsRequest } from './shs-request.js';
2
2
  import { fetchShsEventLog } from './proxy.js';
3
3
  import { decodeShsArchive } from './parser-worker.js';
4
4
  import { collectViaDispatch } from './cli/collect-run.js';
5
+ import { nodeArchiveCodecs } from './cli/native-zstd.js';
5
6
  import { deriveEvidenceAvailability } from './evidence-availability.js';
6
7
  import { mcpError } from './mcp-error.js';
7
8
 
@@ -18,7 +19,7 @@ export const DEFAULT_IDLE_TIMEOUT_MS = envInt('SPARKFORENSICS_SHS_TIMEOUT_MS', 3
18
19
 
19
20
  function collectShsAppModel(zipBytes ) {
20
21
  return collectViaDispatch(
21
- (state, emit) => decodeShsArchive(zipBytes, state, emit),
22
+ (state, emit) => decodeShsArchive(zipBytes, state, emit, nodeArchiveCodecs),
22
23
  (msg) => {
23
24
  const m = msg ;
24
25
  return mcpError(m?.code ?? 'invalid-event-log', m?.message ?? 'Failed to decode SHS archive.');
@@ -135,13 +135,18 @@ export function finalizeStage(
135
135
  const { p50: spillMemP50, p95: spillMemP95, max: spillMemMax } = computeFieldQuantiles(arr, FIELDS.MEM_SPILLED);
136
136
  const { p50: spillDiskP50, p95: spillDiskP95, max: spillDiskMax } = computeFieldQuantiles(arr, FIELDS.DISK_SPILLED);
137
137
 
138
- // Straggler count: tasks with duration > 4 * P50.
138
+ // Straggler count: tasks with duration > 4 * P50, their summed excess over P50, and the longest
139
+ // task that isn't one.
139
140
  const stragglerThreshold = 4 * p50;
140
141
  let stragglerCount = 0;
142
+ let stragglerExcessMs = 0;
143
+ let longestNonStragglerMs = 0;
141
144
  const taskArrCount = arr.length / FIELDS.STRIDE;
142
145
  if (p50 > 0) {
143
146
  for (let i = 0; i < taskArrCount; i++) {
144
- if (arr[i * FIELDS.STRIDE + FIELDS.DURATION] > stragglerThreshold) stragglerCount++;
147
+ const duration = arr[i * FIELDS.STRIDE + FIELDS.DURATION];
148
+ if (duration > stragglerThreshold) { stragglerCount++; stragglerExcessMs += duration - p50; }
149
+ else if (duration > longestNonStragglerMs) longestNonStragglerMs = duration;
145
150
  }
146
151
  }
147
152
 
@@ -160,9 +165,11 @@ export function finalizeStage(
160
165
 
161
166
  const data = {
162
167
  ...stage,
163
- hostStats: hostStatsArr, executorStats: executorStatsArr, failureReasons: failureReasonsArr, localityStats: localityStatsArr, stragglerCount,
168
+ hostStats: hostStatsArr, executorStats: executorStatsArr, failureReasons: failureReasonsArr, localityStats: localityStatsArr, stragglerCount, stragglerExcessMs, longestNonStragglerMs,
164
169
  failedTaskSamples,
165
170
  peakExecutionMemoryMax,
171
+ taskActiveMs: computeTaskActiveMs(arr),
172
+ peakConcurrentTasks: computePeakConcurrentTasks(arr),
166
173
  taskDurationP50: p50,
167
174
  taskDurationP95: p95,
168
175
  taskDurationMax: max,
@@ -178,6 +185,55 @@ export function finalizeStage(
178
185
  return { type: 'stage', data };
179
186
  }
180
187
 
188
+ // Wall-clock time during which at least one of the stage's tasks was running: the union of its
189
+ // [launch, finish) intervals. A stage's submittedAt..completedAt window also covers time it sat
190
+ // open with no task running (waiting for a free slot, or between retried tasks), which no
191
+ // task-level fix can compress. Tasks missing either timestamp are skipped.
192
+ export function computeTaskActiveMs(arr ) {
193
+ const taskCount = arr.length / FIELDS.STRIDE;
194
+ const intervals = [];
195
+ for (let i = 0; i < taskCount; i++) {
196
+ const launch = arr[i * FIELDS.STRIDE + FIELDS.LAUNCH_TIME];
197
+ const finish = arr[i * FIELDS.STRIDE + FIELDS.FINISH_TIME];
198
+ if (launch > 0 && finish > launch) intervals.push([launch, finish]);
199
+ }
200
+ intervals.sort((a, b) => a[0] - b[0]);
201
+ let activeMs = 0, start = -Infinity, end = -Infinity;
202
+ for (const [launch, finish] of intervals) {
203
+ if (launch > end) {
204
+ if (end > start) activeMs += end - start;
205
+ start = launch;
206
+ end = finish;
207
+ } else if (finish > end) {
208
+ end = finish;
209
+ }
210
+ }
211
+ if (end > start) activeMs += end - start;
212
+ return activeMs;
213
+ }
214
+
215
+ // Most tasks running at once, over the same [launch, finish) intervals as computeTaskActiveMs: the
216
+ // slots the stage actually got. A task finishing at the instant another launches frees its slot.
217
+ export function computePeakConcurrentTasks(arr ) {
218
+ const taskCount = arr.length / FIELDS.STRIDE;
219
+ const launches = [];
220
+ const finishes = [];
221
+ for (let i = 0; i < taskCount; i++) {
222
+ const launch = arr[i * FIELDS.STRIDE + FIELDS.LAUNCH_TIME];
223
+ const finish = arr[i * FIELDS.STRIDE + FIELDS.FINISH_TIME];
224
+ if (launch > 0 && finish > launch) { launches.push(launch); finishes.push(finish); }
225
+ }
226
+ launches.sort((a, b) => a - b);
227
+ finishes.sort((a, b) => a - b);
228
+ let peak = 0, running = 0, finished = 0;
229
+ for (const launch of launches) {
230
+ while (finished < finishes.length && finishes[finished] <= launch) { running--; finished++; }
231
+ running++;
232
+ if (running > peak) peak = running;
233
+ }
234
+ return peak;
235
+ }
236
+
181
237
  export function computeFieldQuantiles(arr , fieldIndex ) {
182
238
  const taskCount = arr.length / FIELDS.STRIDE;
183
239
  if (taskCount === 0) return { p50: 0, p95: 0, max: 0 };
@@ -0,0 +1,15 @@
1
+ // cyrb53: fast, non-cryptographic, deterministic 53-bit string hash. Not a
2
+ // security boundary; 53 bits keeps collision risk negligible at the per-node
3
+ // call volume the plan-identity digests (run-comparison.ts, detectors.ts) put
4
+ // through it.
5
+ export function cyrb53(str , seed = 0) {
6
+ let h1 = 0xdeadbeef ^ seed, h2 = 0x41c6ce57 ^ seed;
7
+ for (let i = 0; i < str.length; i++) {
8
+ const ch = str.charCodeAt(i);
9
+ h1 = Math.imul(h1 ^ ch, 2654435761);
10
+ h2 = Math.imul(h2 ^ ch, 1597334677);
11
+ }
12
+ h1 = Math.imul(h1 ^ (h1 >>> 16), 2246822507) ^ Math.imul(h2 ^ (h2 >>> 13), 3266489909);
13
+ h2 = Math.imul(h2 ^ (h2 >>> 16), 2246822507) ^ Math.imul(h1 ^ (h1 >>> 13), 3266489909);
14
+ return (4294967296 * (2097151 & h2) + (h1 >>> 0)).toString(16);
15
+ }
@@ -4,6 +4,17 @@
4
4
  // Only the streaming Decompress class is used by this project (see
5
5
  // src/parser-worker.js) to inflate Spark event logs written with
6
6
  // spark.io.compression.codec=zstd, one block at a time.
7
+ // Local patches, each marked "Local patch" below: Decompress.push loops over
8
+ // frame boundaries instead of recursing; streaming decodes every block into one
9
+ // reused [window | block] buffer (see the note above Decompress) instead of a
10
+ // fresh window per frame and a fresh buffer per block; the sequence loop moves
11
+ // literal runs, non-overlapping matches and window reads longer than CPW_MIN
12
+ // bytes with copyWithin/set instead of a byte loop. On the largest real log
13
+ // they took fzstd from 11.5s to 2.9s in Chrome (5.7s to 2.8s in Node), output
14
+ // byte-identical on all 14 real logs. On 2400 randomly corrupted streams the
15
+ // output and error matched upstream on every one but a corrupt window size
16
+ // that overflows negative: both throw "Invalid typed array length" at the same
17
+ // byte, with a different number in the message (the dev/fuzz-fzstd.mjs check).
7
18
  // Some numerical data is initialized as -1 even when it doesn't need initialization to help the JIT infer types
8
19
  // aliases for shorter compressed code (most minifers don't do this)
9
20
  var ab = ArrayBuffer, u8 = Uint8Array, u16 = Uint16Array, i16 = Int16Array, u32 = Uint32Array, i32 = Int32Array;
@@ -106,7 +117,9 @@ var rzfh = function (dat, w) {
106
117
  }
107
118
  if (ws > 2145386496)
108
119
  err(1);
109
- var buf = new u8((w == 1 ? (fss || ws) : w ? 0 : ws) + 12);
120
+ // Local patch: the streaming Decompress (no `w`) keeps its window in its own buffer (see
121
+ // the note above Decompress), so none is allocated here.
122
+ var buf = new u8((w == 1 ? (fss || ws) : 0) + 12);
110
123
  buf[0] = 1, buf[4] = 4, buf[8] = 8;
111
124
  return {
112
125
  b: bt + fsb,
@@ -394,6 +407,13 @@ var dhu4 = function (dat, out, hu) {
394
407
  dhu(dat.subarray(bt, bt += dat[4] | (dat[5] << 8)), out.subarray(sz2, sz3), hu);
395
408
  dhu(dat.subarray(bt), out.subarray(sz3), hu);
396
409
  };
410
+ // Local patch: runs longer than this are copied natively, where a forward byte copy and
411
+ // copyWithin (memmove) agree: a match whose source ends before its destination starts (offset >=
412
+ // length), and a literal run whose destination doesn't pass its source (corrupt input can push
413
+ // the output past the literals). An overlapping match repeats its own output, which only the byte
414
+ // loop does. A source range past the end of its buffer (corrupt input) stays a loop, which reads
415
+ // undefined and so writes 0 there, as upstream does. 16 and 32 measured the same, 8 and 64 slower.
416
+ var CPW_MIN = 16;
397
417
  // read Zstandard block
398
418
  var rzb = function (dat, st, out) {
399
419
  var _a;
@@ -410,7 +430,7 @@ var rzb = function (dat, st, out) {
410
430
  st.b = bt + 1;
411
431
  if (out) {
412
432
  fill(out, dat[bt], st.y, st.y += sz);
413
- return out;
433
+ return st.hv == null ? out : out.subarray(st.y - sz, st.y);
414
434
  }
415
435
  return fill(new u8(sz), dat[bt]);
416
436
  }
@@ -421,7 +441,7 @@ var rzb = function (dat, st, out) {
421
441
  if (out) {
422
442
  out.set(dat.subarray(bt, ebt), st.y);
423
443
  st.y += sz;
424
- return out;
444
+ return st.hv == null ? out : out.subarray(st.y - sz, st.y);
425
445
  }
426
446
  return slc(dat, bt, ebt);
427
447
  }
@@ -548,9 +568,12 @@ var rzb = function (dat, st, out) {
548
568
  else
549
569
  off = st.o[0];
550
570
  }
551
- for (var i = 0; i < ll; ++i) {
552
- buf[oubt + i] = buf[spl + i];
553
- }
571
+ if (ll > CPW_MIN && spl + ll <= buf.length && oubt <= spl) // see CPW_MIN
572
+ buf.copyWithin(oubt, spl, spl + ll);
573
+ else
574
+ for (var i = 0; i < ll; ++i) {
575
+ buf[oubt + i] = buf[spl + i];
576
+ }
554
577
  oubt += ll, spl += ll;
555
578
  var stin = oubt - off;
556
579
  if (stin < 0) {
@@ -558,14 +581,34 @@ var rzb = function (dat, st, out) {
558
581
  var bs = st.e + stin;
559
582
  if (len > ml)
560
583
  len = ml;
561
- for (var i = 0; i < len; ++i) {
562
- buf[oubt + i] = st.w[bs + i];
584
+ // Local patch: streaming keeps one window buffer for the whole stream (see
585
+ // Decompress.push), so history the frame hasn't written yet reads as the 0s
586
+ // of upstream's fresh zeroed window.
587
+ if (st.hv != null && stin < -st.hv) {
588
+ var z = Math.min(len, -st.hv - stin);
589
+ fill(buf, 0, oubt, oubt + z);
590
+ oubt += z, ml -= z, len -= z, bs += z;
563
591
  }
592
+ // Local patch: in bounds only; an out-of-range read stays a loop so it
593
+ // still yields 0 (subarray would wrap a negative start).
594
+ if (len > CPW_MIN && bs >= 0 && bs + len <= st.w.length && oubt + len <= buf.length) {
595
+ if (st.w.buffer === buf.buffer)
596
+ st.w.copyWithin(st.e + oubt, bs, bs + len);
597
+ else
598
+ buf.set(st.w.subarray(bs, bs + len), oubt);
599
+ }
600
+ else
601
+ for (var i = 0; i < len; ++i) {
602
+ buf[oubt + i] = st.w[bs + i];
603
+ }
564
604
  oubt += len, ml -= len, stin = 0;
565
605
  }
566
- for (var i = 0; i < ml; ++i) {
567
- buf[oubt + i] = buf[stin + i];
568
- }
606
+ if (ml > CPW_MIN && oubt - stin >= ml)
607
+ buf.copyWithin(oubt, stin, stin + ml);
608
+ else
609
+ for (var i = 0; i < ml; ++i) {
610
+ buf[oubt + i] = buf[stin + i];
611
+ }
569
612
  oubt += ml;
570
613
  }
571
614
  if (oubt != spl) {
@@ -575,12 +618,12 @@ var rzb = function (dat, st, out) {
575
618
  }
576
619
  else
577
620
  oubt = buf.length;
578
- if (out)
621
+ if (out && st.hv == null)
579
622
  st.y += oubt;
580
623
  else
581
- buf = slc(buf, 0, oubt);
624
+ buf = buf.subarray(0, oubt); // local patch: buf is this block's own, no copy needed
582
625
  }
583
- else if (out) {
626
+ else if (out && st.hv == null) {
584
627
  st.y += lss;
585
628
  if (spl) {
586
629
  for (var i = 0; i < lss; ++i) {
@@ -589,7 +632,7 @@ var rzb = function (dat, st, out) {
589
632
  }
590
633
  }
591
634
  else if (spl)
592
- buf = slc(buf, spl);
635
+ buf = buf.subarray(spl); // local patch: as above
593
636
  st.b = ebt;
594
637
  return buf;
595
638
  }
@@ -654,6 +697,15 @@ export function decompress(dat, buf) {
654
697
  }
655
698
  return cct(bufs, ol);
656
699
  }
700
+ // Local patch: the streaming Decompress keeps one buffer, this.p, laid out as [window | block]:
701
+ // its first st.e (window size) bytes hold the frame's latest output, and each block is decoded
702
+ // right after them (st.hv tracks how much of the window this frame has written; older positions
703
+ // read as 0, as upstream's fresh zeroed window). A back-reference into an earlier block is then a
704
+ // copy within that one buffer. The block part is zeroed before each block, so a block sees
705
+ // exactly what upstream's fresh block buffer held. Upstream allocated and zeroed a window per
706
+ // frame, a buffer per block, then copied each block out and shifted the window. A chunk passed
707
+ // to ondata is a view of this.p and is only valid until ondata returns: every caller in this
708
+ // project decodes it at once.
657
709
  /**
658
710
  * Decompressor for Zstandard streamed data
659
711
  */
@@ -737,7 +789,22 @@ var Decompress = /*#__PURE__*/ (function () {
737
789
  else
738
790
  this.z = 0;
739
791
  for (;;) {
740
- var blk = rzb(chunk, this.s);
792
+ // Local patch: decode into this.p (see the note above Decompress), grown to fit
793
+ // the block's declared size, keeping the window.
794
+ var st = this.s;
795
+ if (st.hv == null)
796
+ st.hv = 0;
797
+ var hb = st.b, need = Math.max(st.m, (chunk[hb] >> 3) | (chunk[hb + 1] << 5) | (chunk[hb + 2] << 13));
798
+ var P = this.p;
799
+ if (!P || this.pe != st.e || P.length < st.e + need) {
800
+ var np = new u8(st.e + need);
801
+ if (P && this.pe == st.e)
802
+ np.set(P.subarray(0, st.e));
803
+ P = this.p = np, this.pe = st.e;
804
+ }
805
+ st.w = P, st.y = st.e;
806
+ P.fill(0, st.e, st.e + need);
807
+ var blk = rzb(chunk, st, P);
741
808
  if (!blk) {
742
809
  if (final)
743
810
  err(5);
@@ -748,8 +815,17 @@ var Decompress = /*#__PURE__*/ (function () {
748
815
  }
749
816
  else {
750
817
  this.ondata(blk, false);
751
- cpw(this.s.w, 0, blk.length);
752
- this.s.w.set(blk, this.s.w.length - blk.length);
818
+ // Local patch: upstream's window update, on the window part of this.p. It
819
+ // is skipped after a frame's last block, which nothing reads, except
820
+ // that a block longer than the window (corrupt input only) still throws
821
+ // the RangeError upstream's update does.
822
+ if (!st.l) {
823
+ P.copyWithin(0, blk.length, st.e);
824
+ P.set(blk, st.e - blk.length);
825
+ st.hv = Math.min(st.e, st.hv + blk.length);
826
+ }
827
+ else if (blk.length > st.e)
828
+ P.set(blk, st.e - blk.length);
753
829
  }
754
830
  if (this.s.l) {
755
831
  chunk = chunk.subarray(this.s.b);