sparkforensics-mcp 0.2.4 → 0.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (71) hide show
  1. package/LICENSE +21 -0
  2. package/README.md +7 -1
  3. package/bin/sparkforensics-mcp.mjs +41 -10
  4. package/package.json +1 -1
  5. package/vendor-core/analyzer.js +156 -48
  6. package/vendor-core/check-coverage.js +88 -0
  7. package/vendor-core/cli/budgets.js +31 -18
  8. package/vendor-core/cli/collect-run.js +76 -31
  9. package/vendor-core/cli/native-zstd.js +2 -2
  10. package/vendor-core/cli/threshold-config.js +28 -0
  11. package/vendor-core/comparison-verdict.js +177 -0
  12. package/vendor-core/core-source-hash.txt +1 -0
  13. package/vendor-core/core-usage-locality.js +56 -2
  14. package/vendor-core/detector-docs.js +58 -0
  15. package/vendor-core/detectors.js +933 -459
  16. package/vendor-core/docs-config.js +0 -36
  17. package/vendor-core/docs-content/chapters/nav-index.json +31 -0
  18. package/vendor-core/docs-content/detection/cstor.md +9 -0
  19. package/vendor-core/docs-content/detection/fail.md +6 -2
  20. package/vendor-core/docs-site-config.js +3 -0
  21. package/vendor-core/event-handlers.js +191 -6
  22. package/vendor-core/event-schemas.js +29 -0
  23. package/vendor-core/evidence-report.js +440 -112
  24. package/vendor-core/export-data.js +79 -6
  25. package/vendor-core/finding-action-label.js +9 -88
  26. package/vendor-core/finding-filter-predicate.js +9 -0
  27. package/vendor-core/finding-generic-recommendation.js +6 -104
  28. package/vendor-core/finding-names.js +21 -45
  29. package/vendor-core/finding-presentation.js +333 -0
  30. package/vendor-core/finding-tag-help.js +110 -0
  31. package/vendor-core/finding-types.js +361 -0
  32. package/vendor-core/findings-of-type.js +11 -0
  33. package/vendor-core/format-utils.js +92 -27
  34. package/vendor-core/html-export.js +51 -0
  35. package/vendor-core/impact-band.js +21 -8
  36. package/vendor-core/impact-estimator.js +8 -521
  37. package/vendor-core/impact-format.js +114 -0
  38. package/vendor-core/impact-model.js +175 -0
  39. package/vendor-core/ingest.js +2 -0
  40. package/vendor-core/intervals.js +13 -0
  41. package/vendor-core/list-runs.js +2 -3
  42. package/vendor-core/load-vendored.js +70 -5
  43. package/vendor-core/mcp-server-factory.js +14 -10
  44. package/vendor-core/mcp-tools.js +105 -45
  45. package/vendor-core/model-assembler.js +12 -0
  46. package/vendor-core/occupancy.js +1 -1
  47. package/vendor-core/parser-worker.js +22 -5
  48. package/vendor-core/plan-graph-model.js +3 -2
  49. package/vendor-core/plan-node-detail.js +1 -1
  50. package/vendor-core/recommendation-rollup.js +63 -3
  51. package/vendor-core/redact.js +68 -28
  52. package/vendor-core/run-comparison.js +40 -7
  53. package/vendor-core/run-interpretation.js +290 -0
  54. package/vendor-core/run-outcome.js +74 -0
  55. package/vendor-core/run-payload.js +17 -0
  56. package/vendor-core/run-shape.js +40 -0
  57. package/vendor-core/run-verdict.js +353 -0
  58. package/vendor-core/scaling-sim.js +4 -5
  59. package/vendor-core/scorecard-estimates.js +62 -0
  60. package/vendor-core/shs-fetch.js +175 -65
  61. package/vendor-core/shs-load.js +1 -1
  62. package/vendor-core/sql-stages.js +11 -0
  63. package/vendor-core/stage-quantiles.js +14 -0
  64. package/vendor-core/task-failure.js +151 -0
  65. package/vendor-core/threshold-overrides.js +160 -0
  66. package/vendor-core/threshold-summary.js +11 -33
  67. package/vendor-core/types.js +6 -42
  68. package/vendor-core/vendor/fflate.js +1 -1
  69. package/vendor-core/wall-clock.js +1 -12
  70. package/vendor-core/wasted-core-hours.js +2 -2
  71. package/vendor-core/zip-archive.js +167 -0
@@ -0,0 +1,160 @@
1
+ // User threshold overrides: validation of a parsed config file, and which of a detector's
2
+ // thresholds an override actually moved off its specification default. Only the CLI and the MCP
3
+ // server accept overrides; the dashboard always runs the defaults. Reading the file is
4
+ // cli/threshold-config.ts's job, so this module stays free of Node APIs.
5
+ import {
6
+ DETECTORS, ENTRY_BY_TYPE, detectorCatalog,
7
+ } from './detectors.js';
8
+
9
+
10
+ const entries = DETECTORS;
11
+
12
+ function isPlainObject(value ) {
13
+ return typeof value === 'object' && value !== null && !Array.isArray(value);
14
+ }
15
+
16
+ function isNonNegativeNumber(value ) {
17
+ return typeof value === 'number' && Number.isFinite(value) && value >= 0;
18
+ }
19
+
20
+ // Entries a user may tune: not config-scope (those audits compare against Spark's own defaults,
21
+ // such as its memoryOverhead floor), and with at least one threshold.
22
+ function tunableEntry(type ) {
23
+ const entry = entries.find((d) => d.type === type);
24
+ if (!entry) {
25
+ const tunable = entries.filter((d) => d.scope !== 'config' && Object.keys(d.thresholds).length > 0).map((d) => d.type);
26
+ throw new Error(`unknown detector "${type}" (tunable detectors: ${tunable.join(', ')})`);
27
+ }
28
+ if (entry.scope === 'config') throw new Error(`"${type}" is not tunable: its checks compare against Spark's own defaults`);
29
+ if (Object.keys(entry.thresholds).length === 0) throw new Error(`"${type}" has no thresholds to tune`);
30
+ return entry;
31
+ }
32
+
33
+ function checkValue(type , name , fallback , value ) {
34
+ if (!Array.isArray(fallback)) {
35
+ if (!isNonNegativeNumber(value)) throw new Error(`"${type}.${name}" must be a non-negative number`);
36
+ return value;
37
+ }
38
+ // Tier tables are indexed by position and read as ascending bands.
39
+ const ascending = Array.isArray(value) && value.length === fallback.length
40
+ && value.every(isNonNegativeNumber) && value.every((v, i) => i === 0 || v >= value[i - 1]);
41
+ if (!ascending) throw new Error(`"${type}.${name}" must be an ascending list of ${fallback.length} non-negative numbers`);
42
+ return Object.freeze([...value]);
43
+ }
44
+
45
+ /**
46
+ * Validates a parsed thresholds config: `{ "<detector type>": { "<threshold>": value } }`, with each
47
+ * value shaped like that threshold's default (see a report's `detectors` catalog for the names,
48
+ * units and defaults). Throws an Error naming the first problem; never drops a bad entry silently.
49
+ */
50
+ export function parseThresholdOverrides(raw ) {
51
+ if (!isPlainObject(raw)) throw new Error('expected a JSON object keyed by detector type');
52
+ const result = {};
53
+ for (const [type, perDetector] of Object.entries(raw)) {
54
+ const entry = tunableEntry(type);
55
+ if (!isPlainObject(perDetector)) throw new Error(`"${type}" must be an object of threshold values`);
56
+ const values = {};
57
+ for (const [name, value] of Object.entries(perDetector)) {
58
+ // Own keys only: an inherited name such as `constructor` or `toString` is no threshold.
59
+ if (!Object.hasOwn(entry.thresholds, name)) {
60
+ throw new Error(`unknown threshold "${type}.${name}" (${type} thresholds: ${Object.keys(entry.thresholds).join(', ')})`);
61
+ }
62
+ values[name] = checkValue(type, name, entry.thresholds[name], value);
63
+ }
64
+ result[type] = Object.freeze(values);
65
+ }
66
+ return Object.freeze(result) ;
67
+ }
68
+
69
+ /** The overrides for one entry, as the untyped map withThresholds() takes. */
70
+ export function overridesFor(entry , overrides ) {
71
+ return (overrides )?.[entry.type];
72
+ }
73
+
74
+ function sameValue(a , b ) {
75
+ if (Array.isArray(a) && Array.isArray(b)) return a.length === b.length && a.every((v, i) => v === b[i]);
76
+ return a === b;
77
+ }
78
+
79
+ /** The entry's thresholds that `overrides` moves off the default, or null when none does (no
80
+ * override, or one equal to the default: that run is the specification's). */
81
+ export function tunedThresholdsOf(entry , overrides ) {
82
+ const tuned = {};
83
+ for (const [name, value] of Object.entries(overridesFor(entry, overrides) ?? {})) {
84
+ if (!Object.hasOwn(entry.thresholds, name)) continue;
85
+ const fallback = entry.thresholds[name];
86
+ if (!sameValue(value, fallback)) tuned[name] = { value, default: fallback };
87
+ }
88
+ return Object.keys(tuned).length > 0 ? tuned : null;
89
+ }
90
+
91
+ /** The tuned thresholds an entry's findings carry: its own, plus its `suppressedBy` entry's
92
+ * (named `<suppressor>.<threshold>`), since tuning the suppressor changes which of them survive. */
93
+ export function findingTunedThresholds(entry , overrides ) {
94
+ const own = tunedThresholdsOf(entry, overrides);
95
+ const suppressor = entry.suppressedBy ? entries.find((d) => d.type === entry.suppressedBy) : undefined;
96
+ const bySuppressor = suppressor ? tunedThresholdsOf(suppressor, overrides) : null;
97
+ if (!suppressor || !bySuppressor) return own;
98
+ const prefixed = Object.fromEntries(Object.entries(bySuppressor).map(([name, t]) => [`${suppressor.type}.${name}`, t]));
99
+ return { ...own, ...prefixed };
100
+ }
101
+
102
+ /** findingTunedThresholds() for the first entry emitting finding type `type`, the entry whose
103
+ * thresholds its clean-check summary reads. */
104
+ export function tunedThresholdsForType(type , overrides ) {
105
+ const entry = ENTRY_BY_TYPE.get(type);
106
+ return entry ? findingTunedThresholds(entry, overrides) : null;
107
+ }
108
+
109
+ /** Every tuned detector's overridden thresholds, keyed by entry type, or null when none is tuned. */
110
+ export function tunedDetectors(overrides ) {
111
+ const byType = {};
112
+ for (const entry of entries) {
113
+ const tuned = tunedThresholdsOf(entry, overrides);
114
+ if (tuned) byType[entry.type] = tuned;
115
+ }
116
+ return Object.keys(byType).length > 0 ? byType : null;
117
+ }
118
+
119
+ /** The thresholds an entry runs with under `overrides`: its own, with any override merged in. */
120
+ export function effectiveThresholds(entry , overrides ) {
121
+ const own = overridesFor(entry, overrides);
122
+ return own ? { ...entry.thresholds, ...own } : entry.thresholds;
123
+ }
124
+
125
+ /** detectorCatalog() as a run under `overrides` used it: each row's effective thresholds, plus
126
+ * `tunedThresholds` on a row an override moved off its defaults. */
127
+ export function tunedDetectorCatalog(overrides ) {
128
+ return detectorCatalog().map((row, i) => {
129
+ const entry = entries[i];
130
+ const tuned = tunedThresholdsOf(entry, overrides);
131
+ return tuned ? { ...row, thresholds: effectiveThresholds(entry, overrides), tunedThresholds: tuned } : row;
132
+ });
133
+ }
134
+
135
+ function formatThresholdValue(value ) {
136
+ return Array.isArray(value) ? `[${value.join(', ')}]` : String(value);
137
+ }
138
+
139
+ /** "ratioWarn 5 (default 3), minTasksForP95 40 (default 20)". */
140
+ export function describeTunedThresholds(tuned ) {
141
+ return Object.entries(tuned)
142
+ .map(([name, { value, default: fallback }]) => `${name} ${formatThresholdValue(value)} (default ${formatThresholdValue(fallback)})`)
143
+ .join(', ');
144
+ }
145
+
146
+ /** A tuned run's report line after its label: every tuned detector's thresholds, then why its
147
+ * findings' estimates are uncalibrated. `byType` is tunedDetectors()'s result. */
148
+ export function tunedRunNote(byType ) {
149
+ const tuned = Object.entries(byType).map(([type, t]) => `${type} ${describeTunedThresholds(t)}`).join('; ');
150
+ return `${tuned}. Findings from these detectors are marked, and their impact estimates are uncalibrated: the estimates are calibrated against the default thresholds.`;
151
+ }
152
+
153
+ /** The caveat a tuned finding carries in `validationRequired`: which thresholds produced it and,
154
+ * when it has an estimate figure (`hasEstimate`), that the figure was never calibrated. */
155
+ export function tunedThresholdsNote(tuned , hasEstimate ) {
156
+ const label = `Produced with tuned thresholds: ${describeTunedThresholds(tuned)}.`;
157
+ return hasEstimate
158
+ ? `${label} Impact estimates are calibrated against the default thresholds, so this finding's estimate is unvalidated.`
159
+ : label;
160
+ }
@@ -1,35 +1,13 @@
1
- const THRESHOLD_SUMMARIES = {
2
- spill: 'single-task disk spill above 1 GiB',
3
- shuffle: 'shuffle read above the configured minimum byte threshold',
4
- skew: 'task duration skew above the configured ratio',
5
- gc: 'JVM GC time share above the configured ratio',
6
- slowHost: 'a host running 2x+ slower than its peers by mean task duration (per-executor byte/time dimensions use a separate, narrower ratio ladder starting at 1.33x; only those can reach critical on ratio alone)',
7
- stageSlowness: 'a stage running far longer than its peers, not attributable to a single slow host',
8
- straggler: 'one or more tasks finishing far after the rest of their stage',
9
- speculationWaste: 'speculative task attempts that completed after the original',
10
- tinyTask: 'median task duration below the configured floor',
11
- partitionSizing: 'partition byte size outside the configured target range',
12
- stageShape: 'low parallelism, data explosion, or task-count skew relative to core count',
13
- stageFailed: 'a stage that failed outright',
14
- failures: 'task failures above the configured rate',
15
- retryWaste: 'retried task attempts consuming executor time',
16
- coldStart: 'executor startup time above the configured floor',
17
- incompleteRun: 'an event log missing its terminal ApplicationEnd/job-completion event',
18
- utilization: 'core occupancy below the configured floor across the run',
19
- memoryUtilization: 'executor heap usage outside the configured band',
20
- cacheUtilization: 'cached partitions evicted or spilled to disk',
21
- coreLocality: 'task placement missing data-local core assignment',
22
- cachingOpportunity: 'a dataset re-read from source multiple times with no cache/persist',
23
- jobFailureRate: 'job failure rate above the configured threshold',
24
- autoscalingChurn: 'executor add/remove churn above the configured rate',
25
- configAudit: 'a Spark conf value outside the recommended range',
26
- duplicatePlanSubtree: 'the same physical plan subtree executed more than once',
27
- smallFiles: 'output files below the configured target size',
28
- overBroadcast: 'a broadcast join above the configured size ceiling',
29
- underBroadcast: 'a join below the configured size floor that skipped broadcast',
30
- broadcastSizing: 'a broadcast join outside the configured size range in either direction',
31
- };
1
+ import { ENTRY_BY_TYPE, } from './detectors.js';
2
+ import { presentationOf } from './finding-presentation.js';
3
+ import { effectiveThresholds } from './threshold-overrides.js';
32
4
 
33
- export function getThresholdSummary(type ) {
34
- return THRESHOLD_SUMMARIES[type] ?? 'criteria not met';
5
+ /** The clean-check criterion for a finding type, built from the thresholds of the first DETECTORS
6
+ * entry that emits it (configAudit's four entries share one summary), with `overrides` merged in
7
+ * when the run was tuned. */
8
+ export function getThresholdSummary(type , overrides ) {
9
+ const presentation = presentationOf(type);
10
+ const entry = ENTRY_BY_TYPE.get(type);
11
+ if (!presentation || !entry) return 'criteria not met';
12
+ return presentation.thresholdSummary(effectiveThresholds(entry, overrides) );
35
13
  }
@@ -1,3 +1,6 @@
1
+
2
+
3
+
1
4
 
2
5
 
3
6
 
@@ -287,48 +290,9 @@
287
290
 
288
291
 
289
292
 
290
-
291
-
292
-
293
-
294
-
295
-
296
-
297
-
298
-
299
-
300
-
301
-
302
-
303
-
304
-
305
-
306
-
307
-
308
-
309
-
310
-
311
-
312
-
313
-
314
-
315
-
316
-
317
-
318
-
319
-
320
-
321
-
322
-
323
-
324
-
325
-
326
-
327
-
328
-
329
-
330
-
331
-
293
+ // `Finding` is a union discriminated on `type`, one member per emitted finding type: see
294
+ // finding-types.ts for each detector's shape and which of its fields are public evidence.
295
+
332
296
 
333
297
 
334
298
 
@@ -1,6 +1,6 @@
1
1
  // Vendored from fflate@0.8.3 (esm/browser.js), MIT license.
2
2
  // https://github.com/101arrowz/fflate — sha256 b7ca4450b19559a1d50eb381adcee94b82449674be4cd17789d9beba7e6122a1
3
- // Only unzipSync/gunzipSync/zipSync/strToU8/strFromU8 are used by this project (see src/lz4-block.js, src/parser-worker.js).
3
+ // Only Gunzip/UnzipInflate/gunzipSync/strFromU8 are used by this project's source (see src/shs-fetch.ts, src/zip-archive.ts); tests also use the zip writers and strToU8.
4
4
  // DEFLATE is a complex format; to read this code, you should probably check the RFC first:
5
5
  // https://tools.ietf.org/html/rfc1951
6
6
  // You may also wish to take a look at the guide I made about this program:
@@ -1,15 +1,4 @@
1
- export function mergeIntervals(intervals ) {
2
- if (intervals.length === 0) return [];
3
- const sorted = [...intervals].sort((a, b) => a[0] - b[0]);
4
- const out = [[sorted[0][0], sorted[0][1]]];
5
- for (let i = 1; i < sorted.length; i++) {
6
- const last = out[out.length - 1];
7
- const cur = sorted[i];
8
- if (cur[0] <= last[1]) last[1] = Math.max(last[1], cur[1]);
9
- else out.push([cur[0], cur[1]]);
10
- }
11
- return out;
12
- }
1
+ import { mergeIntervals } from './intervals.js';
13
2
 
14
3
  export function computeWallClock(app , stages ) {
15
4
  const start = app?.startTime ?? 0;
@@ -3,11 +3,11 @@
3
3
  // core-time that actually ran tasks. Feeds a report widget only, no detector.
4
4
 
5
5
  import { computeTotalCores } from './core-count.js';
6
+ import { MS_PER_CORE_HOUR } from './format-utils.js';
6
7
 
7
- export const MS_PER_CORE_HOUR = 3.6e6;
8
8
  const TOP_N = 5;
9
9
 
10
-
10
+
11
11
 
12
12
 
13
13
 
@@ -0,0 +1,167 @@
1
+ // Random-access zip reader for Spark History Server log archives. Reads the
2
+ // central directory from the archive's tail, then inflates one entry at a time
3
+ // in bounded slices through fflate's streaming UnzipInflate: the source is read
4
+ // a slice at a time and no decompressed entry is ever held whole in memory.
5
+ //
6
+ // Why the central directory and not fflate's forward-scanning `Unzip`: the
7
+ // History Server writes its zip with Java's ZipOutputStream, which leaves
8
+ // every local header's sizes blank and puts them in a data descriptor after
9
+ // the entry. `Unzip` then has to find each entry's end by scanning the
10
+ // compressed bytes for the descriptor signature, and it hands rolling-log
11
+ // parts over in archive order, not the order they must be parsed in. The
12
+ // central directory has the real sizes and lets the caller pick the order.
13
+ import { UnzipInflate, strFromU8 } from './vendor/fflate.js';
14
+
15
+
16
+
17
+
18
+
19
+
20
+
21
+
22
+
23
+
24
+
25
+
26
+
27
+
28
+ const LOCAL_HEADER_SIG = 0x04034b50;
29
+ const CENTRAL_HEADER_SIG = 0x02014b50;
30
+ const EOCD_SIG = 0x06054b50;
31
+ const ZIP64_EOCD_LOCATOR_SIG = 0x07064b50;
32
+ const ZIP64_EOCD_SIG = 0x06064b50;
33
+ const ZIP64_EXTRA_ID = 0x0001;
34
+ const EOCD_MIN_SIZE = 22;
35
+ // The end-of-central-directory record ends with a comment of at most 65535 bytes.
36
+ const EOCD_MAX_SEARCH = EOCD_MIN_SIZE + 0xffff;
37
+ const UINT32_MAX = 0xffffffff;
38
+
39
+ // fflate's UnzipInflate is untyped vendor JS; this local shape types the call site.
40
+
41
+
42
+
43
+
44
+
45
+
46
+ const u16 = (d , i ) => d[i] | (d[i + 1] << 8);
47
+ const u32 = (d , i ) => (d[i] | (d[i + 1] << 8) | (d[i + 2] << 16) | (d[i + 3] << 24)) >>> 0;
48
+ const u64 = (d , i ) => u32(d, i) + u32(d, i + 4) * 2 ** 32;
49
+
50
+ async function readRange(source , start , end ) {
51
+ if (start < 0 || end > source.size || start > end) throw new Error('zip structure points outside the file');
52
+ return new Uint8Array(await source.slice(start, end).arrayBuffer());
53
+ }
54
+
55
+ /** True when `header` starts with a zip local-file header or an empty archive's EOCD record. */
56
+ export function isZip(header ) {
57
+ if (header.length < 4) return false;
58
+ const sig = u32(header, 0);
59
+ return sig === LOCAL_HEADER_SIG || sig === EOCD_SIG;
60
+ }
61
+
62
+ // Locates the central directory via the EOCD record (or its zip64 variant).
63
+ async function readDirectoryLocation(source ) {
64
+ const tailStart = Math.max(0, source.size - EOCD_MAX_SEARCH);
65
+ const tail = await readRange(source, tailStart, source.size);
66
+ let eocd = tail.length - EOCD_MIN_SIZE;
67
+ while (eocd >= 0 && u32(tail, eocd) !== EOCD_SIG) eocd--;
68
+ if (eocd < 0) throw new Error('no end-of-central-directory record');
69
+
70
+ const location = { count: u16(tail, eocd + 10), size: u32(tail, eocd + 12), offset: u32(tail, eocd + 16) };
71
+ const locator = tailStart + eocd - 20;
72
+ if (locator >= 0) {
73
+ const loc = await readRange(source, locator, locator + 20);
74
+ if (u32(loc, 0) === ZIP64_EOCD_LOCATOR_SIG) {
75
+ const zip64Offset = u64(loc, 8);
76
+ const z = await readRange(source, zip64Offset, zip64Offset + 56);
77
+ if (u32(z, 0) !== ZIP64_EOCD_SIG) throw new Error('bad zip64 end-of-central-directory record');
78
+ return { count: u64(z, 32), size: u64(z, 40), offset: u64(z, 48) };
79
+ }
80
+ }
81
+ return location;
82
+ }
83
+
84
+ // Reads the zip64 extended-information extra field, which carries (in this
85
+ // order) only the sizes/offset whose 32-bit central-directory field is 0xFFFFFFFF.
86
+ function applyZip64Extra(dir , extraStart , extraEnd , fields ) {
87
+ for (let p = extraStart; p + 4 <= extraEnd;) {
88
+ const id = u16(dir, p), len = u16(dir, p + 2);
89
+ if (id === ZIP64_EXTRA_ID) {
90
+ let q = p + 4;
91
+ for (const key of ['uncompressed', 'compressed', 'offset'] ) {
92
+ if (fields[key] === UINT32_MAX && q + 8 <= p + 4 + len) { fields[key] = u64(dir, q); q += 8; }
93
+ }
94
+ return;
95
+ }
96
+ p += 4 + len;
97
+ }
98
+ }
99
+
100
+ /** Lists every entry in the archive's central directory, in directory order. */
101
+ export async function listZipEntries(source ) {
102
+ const { count, offset, size } = await readDirectoryLocation(source);
103
+ const dir = await readRange(source, offset, offset + size);
104
+ const entries = [];
105
+ let p = 0;
106
+ for (let i = 0; i < count; i++) {
107
+ if (p + 46 > dir.length || u32(dir, p) !== CENTRAL_HEADER_SIG) throw new Error('bad central-directory header');
108
+ const nameLength = u16(dir, p + 28), extraLength = u16(dir, p + 30), commentLength = u16(dir, p + 32);
109
+ const utf8 = (u16(dir, p + 8) & 0x800) !== 0;
110
+ const nameStart = p + 46, extraStart = nameStart + nameLength;
111
+ const fields = { uncompressed: u32(dir, p + 24), compressed: u32(dir, p + 20), offset: u32(dir, p + 42) };
112
+ applyZip64Extra(dir, extraStart, extraStart + extraLength, fields);
113
+ entries.push({
114
+ name: strFromU8(dir.subarray(nameStart, extraStart), !utf8),
115
+ compression: u16(dir, p + 10),
116
+ compressedSize: fields.compressed,
117
+ localHeaderOffset: fields.offset,
118
+ });
119
+ p = extraStart + extraLength + commentLength;
120
+ }
121
+ return entries;
122
+ }
123
+
124
+ /**
125
+ * Streams one entry's decompressed bytes to `onChunk`, reading `chunkSize`
126
+ * compressed bytes at a time and awaiting `onChunk` before reading more, so an
127
+ * async consumer (an off-thread zstd decoder) applies backpressure. `onRead`
128
+ * reports compressed bytes consumed, for progress.
129
+ */
130
+ export async function streamZipEntry(
131
+ source ,
132
+ entry ,
133
+ onChunk ,
134
+ { chunkSize, onRead } ,
135
+ ) {
136
+ const header = await readRange(source, entry.localHeaderOffset, entry.localHeaderOffset + 30);
137
+ if (u32(header, 0) !== LOCAL_HEADER_SIG) throw new Error(`bad local header for "${entry.name}"`);
138
+ const dataStart = entry.localHeaderOffset + 30 + u16(header, 26) + u16(header, 28);
139
+ const dataEnd = dataStart + entry.compressedSize;
140
+ if (dataEnd > source.size) throw new Error(`"${entry.name}" is truncated`);
141
+ if (entry.compression !== 0 && entry.compression !== 8) {
142
+ throw new Error(`"${entry.name}" uses unsupported zip compression method ${entry.compression}`);
143
+ }
144
+
145
+ const pending = [];
146
+ let inflateError = null;
147
+ const inflater = entry.compression === 8 ? new (UnzipInflate )() : null;
148
+ if (inflater) {
149
+ inflater.ondata = (err, data) => {
150
+ if (err) inflateError = err;
151
+ else if (data.length) pending.push(data);
152
+ };
153
+ }
154
+
155
+ for (let offset = dataStart; offset < dataEnd;) {
156
+ const end = Math.min(dataEnd, offset + chunkSize);
157
+ const slice = await readRange(source, offset, end);
158
+ offset = end;
159
+ if (inflater) inflater.push(slice, offset >= dataEnd);
160
+ else pending.push(slice);
161
+ if (inflateError) throw inflateError;
162
+ onRead?.(slice.length);
163
+ // Inflate output chunks are views that fflate may reuse on the next push,
164
+ // so each is fully consumed here before the loop reads further.
165
+ while (pending.length) await onChunk(pending.shift() );
166
+ }
167
+ }