sparkforensics-mcp 0.2.4 → 0.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/LICENSE ADDED
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Agustin Recoba
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
package/README.md CHANGED
@@ -2,5 +2,11 @@
2
2
 
3
3
  MCP server exposing Apache Spark event-log diagnostics over stdio.
4
4
 
5
+ ```bash
6
+ npx sparkforensics-mcp
7
+ ```
8
+
9
+ Point your MCP client at that command.
10
+
5
11
  See the [main README](https://github.com/shuffle-works/sparkforensics#readme)
6
- for install instructions and the full tool reference.
12
+ for the full tool reference.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "sparkforensics-mcp",
3
- "version": "0.2.4",
3
+ "version": "0.3.0",
4
4
  "mcpName": "io.github.shuffle-works/sparkforensics-mcp",
5
5
  "type": "module",
6
6
  "license": "MIT",
@@ -343,8 +343,8 @@ export function createThreadedZstdDecoder(
343
343
  }
344
344
 
345
345
  // Node's native zstd where this Node has it (22.15+/23.8+); older Nodes keep the vendored fzstd.
346
- // runParse/runParseFiles (collectRun) take the threaded decoder. decodeShsArchive decodes each
347
- // archive entry in one synchronous call, so the SHS archive loader takes the inline one.
346
+ // runParse/runParseFiles (collectRun) take the threaded decoder; the SHS archive loader
347
+ // (decodeShsArchive over an in-memory download) takes the inline one.
348
348
  export const nodeParseCodecs =
349
349
  nativeZstdAvailable ? { zstdDecoder: createThreadedZstdDecoder } : {};
350
350
  export const nodeArchiveCodecs =
@@ -6,6 +6,7 @@ import { computeCoreLocalityRatio } from './core-locality-ratio.js';
6
6
  import { estimateSingleStage, tailRecoveryMs, tailRemovedWorkMs, stragglerFixLongestTaskMs, } from './occupancy.js';
7
7
  import { isExchangeNode, isBroadcastExchangeNode } from './plan-node-detail.js';
8
8
  import { cyrb53 } from './string-hash.js';
9
+ import { MAX_FAILURE_GROUPS, describeTaskFailure, } from './task-failure.js';
9
10
 
10
11
 
11
12
  const MB = 1024 * 1024;
@@ -73,6 +74,7 @@ const TB = 1024 * GB;
73
74
 
74
75
 
75
76
 
77
+
76
78
 
77
79
 
78
80
 
@@ -1206,7 +1208,7 @@ export const DETECTORS = [
1206
1208
  },
1207
1209
  },
1208
1210
  {
1209
- type: 'failures', scope: 'stage', order: 40, fixEffort: 'code', version: 1,
1211
+ type: 'failures', scope: 'stage', order: 40, fixEffort: 'code', version: 2,
1210
1212
  docAnchor: '#bottleneck-failures',
1211
1213
  thresholds: { minTasks: 10, warnRate: 0.05, critRate: 0.20 },
1212
1214
  detect(
@@ -1219,13 +1221,25 @@ export const DETECTORS = [
1219
1221
  if (failureRate <= this.thresholds.warnRate) return null;
1220
1222
  const value = Math.round(failureRate * 1000) / 10;
1221
1223
  const dominantReason = pickDominantReason(stage.failureReasons);
1224
+ // Groups arrive most frequent first. Name the dominant error from the largest group under the
1225
+ // dominant tag, so the error and the tag agree even when one tag splits into many messages.
1226
+ const allGroups = stage.failureGroups ?? [];
1227
+ const dominantGroup = allGroups.find((g) => g.reason === dominantReason);
1228
+ const dominantError = (dominantGroup ? describeTaskFailure(dominantGroup) : null) ?? dominantReason;
1229
+ const failureGroups = allGroups.slice(0, MAX_FAILURE_GROUPS);
1230
+ const groupedTasks = failureGroups.reduce((sum, g) => sum + g.count, 0);
1222
1231
  return {
1223
1232
  type: 'failures', stageId: stage.id,
1224
1233
  impactBand: failureRate > this.thresholds.critRate ? 'critical' : 'warning',
1225
1234
  metric: 'failureRate', value,
1226
1235
  failedTasks: stage.failedTasks,
1227
1236
  dominantReason,
1228
- recommendation: `${value}% of tasks failed${dominantReason ? ` (dominant reason: ${dominantReason})` : ''}: investigate driver logs for executor instability or data-driven errors.`,
1237
+ dominantError,
1238
+ // One entry per distinct error (tag, class, message, loss reason), each with one bounded
1239
+ // stack excerpt; otherFailedTasks counts the failed tasks no shown group covers.
1240
+ failureGroups,
1241
+ otherFailedTasks: Math.max(0, stage.failedTasks - groupedTasks),
1242
+ recommendation: `${value}% of tasks failed${dominantError ? ` (dominant error: ${dominantError})` : ''}: investigate driver logs for executor instability or data-driven errors.`,
1229
1243
  };
1230
1244
  },
1231
1245
  },
@@ -1,5 +1,9 @@
1
1
  ### `FAIL`: Failed tasks {#fail}
2
2
 
3
3
  Tasks fail often enough to affect the stage. Failed tasks point to executor
4
- instability or data-driven errors: check driver logs for the dominant
5
- failure reason.
4
+ instability or data-driven errors. The finding names the dominant error: the
5
+ exception class, or the executor loss reason (for example "Container killed
6
+ by YARN for exceeding memory limits"). It lists up to five distinct failures,
7
+ each with its message and a short stack excerpt. With redaction on, messages
8
+ and the message text inside excerpts are replaced, since they can carry file
9
+ paths and data values; class names and stack frames stay.
@@ -19,6 +19,7 @@ import {
19
19
  } from './event-schemas.js';
20
20
  import { assertNever } from './assert-never.js';
21
21
  import { finalizeStage } from './stage-quantiles.js';
22
+ import { MAX_FAILURE_DETAILS_PER_STAGE, extractTaskFailureDetail, taskFailureKey, } from './task-failure.js';
22
23
  import { computeRunAggregates } from './run-aggregates.js';
23
24
 
24
25
 
@@ -94,6 +95,9 @@ const MAX_TASK_SAMPLES = 20;
94
95
 
95
96
 
96
97
 
98
+
99
+
100
+
97
101
 
98
102
 
99
103
 
@@ -136,6 +140,8 @@ const MAX_TASK_SAMPLES = 20;
136
140
 
137
141
 
138
142
 
143
+
144
+
139
145
 
140
146
 
141
147
 
@@ -485,6 +491,19 @@ function taskRecordToSample(t ) {
485
491
  };
486
492
  }
487
493
 
494
+ // One shared detail object per distinct failure, so a stage with thousands of failed attempts
495
+ // holds each bounded stack excerpt once. Past the cap an attempt keeps only its reason tag.
496
+ function internTaskFailure(stage , endReason ) {
497
+ const detail = extractTaskFailureDetail(endReason);
498
+ if (!detail || !stage.failureDetails) return null;
499
+ const key = taskFailureKey(detail);
500
+ const known = stage.failureDetails.get(key);
501
+ if (known) return known;
502
+ if (stage.failureDetails.size >= MAX_FAILURE_DETAILS_PER_STAGE) return null;
503
+ stage.failureDetails.set(key, detail);
504
+ return detail;
505
+ }
506
+
488
507
  export function accumulateTask(event , state ) {
489
508
  const stageId = event['Stage ID'];
490
509
  const stage = state.stages.get(stageId);
@@ -521,6 +540,7 @@ export function accumulateTask(event , state
521
540
  launchTime: info['Launch Time'] ?? 0,
522
541
  finishTime: info['Finish Time'] ?? 0,
523
542
  reason: event['Task End Reason']?.['Reason'] ?? null,
543
+ failure: failed ? internTaskFailure(stage, event['Task End Reason']) : null,
524
544
  speculative: info['Speculative'] === true,
525
545
  host: info['Host'] ?? '',
526
546
  executorId: info['Executor ID'] ?? '',
@@ -802,6 +822,7 @@ export function submitStage(event , st
802
822
  failureReasons: new Map(),
803
823
  stageFailureReason: null,
804
824
  taskAttempts: new Map(),
825
+ failureDetails: new Map(),
805
826
  retryTaskSamples: [],
806
827
  retryWasteMs: 0,
807
828
  wastedAttempts: 0,
@@ -238,6 +238,14 @@ export const TaskEndEventSchema = z.object({
238
238
  'Stage Attempt ID': z.number().optional(),
239
239
  'Task End Reason': z.object({
240
240
  Reason: z.string().optional(),
241
+ // The error behind a failed attempt, read by extractTaskFailureDetail (task-failure.ts), which
242
+ // keeps only string values: unknown here so an unexpected shape drops the detail, never the task.
243
+ 'Class Name': z.unknown().optional(),
244
+ Description: z.unknown().optional(),
245
+ 'Full Stack Trace': z.unknown().optional(),
246
+ 'Loss Reason': z.unknown().optional(),
247
+ Message: z.unknown().optional(),
248
+ 'Kill Reason': z.unknown().optional(),
241
249
  }).optional(),
242
250
  'Task Info': z.object({
243
251
  'Launch Time': z.number().optional(),
@@ -6,6 +6,7 @@ import { detectorCatalog } from './detectors.js';
6
6
  import { typeTag, formatBytes, formatDuration, IMPACT_BAND_ORDER } from './format-utils.js';
7
7
  import { FINDING_NAMES, titleCase } from './finding-names.js';
8
8
  import { redactReport } from './redact.js';
9
+ import { formatTaskFailureHeadline, } from './task-failure.js';
9
10
  import { coreFindingActionLabel } from './finding-action-label.js';
10
11
  import { matchesFindingFilterCriteria } from './finding-filter-predicate.js';
11
12
  import { buildRecommendationRollup, isEligible, rankFindings, } from './recommendation-rollup.js';
@@ -269,6 +270,21 @@ function renderEvidenceValue(key , value ) {
269
270
  return String(value);
270
271
  }
271
272
 
273
+ // The `failures` finding's distinct errors: a headline per group, its stack excerpt as an indented
274
+ // code block (indented, not fenced, so no excerpt content can close it early).
275
+ function renderFailureGroups(groups ) {
276
+ const lines = [` - failureGroups: ${groups.length}`];
277
+ for (const g of groups) {
278
+ lines.push(` - ${g.count} task(s): ${formatTaskFailureHeadline(g)}`);
279
+ if (g.stackExcerpt) {
280
+ lines.push('');
281
+ for (const l of g.stackExcerpt.split('\n')) lines.push(` ${l}`);
282
+ lines.push('');
283
+ }
284
+ }
285
+ return lines;
286
+ }
287
+
272
288
  function formatWallClockRange(low , high ) {
273
289
  const fmtMs = (ms ) => (ms === 0 ? '0s' : formatDuration(ms));
274
290
  return low === high ? `Estimated ${fmtMs(high)}` : `Estimated ${fmtMs(low)}-${fmtMs(high)}`;
@@ -336,7 +352,10 @@ function renderMarkdown(json ) {
336
352
  const evidence = Object.entries(r.evidence ?? {}).sort(([a], [b]) => (a < b ? -1 : a > b ? 1 : 0));
337
353
  if (evidence.length) {
338
354
  lines.push('- evidence:');
339
- for (const [k, v] of evidence) lines.push(` - ${k}: ${renderEvidenceValue(k, v)}`);
355
+ for (const [k, v] of evidence) {
356
+ if (k === 'failureGroups' && Array.isArray(v)) lines.push(...renderFailureGroups(v ));
357
+ else lines.push(` - ${k}: ${renderEvidenceValue(k, v)}`);
358
+ }
340
359
  }
341
360
  lines.push('');
342
361
  }
@@ -4,7 +4,8 @@ import { Decompress as ZstdDecompress } from './vendor/fzstd.js';
4
4
  import { createSnappyBlockDecoder } from './snappy-block.js';
5
5
  import { createState, dispatchLine, buildChunkDecoder, emitParseCompletion, } from './event-handlers.js';
6
6
  import { TASK_FIELD_NAMES } from './stage-quantiles.js';
7
- import { runParseFromUrl, sniffCodec } from './shs-fetch.js';
7
+ import { runParseFromUrl, sniffCodec, parseZipArchive } from './shs-fetch.js';
8
+ import { isZip } from './zip-archive.js';
8
9
  import { createWorkerZstdDecoders } from './zstd-worker-client.js';
9
10
 
10
11
  export {
@@ -35,6 +36,8 @@ const MIN_PROGRESS_STEPS = 100;
35
36
  const PROGRESS_EMIT_LINES = 300;
36
37
 
37
38
 
39
+ const NOT_AN_EVENT_LOG = 'Not a Spark event log: no application-start event found. Choose a Spark event log file, or check the docs for supported formats.';
40
+
38
41
  // zstdDecoder replaces the vendored fzstd for zstd input; the Node CLI/MCP path passes
39
42
  // cli/native-zstd.ts's native-zlib decoder, which a browser bundle can't import.
40
43
 
@@ -140,6 +143,20 @@ export async function runParse(
140
143
  return;
141
144
  }
142
145
 
146
+ // A Spark History Server download (the UI's download link, or GET
147
+ // /api/v1/applications/<id>/logs) is a zip holding the log file, or a
148
+ // rolling log's parts: unwrap it through the same path the SHS fetch uses.
149
+ if (isZip(new Uint8Array(await file.slice(0, Math.min(4, file.size)).arrayBuffer()))) {
150
+ await parseZipArchive(file, state, emit, {
151
+ zstdDecoder,
152
+ chunkSize,
153
+ progressEvery: PROGRESS_EMIT_LINES,
154
+ reportPct: true,
155
+ onInvalid: (detail) => emit({ type: 'error', message: detail ?? NOT_AN_EVENT_LOG }),
156
+ });
157
+ return;
158
+ }
159
+
143
160
  const decoder = buildChunkDecoder();
144
161
  const joined = [];
145
162
  let linesProcessed = 0;
@@ -168,7 +185,7 @@ export async function runParse(
168
185
  }
169
186
 
170
187
  if (!state.app) {
171
- emit({ type: 'error', message: 'Not a Spark event log: no application-start event found. Choose a Spark event log file, or check the docs for supported formats.' });
188
+ emit({ type: 'error', message: NOT_AN_EVENT_LOG });
172
189
  return;
173
190
  }
174
191
 
@@ -227,7 +244,7 @@ export async function runParseFiles(
227
244
  }
228
245
 
229
246
  if (!state.app) {
230
- emit({ type: 'error', message: 'Not a Spark event log: no application-start event found. Choose a Spark event log file, or check the docs for supported formats.' });
247
+ emit({ type: 'error', message: NOT_AN_EVENT_LOG });
231
248
  return;
232
249
  }
233
250
 
@@ -242,7 +259,7 @@ if (isWorker) {
242
259
  let workerState = null;
243
260
  // Dropped zstd files decompress in a second worker, overlapping with parsing here. The
244
261
  // `new Worker(new URL(...))` stays inline for Vite's worker detection (see ingest.ts). The SHS
245
- // path (runParseFromUrl) decodes whole zip entries synchronously and keeps in-thread fzstd.
262
+ // path (runParseFromUrl) keeps in-thread fzstd.
246
263
  const zstdDecoder = createWorkerZstdDecoders(
247
264
  () => new Worker(new URL('./zstd-worker.js', import.meta.url), { type: 'module' }),
248
265
  (onChunk) => new (ZstdDecompress )(onChunk),
@@ -8,6 +8,7 @@
8
8
  // and non-mutating (returns a fresh, deep-copied tree).
9
9
 
10
10
 
11
+ import { redactTaskFailureGroup, } from './task-failure.js';
11
12
 
12
13
  // Host / IP identifier patterns. Used to enumerate host names that surface only
13
14
  // inside free text: recommendation strings, `stageFailed`'s failure-reason
@@ -79,6 +80,25 @@ function collectHostFields(node , hosts ) {
79
80
  }
80
81
  }
81
82
 
83
+ // A failed-task error's message and stack text can carry file paths and data values that no
84
+ // host/app-id pattern recognizes, so they are dropped rather than pseudonymized: every array under
85
+ // a key named `failureGroups` (a `failures` finding's, or a stage's in the HTML export) gets
86
+ // redactTaskFailureGroup applied. Walking by key name, like collectHostFields, needs no path list.
87
+ // Returns a fresh tree.
88
+ function redactFailureGroups (node ) {
89
+ if (Array.isArray(node)) return node.map((n) => redactFailureGroups(n)) ;
90
+ if (node && typeof node === 'object') {
91
+ const out = {};
92
+ for (const [k, v] of Object.entries(node)) {
93
+ out[k] = k === 'failureGroups' && Array.isArray(v)
94
+ ? v.map((g) => (g && typeof g === 'object' ? redactTaskFailureGroup(g ) : g))
95
+ : redactFailureGroups(v);
96
+ }
97
+ return out ;
98
+ }
99
+ return node;
100
+ }
101
+
82
102
  // Spark config keys ending in `host`/`hostname` (e.g. spark.driver.host,
83
103
  // spark.yarn.am.hostname) carry plain FQDN host names that neither
84
104
  // HOST_PATTERNS matches (no IP/EC2 shape) nor collectHostFields's by-key-name
@@ -172,7 +192,8 @@ function applyReplacements (node , ids
172
192
  return deepReplace(node, merged) ;
173
193
  }
174
194
 
175
- export function redactReport (report ) {
195
+ export function redactReport (input ) {
196
+ const report = redactFailureGroups(input);
176
197
  const { appIds, hosts } = collectIds(report);
177
198
  return applyReplacements(report, { appIds, hosts });
178
199
  }
@@ -216,7 +237,8 @@ export function redactComparison (comparison ) {
216
237
  // this also walks executors.added/removed for their literal `host` field
217
238
  // (ExecutorAddedEvent.host), since raw executor records: not just findings
218
239
  //: reach data.js.
219
- export function redactExportData(data ) {
240
+ export function redactExportData(input ) {
241
+ const data = redactFailureGroups(input);
220
242
  const appIds = new Set ();
221
243
  const hosts = new Set ();
222
244
  const appId = data.app?.id;
@@ -180,6 +180,14 @@ function skewRatios(stages ) {
180
180
  return out;
181
181
  }
182
182
 
183
+ // Every key metricDeltas() emits, in emission order. The CLI validates
184
+ // --regression-metric against it, so a misspelled key is a usage error rather
185
+ // than an inconclusive budget. A contract test keeps it in sync.
186
+ export const COMPARISON_METRIC_KEYS = [
187
+ 'wallClock', 'shuffleSpill', 'taskSkew', 'failedTaskRate', 'diskSpill', 'gcTime',
188
+ 'inputBytes', 'outputBytes', 'executorRunTime', 'taskCount', 'executorsAdded',
189
+ ];
190
+
183
191
  // Volume/count metrics, not cost metrics: more or less input/output data, or
184
192
  // tasks/executors, isn't inherently better or worse (it may just reflect a
185
193
  // differently-sized job), unlike wall-clock, spill, GC, etc. Exported as the
@@ -1,4 +1,4 @@
1
- import { unzipSync, Gunzip } from './vendor/fflate.js';
1
+ import { Gunzip } from './vendor/fflate.js';
2
2
  import { createLz4BlockDecoder } from './lz4-block.js';
3
3
  import { Decompress as ZstdDecompress } from './vendor/fzstd.js';
4
4
  import { createSnappyBlockDecoder } from './snappy-block.js';
@@ -6,6 +6,8 @@ import { buildProxyRequestUrl, isShsErrorCode } from './shs-request.js';
6
6
  import { dispatchLine, buildChunkDecoder, emitParseCompletion, } from './event-handlers.js';
7
7
  import { ShsProxyErrorBodySchema } from './shs-schemas.js';
8
8
  import { naturalCompare, reassembleRollingEntries } from './rolling-log-reassembly.js';
9
+ import { listZipEntries, streamZipEntry, } from './zip-archive.js';
10
+
9
11
 
10
12
  export { naturalCompare, reassembleRollingEntries };
11
13
 
@@ -24,48 +26,84 @@ export function sniffCodec(bytes )
24
26
  return null;
25
27
  }
26
28
 
27
- // Streams one already-in-memory zip entry through its codec's block-by-block
28
- // decoder, invoking `onChunk` per decompressed piece; never materializes the
29
- // whole decompressed entry as one buffer. A giant single-segment zstd frame
30
- // (or gzip member) declares its full content size in the header; one-shotting
31
- // it (fzstd's `decompress()`, fflate's `gunzipSync()`) allocates that entire
32
- // size as a single ArrayBuffer up front, which can exceed what the browser
33
- // will allocate for a multi-GB event log ("Array buffer allocation failed").
29
+ // Decodes one zip entry's (possibly compressed) bytes as they stream out of
30
+ // the archive, block by block, invoking `onChunk` per decompressed piece;
31
+ // never materializes the whole decompressed entry as one buffer. A giant
32
+ // single-segment zstd frame (or gzip member) declares its full content size in
33
+ // the header; one-shotting it (fzstd's `decompress()`, fflate's `gunzipSync()`)
34
+ // allocates that entire size as a single ArrayBuffer up front, which can exceed
35
+ // what the browser will allocate for a multi-GB event log ("Array buffer
36
+ // allocation failed"). The codec is sniffed from the entry's first 8 bytes,
37
+ // falling back to the entry name's suffix.
34
38
  // fflate's Gunzip and fzstd's Decompress are untyped vendor JS, so TS can't
35
39
  // infer a construct signature for them; this local shape types the call sites
36
40
  // without touching the vendored files.
37
41
 
38
42
 
39
- // Like parser-worker.ts's RunOpts.zstdDecoder, but synchronous: decodeEntry never awaits push(),
40
- // so its result is typed `undefined` (not `void`, which would also accept an async decoder's
41
- // Promise). Node callers pass cli/native-zstd.ts's createNativeZstdDecoder (nodeArchiveCodecs).
42
-
43
-
44
- function decodeEntry(
45
- name , raw , onChunk , zstdDecoder ,
46
- ) {
47
- const codec = sniffCodec(raw);
43
+
44
+
45
+ function openCodecSink(
46
+ codec , name , onChunk , zstdDecoder ,
47
+ ) {
48
48
  if (codec === 'lz4' || name.endsWith('.lz4')) {
49
49
  const lz4 = createLz4BlockDecoder(onChunk);
50
- lz4.push(raw);
51
- lz4.end();
52
- } else if (codec === 'gz' || name.endsWith('.gz')) {
53
- new (Gunzip )(onChunk).push(raw, true);
54
- } else if (codec === 'zstd' || name.endsWith('.zstd') || name.endsWith('.zst')) {
55
- (zstdDecoder ? zstdDecoder(onChunk) : new (ZstdDecompress )(onChunk)).push(raw, true);
56
- } else if (codec === 'snappy' || name.endsWith('.snappy')) {
50
+ return { push(chunk, final) { lz4.push(chunk); if (final) lz4.end(); } };
51
+ }
52
+ if (codec === 'gz' || name.endsWith('.gz')) {
53
+ const gunzip = new (Gunzip )(onChunk);
54
+ return { push: (chunk, final) => gunzip.push(chunk, final) };
55
+ }
56
+ if (codec === 'zstd' || name.endsWith('.zstd') || name.endsWith('.zst')) {
57
+ const zstd = zstdDecoder ? zstdDecoder(onChunk) : new (ZstdDecompress )(onChunk);
58
+ return { push: (chunk, final) => zstd.push(chunk, final), cancel: () => (zstd ).cancel?.() };
59
+ }
60
+ if (codec === 'snappy' || name.endsWith('.snappy')) {
57
61
  const snappy = createSnappyBlockDecoder(onChunk);
58
- snappy.push(raw);
59
- snappy.end();
60
- } else {
61
- onChunk(raw);
62
+ return { push(chunk, final) { snappy.push(chunk); if (final) snappy.end(); } };
62
63
  }
64
+ return { push: (chunk) => onChunk(chunk) };
65
+ }
66
+
67
+ // Wraps openCodecSink for a stream of unknown length: holds back the latest
68
+ // chunk so the last one can be pushed with `final` set (codecs reject a
69
+ // trailing empty push less uniformly than a real last chunk), and buffers the
70
+ // start of the entry until it has the 8 bytes sniffCodec needs.
71
+ function createEntryDecoder(name , onChunk , zstdDecoder ) {
72
+ let head = new Uint8Array(0);
73
+ let held = null;
74
+ let sink = null;
75
+ const open = () => {
76
+ sink = openCodecSink(sniffCodec(head), name, onChunk, zstdDecoder);
77
+ held = head;
78
+ };
79
+ return {
80
+ async push(chunk ) {
81
+ if (!sink) {
82
+ const joined = new Uint8Array(head.length + chunk.length);
83
+ joined.set(head);
84
+ joined.set(chunk, head.length);
85
+ head = joined;
86
+ if (head.length >= 8) open();
87
+ return;
88
+ }
89
+ await sink.push(held , false);
90
+ held = chunk;
91
+ },
92
+ async end() {
93
+ if (!sink) {
94
+ if (!head.length) return;
95
+ open();
96
+ }
97
+ await sink .push(held , true);
98
+ },
99
+ cancel() { sink?.cancel?.(); },
100
+ };
63
101
  }
64
102
 
65
103
 
66
104
 
67
- function emitShsError(emit , code ) {
68
- emit({ type: 'error', source: 'shs', code });
105
+ function emitShsError(emit , code , message ) {
106
+ emit({ type: 'error', source: 'shs', code, ...(message ? { message } : {}) });
69
107
  }
70
108
 
71
109
  export async function runParseFromUrl(
@@ -141,55 +179,127 @@ export async function runParseFromUrl(
141
179
  if (total) {
142
180
  emit({ type: 'progress', pct: 0.5, linesProcessed: 0 });
143
181
  }
144
- decodeShsArchive(zipBytes, state, emit);
182
+ await decodeShsArchive(zipBytes, state, emit);
183
+ }
184
+
185
+ // Reads the in-memory SHS download through the same random-access interface a
186
+ // dropped File offers, so both paths share parseZipArchive.
187
+ function bytesZipSource(bytes ) {
188
+ return {
189
+ size: bytes.length,
190
+ slice: (start, end) => ({ arrayBuffer: async () => bytes.slice(start, end).buffer }),
191
+ };
145
192
  }
146
193
 
147
194
  export function decodeShsArchive(
148
195
  zipBytes , state , emit , { zstdDecoder } = {},
149
- ) {
150
- let entries ;
196
+ ) {
197
+ return parseZipArchive(bytesZipSource(zipBytes), state, emit, {
198
+ zstdDecoder,
199
+ onInvalid: (detail) => emitShsError(emit, 'invalid-event-log', detail ?? undefined),
200
+ });
201
+ }
202
+
203
+ const baseName = (e ) => e.name.slice(e.name.lastIndexOf('/') + 1);
204
+ const isRollingPart = (e ) => /^events_\d+_/.test(baseName(e));
205
+
206
+ // The application attempts a History Server zip holds. Without an attempt ID
207
+ // the History Server packs every attempt into one download: each single-file
208
+ // log is its own root entry, each rolling log its own `eventlog_v2_` directory.
209
+ // Directory entries and `appstatus` markers are not attempts.
210
+ function countLogAttempts(entries ) {
211
+ const attempts = new Set ();
212
+ for (const e of entries) {
213
+ if (e.name.endsWith('/') || baseName(e).toLowerCase().startsWith('appstatus')) continue;
214
+ const dir = e.name.slice(0, e.name.indexOf('/') + 1);
215
+ attempts.add(dir || (isRollingPart(e) ? '' : e.name));
216
+ }
217
+ return attempts.size;
218
+ }
219
+
220
+ // The event-log entries a single-attempt History Server zip holds, in parse
221
+ // order. A rolling log's parts sit under an `eventlog_v2_<appId>/` directory
222
+ // entry; they are matched and reassembled by base name. A single-file log is
223
+ // one entry at the root. Directory entries and the rolling log's `appstatus`
224
+ // marker are skipped.
225
+ function selectLogEntries(entries ) {
226
+ const files = entries.filter((e) => !e.name.endsWith('/'));
227
+ if (files.some(isRollingPart)) {
228
+ const byBase = new Map(files.map((e) => [baseName(e), e]));
229
+ return reassembleRollingEntries(files.map(baseName)).map((n) => byBase.get(n) );
230
+ }
231
+ return files.filter((e) => baseName(e).toLowerCase() !== 'appstatus');
232
+ }
233
+
234
+
235
+
236
+
237
+
238
+
239
+
240
+
241
+
242
+
243
+
244
+
245
+
246
+
247
+ // Parses a single-entry or rolling Spark History Server zip, streaming each
248
+ // entry out of `source` in `chunkSize` slices. One NDJSON line decoder spans
249
+ // all entries, since a file-roll boundary need not fall on a line boundary.
250
+ export async function parseZipArchive(
251
+ source ,
252
+ state ,
253
+ emit ,
254
+ { zstdDecoder, chunkSize = 512 * 1024, onInvalid, progressEvery = 2000, reportPct = false } ,
255
+ ) {
256
+ let logEntries ;
257
+ let attempts ;
151
258
  try {
152
- entries = unzipSync(zipBytes);
153
- } catch {
154
- emitShsError(emit, 'invalid-event-log');
259
+ const entries = await listZipEntries(source);
260
+ attempts = countLogAttempts(entries);
261
+ logEntries = attempts > 1 ? [] : selectLogEntries(entries);
262
+ } catch (e) {
263
+ onInvalid(`Could not read the zip archive: ${e instanceof Error ? e.message : String(e)}`);
155
264
  return;
156
265
  }
157
- const allNames = Object.keys(entries);
158
- const isRolling = allNames.some(n => /^events_\d+_/.test(n));
159
- let names;
160
- if (isRolling) {
161
- try {
162
- names = reassembleRollingEntries(allNames);
163
- } catch {
164
- emitShsError(emit, 'invalid-event-log');
165
- return;
166
- }
167
- } else {
168
- names = allNames.filter(n => n.toLowerCase() !== 'appstatus').sort(naturalCompare);
266
+ if (attempts > 1) {
267
+ onInvalid(`The zip archive holds ${attempts} application attempts. Download a single attempt, for example GET /api/v1/applications/<appId>/<attemptId>/logs.`);
268
+ return;
169
269
  }
170
- if (names.length === 0) {
171
- emitShsError(emit, 'invalid-event-log');
270
+ if (logEntries.length === 0) {
271
+ onInvalid('The zip archive contains no event log.');
172
272
  return;
173
273
  }
174
274
 
275
+ const totalBytes = logEntries.reduce((sum, e) => sum + e.compressedSize, 0);
276
+ let bytesRead = 0;
175
277
  const decoder = buildChunkDecoder();
176
278
  const joined = [];
177
279
  let linesProcessed = 0;
178
- for (const name of names) {
280
+ const feed = (bytes ) => {
281
+ joined.length = 0;
282
+ const lines = decoder.decode(bytes, joined);
283
+ for (let i = 0, j = 0; i < lines.length; i++) {
284
+ dispatchLine(lines[i], state, emit, joined[j]?.index === i ? joined[j++] : undefined);
285
+ linesProcessed++;
286
+ if (linesProcessed % progressEvery === 0) {
287
+ emit({ type: 'progress', pct: reportPct && totalBytes ? bytesRead / totalBytes : null, linesProcessed });
288
+ }
289
+ }
290
+ };
291
+
292
+ for (const entry of logEntries) {
293
+ const entryDecoder = createEntryDecoder(entry.name, feed, zstdDecoder);
179
294
  try {
180
- decodeEntry(name, entries[name], (bytes) => {
181
- joined.length = 0;
182
- const lines = decoder.decode(bytes, joined);
183
- for (let i = 0, j = 0; i < lines.length; i++) {
184
- dispatchLine(lines[i], state, emit, joined[j]?.index === i ? joined[j++] : undefined);
185
- linesProcessed++;
186
- if (linesProcessed % 2000 === 0) {
187
- emit({ type: 'progress', pct: null, linesProcessed });
188
- }
189
- }
190
- }, zstdDecoder);
191
- } catch {
192
- emitShsError(emit, 'invalid-event-log');
295
+ await streamZipEntry(source, entry, (chunk) => entryDecoder.push(chunk), {
296
+ chunkSize,
297
+ onRead: (bytes) => { bytesRead += bytes; },
298
+ });
299
+ await entryDecoder.end();
300
+ } catch (e) {
301
+ entryDecoder.cancel();
302
+ onInvalid(`Could not decompress "${entry.name}" in the zip archive: ${e instanceof Error ? e.message : String(e)}`);
193
303
  return;
194
304
  }
195
305
  }
@@ -198,7 +308,7 @@ export function decodeShsArchive(
198
308
  }
199
309
 
200
310
  if (!state.app) {
201
- emitShsError(emit, 'invalid-event-log');
311
+ onInvalid(null);
202
312
  return;
203
313
  }
204
314
  emitParseCompletion(state, emit, linesProcessed);
@@ -19,7 +19,7 @@ export const DEFAULT_IDLE_TIMEOUT_MS = envInt('SPARKFORENSICS_SHS_TIMEOUT_MS', 3
19
19
 
20
20
  function collectShsAppModel(zipBytes ) {
21
21
  return collectViaDispatch(
22
- (state, emit) => decodeShsArchive(zipBytes, state, emit, nodeArchiveCodecs),
22
+ (state, emit, reject) => { decodeShsArchive(zipBytes, state, emit, nodeArchiveCodecs).catch(reject); },
23
23
  (msg) => {
24
24
  const m = msg ;
25
25
  return mcpError(m?.code ?? 'invalid-event-log', m?.message ?? 'Failed to decode SHS archive.');
@@ -1,4 +1,5 @@
1
1
 
2
+
2
3
 
3
4
  // Single source of truth for the packed per-task numeric array: FIELDS (offset
4
5
  // constants), TASK_FIELD_NAMES (display labels), and the finalizeStage hot-loop
@@ -74,6 +75,8 @@ export function finalizeStage(
74
75
  const executorStats = new Map();
75
76
  const failureReasons = new Map();
76
77
  const failedTaskSamples = [];
78
+ // Keyed by the interned detail object: accumulateTask shares one per distinct failure.
79
+ const failureGroups = new Map ();
77
80
  const localityStats = new Map();
78
81
  let peakExecutionMemoryMax = 0;
79
82
 
@@ -82,6 +85,8 @@ export function finalizeStage(
82
85
  if (t.failed) {
83
86
  failedTasks++;
84
87
  if (t.reason) failureReasons.set(t.reason, (failureReasons.get(t.reason) ?? 0) + 1);
88
+ const failure = (t ).failure;
89
+ if (failure) failureGroups.set(failure, (failureGroups.get(failure) ?? 0) + 1);
85
90
  if (failedTaskSamples.length < MAX_TASK_SAMPLES) {
86
91
  failedTaskSamples.push({
87
92
  taskId: t.taskId, attemptNumber: t.attemptNumber, host: t.host, executorId: t.executorId,
@@ -125,6 +130,7 @@ export function finalizeStage(
125
130
  stage.failedTasks = failedTasks;
126
131
  stage.speculativeTasks = speculativeTasks;
127
132
  stage.taskAttempts = null; // no longer needed after finalize, freeing memory
133
+ stage.failureDetails = null;
128
134
 
129
135
  const arr = new Float64Array(buf);
130
136
  state.taskStore.set(stageId, arr);
@@ -164,6 +170,10 @@ export function finalizeStage(
164
170
  const failureReasonsArr = [...failureReasons.entries()].map(
165
171
  ([reason, count]) => ({ reason, count })
166
172
  );
173
+ // Most frequent first; ties keep first-seen order (Array.prototype.sort is stable).
174
+ const failureGroupsArr = [...failureGroups.entries()]
175
+ .map(([detail, count]) => ({ ...detail, count }))
176
+ .sort((a, b) => b.count - a.count);
167
177
  const localityStatsArr = [...localityStats.entries()].map(
168
178
  ([locality, count]) => ({ locality, count })
169
179
  );
@@ -173,6 +183,7 @@ export function finalizeStage(
173
183
  hostStats: hostStatsArr, executorStats: executorStatsArr, failureReasons: failureReasonsArr, localityStats: localityStatsArr, stragglerCount, stragglerExcessMs, longestNonStragglerMs,
174
184
  tailReplayRecoveryMs,
175
185
  failedTaskSamples,
186
+ failureGroups: failureGroupsArr,
176
187
  peakExecutionMemoryMax,
177
188
  taskActiveMs: computeTaskActiveMs(arr),
178
189
  peakConcurrentTasks,
@@ -187,6 +198,7 @@ export function finalizeStage(
187
198
  stageType: acc.shuffleReadBytes > 0 ? 'REDUCE' : 'MAP',
188
199
  };
189
200
  delete data.taskAttempts; // internal-only field, already nulled above; never part of the public message
201
+ delete data.failureDetails; // internal-only intern table, summarized by failureGroups
190
202
 
191
203
  return { type: 'stage', data };
192
204
  }
@@ -0,0 +1,151 @@
1
+ // What actually went wrong behind a failed task attempt, read from its TaskEnd "Task End Reason".
2
+ // Spark's `Reason` is only the end-reason tag (ExceptionFailure, ExecutorLostFailure, ...); the
3
+ // real error lives in tag-specific fields (JsonProtocol.taskEndReasonToJson):
4
+ // ExceptionFailure Class Name, Description, Full Stack Trace
5
+ // ExecutorLostFailure Loss Reason (e.g. "Container killed by YARN for exceeding memory limits")
6
+ // FetchFailed Message (often a whole exception string, stack included)
7
+ // TaskKilled Kill Reason
8
+ // Every text field is bounded here, at ingest, so neither parser memory nor a report grows with
9
+ // the size of the traces Spark wrote.
10
+
11
+
12
+
13
+
14
+
15
+
16
+
17
+
18
+
19
+
20
+
21
+
22
+
23
+ const MAX_TEXT_CHARS = 300;
24
+ const MAX_EXCERPT_FRAMES = 8;
25
+ const MAX_EXCERPT_LINE_CHARS = 300;
26
+ const MAX_EXCERPT_CHARS = 2000;
27
+ // Distinct failures kept per stage while parsing, and shown per finding.
28
+ export const MAX_FAILURE_DETAILS_PER_STAGE = 50;
29
+ export const MAX_FAILURE_GROUPS = 5;
30
+
31
+ export const REDACTED_TEXT = '[redacted]';
32
+
33
+ function truncate(s , max ) {
34
+ return s.length > max ? `${s.slice(0, max - 3)}...` : s;
35
+ }
36
+
37
+ function text(v ) {
38
+ return typeof v === 'string' && v.trim().length > 0 ? v : null;
39
+ }
40
+
41
+ // First non-blank line, whitespace collapsed: a message can carry a whole stack trace after it.
42
+ function firstLine(s ) {
43
+ if (s == null) return null;
44
+ const line = s.split('\n').map((l) => l.trim()).find((l) => l.length > 0);
45
+ return line ? truncate(line.replace(/\s+/g, ' '), MAX_TEXT_CHARS) : null;
46
+ }
47
+
48
+ const PYTHON_TRACEBACK_HEAD = /(?:^|: )(?:Traceback \(most recent call last\):|An exception was thrown from the Python worker)/;
49
+ const JAVA_FRAME = /^\s*at \S/;
50
+
51
+ // A PySpark error's text is a whole Python traceback whose own error line (`ValueError: bad row`)
52
+ // comes last, before any JVM frames: index of that line among non-blank `lines`, or -1.
53
+ function pythonErrorLineIndex(lines ) {
54
+ if (lines.length < 2 || !PYTHON_TRACEBACK_HEAD.test(lines[0].trim())) return -1;
55
+ let last = -1;
56
+ for (let i = 1; i < lines.length && !JAVA_FRAME.test(lines[i]); i++) last = i;
57
+ return last;
58
+ }
59
+
60
+ // The message's headline: its first line, or a Python traceback's error line.
61
+ function messageLine(s ) {
62
+ if (s == null) return null;
63
+ const lines = s.split('\n').filter((l) => l.trim().length > 0);
64
+ const pyIdx = pythonErrorLineIndex(lines);
65
+ return pyIdx >= 0 ? firstLine(lines[pyIdx]) : firstLine(s);
66
+ }
67
+
68
+ /** Header line, the first frames, and (when cut off) a Python traceback's error line and the last
69
+ * `Caused by:` line, which usually name the root cause. Bounded by frame count, line length and
70
+ * total length. */
71
+ export function buildStackExcerpt(trace ) {
72
+ if (trace == null) return null;
73
+ const lines = trace.split('\n').map((l) => l.replace(/\s+$/, '')).filter((l) => l.trim().length > 0);
74
+ if (lines.length === 0) return null;
75
+ const headCount = 1 + MAX_EXCERPT_FRAMES;
76
+ const kept = lines.slice(0, headCount);
77
+ if (lines.length > headCount) {
78
+ const pyIdx = pythonErrorLineIndex(lines);
79
+ let causeIdx = -1;
80
+ for (let i = lines.length - 1; i >= headCount; i--) {
81
+ if (lines[i].startsWith('Caused by:')) { causeIdx = i; break; }
82
+ }
83
+ if (pyIdx >= headCount) kept.push('\t...', lines[pyIdx]);
84
+ if (pyIdx < lines.length - 1) kept.push('\t...');
85
+ if (causeIdx >= 0) kept.push(...lines.slice(causeIdx, causeIdx + 2));
86
+ }
87
+ return truncate(kept.map((l) => truncate(l, MAX_EXCERPT_LINE_CHARS)).join('\n'), MAX_EXCERPT_CHARS);
88
+ }
89
+
90
+ /** Bounded failure detail for a failed attempt's end reason, or null when there is no end reason. */
91
+ export function extractTaskFailureDetail(endReason ) {
92
+ if (!endReason) return null;
93
+ const reason = text(endReason['Reason']);
94
+ const rawMessage = text(endReason['Description']) ?? text(endReason['Message']) ?? text(endReason['Kill Reason']);
95
+ // FetchFailed has no Full Stack Trace field, but its Message is a full exception string.
96
+ const trace = text(endReason['Full Stack Trace'])
97
+ ?? (rawMessage != null && rawMessage.trim().includes('\n') ? rawMessage : null);
98
+ return {
99
+ reason,
100
+ className: firstLine(text(endReason['Class Name'])),
101
+ message: messageLine(rawMessage),
102
+ lossReason: firstLine(text(endReason['Loss Reason'])),
103
+ stackExcerpt: buildStackExcerpt(trace),
104
+ };
105
+ }
106
+
107
+ /** Grouping key: one group per distinct error. The excerpt is left out, so two traces of the same
108
+ * error share a group and the first one seen is kept. */
109
+ export function taskFailureKey(d ) {
110
+ return JSON.stringify([d.reason, d.className, d.message, d.lossReason]);
111
+ }
112
+
113
+ /** Short name for the error: the exception class, or the end-reason tag with its loss reason. */
114
+ export function describeTaskFailure(d ) {
115
+ if (d.className) return d.className;
116
+ if (d.lossReason) return d.reason ? `${d.reason}: ${d.lossReason}` : d.lossReason;
117
+ return d.reason;
118
+ }
119
+
120
+ const FRAME_LINE = /^\s*(?:at \S.*|\.\.\.(?: \d+ more)?)$/;
121
+ const CLASS_HEADER = /^((?:Caused by: |Suppressed: )?[\w$]+(?:\.[\w$]+)+)(?::.*)?$/;
122
+
123
+ // Keeps stack frames and exception class names only: a message (the text after "Class: ", and any
124
+ // continuation line such as a Python traceback's `File "/path"` lines) is dropped.
125
+ function stripExcerptMessages(excerpt ) {
126
+ const kept = [];
127
+ for (const line of excerpt.split('\n')) {
128
+ if (FRAME_LINE.test(line)) { kept.push(line); continue; }
129
+ const header = CLASS_HEADER.exec(line.trim());
130
+ if (header) kept.push(header[1]);
131
+ }
132
+ return kept.join('\n');
133
+ }
134
+
135
+ /** Redacted copy of a failure group: the message and the message text inside the stack excerpt
136
+ * can carry file paths and data values. Class names, frames and the loss reason are kept (hosts in
137
+ * a loss reason are pseudonymized by the caller's host scan like any other free text). */
138
+ export function redactTaskFailureGroup (g ) {
139
+ return {
140
+ ...g,
141
+ message: g.message == null ? null : REDACTED_TEXT,
142
+ stackExcerpt: g.stackExcerpt == null ? null : stripExcerptMessages(g.stackExcerpt),
143
+ };
144
+ }
145
+
146
+ /** One-line headline for a failure group, for reports and the Failed Tasks widget. */
147
+ export function formatTaskFailureHeadline(d ) {
148
+ const name = d.className ?? d.reason ?? 'Unknown failure';
149
+ const detail = [d.message, d.lossReason].filter((s) => s != null).join(' · ');
150
+ return detail ? `${name}: ${detail}` : name;
151
+ }
@@ -1,6 +1,6 @@
1
1
  // Vendored from fflate@0.8.3 (esm/browser.js), MIT license.
2
2
  // https://github.com/101arrowz/fflate — sha256 b7ca4450b19559a1d50eb381adcee94b82449674be4cd17789d9beba7e6122a1
3
- // Only unzipSync/gunzipSync/zipSync/strToU8/strFromU8 are used by this project (see src/lz4-block.js, src/parser-worker.js).
3
+ // Only Gunzip/UnzipInflate/gunzipSync/strFromU8 are used by this project's source (see src/shs-fetch.ts, src/zip-archive.ts); tests also use the zip writers and strToU8.
4
4
  // DEFLATE is a complex format; to read this code, you should probably check the RFC first:
5
5
  // https://tools.ietf.org/html/rfc1951
6
6
  // You may also wish to take a look at the guide I made about this program:
@@ -0,0 +1,167 @@
1
+ // Random-access zip reader for Spark History Server log archives. Reads the
2
+ // central directory from the archive's tail, then inflates one entry at a time
3
+ // in bounded slices through fflate's streaming UnzipInflate: the source is read
4
+ // a slice at a time and no decompressed entry is ever held whole in memory.
5
+ //
6
+ // Why the central directory and not fflate's forward-scanning `Unzip`: the
7
+ // History Server writes its zip with Java's ZipOutputStream, which leaves
8
+ // every local header's sizes blank and puts them in a data descriptor after
9
+ // the entry. `Unzip` then has to find each entry's end by scanning the
10
+ // compressed bytes for the descriptor signature, and it hands rolling-log
11
+ // parts over in archive order, not the order they must be parsed in. The
12
+ // central directory has the real sizes and lets the caller pick the order.
13
+ import { UnzipInflate, strFromU8 } from './vendor/fflate.js';
14
+
15
+
16
+
17
+
18
+
19
+
20
+
21
+
22
+
23
+
24
+
25
+
26
+
27
+
28
+ const LOCAL_HEADER_SIG = 0x04034b50;
29
+ const CENTRAL_HEADER_SIG = 0x02014b50;
30
+ const EOCD_SIG = 0x06054b50;
31
+ const ZIP64_EOCD_LOCATOR_SIG = 0x07064b50;
32
+ const ZIP64_EOCD_SIG = 0x06064b50;
33
+ const ZIP64_EXTRA_ID = 0x0001;
34
+ const EOCD_MIN_SIZE = 22;
35
+ // The end-of-central-directory record ends with a comment of at most 65535 bytes.
36
+ const EOCD_MAX_SEARCH = EOCD_MIN_SIZE + 0xffff;
37
+ const UINT32_MAX = 0xffffffff;
38
+
39
+ // fflate's UnzipInflate is untyped vendor JS; this local shape types the call site.
40
+
41
+
42
+
43
+
44
+
45
+
46
+ const u16 = (d , i ) => d[i] | (d[i + 1] << 8);
47
+ const u32 = (d , i ) => (d[i] | (d[i + 1] << 8) | (d[i + 2] << 16) | (d[i + 3] << 24)) >>> 0;
48
+ const u64 = (d , i ) => u32(d, i) + u32(d, i + 4) * 2 ** 32;
49
+
50
+ async function readRange(source , start , end ) {
51
+ if (start < 0 || end > source.size || start > end) throw new Error('zip structure points outside the file');
52
+ return new Uint8Array(await source.slice(start, end).arrayBuffer());
53
+ }
54
+
55
+ /** True when `header` starts with a zip local-file header or an empty archive's EOCD record. */
56
+ export function isZip(header ) {
57
+ if (header.length < 4) return false;
58
+ const sig = u32(header, 0);
59
+ return sig === LOCAL_HEADER_SIG || sig === EOCD_SIG;
60
+ }
61
+
62
+ // Locates the central directory via the EOCD record (or its zip64 variant).
63
+ async function readDirectoryLocation(source ) {
64
+ const tailStart = Math.max(0, source.size - EOCD_MAX_SEARCH);
65
+ const tail = await readRange(source, tailStart, source.size);
66
+ let eocd = tail.length - EOCD_MIN_SIZE;
67
+ while (eocd >= 0 && u32(tail, eocd) !== EOCD_SIG) eocd--;
68
+ if (eocd < 0) throw new Error('no end-of-central-directory record');
69
+
70
+ const location = { count: u16(tail, eocd + 10), size: u32(tail, eocd + 12), offset: u32(tail, eocd + 16) };
71
+ const locator = tailStart + eocd - 20;
72
+ if (locator >= 0) {
73
+ const loc = await readRange(source, locator, locator + 20);
74
+ if (u32(loc, 0) === ZIP64_EOCD_LOCATOR_SIG) {
75
+ const zip64Offset = u64(loc, 8);
76
+ const z = await readRange(source, zip64Offset, zip64Offset + 56);
77
+ if (u32(z, 0) !== ZIP64_EOCD_SIG) throw new Error('bad zip64 end-of-central-directory record');
78
+ return { count: u64(z, 32), size: u64(z, 40), offset: u64(z, 48) };
79
+ }
80
+ }
81
+ return location;
82
+ }
83
+
84
+ // Reads the zip64 extended-information extra field, which carries (in this
85
+ // order) only the sizes/offset whose 32-bit central-directory field is 0xFFFFFFFF.
86
+ function applyZip64Extra(dir , extraStart , extraEnd , fields ) {
87
+ for (let p = extraStart; p + 4 <= extraEnd;) {
88
+ const id = u16(dir, p), len = u16(dir, p + 2);
89
+ if (id === ZIP64_EXTRA_ID) {
90
+ let q = p + 4;
91
+ for (const key of ['uncompressed', 'compressed', 'offset'] ) {
92
+ if (fields[key] === UINT32_MAX && q + 8 <= p + 4 + len) { fields[key] = u64(dir, q); q += 8; }
93
+ }
94
+ return;
95
+ }
96
+ p += 4 + len;
97
+ }
98
+ }
99
+
100
+ /** Lists every entry in the archive's central directory, in directory order. */
101
+ export async function listZipEntries(source ) {
102
+ const { count, offset, size } = await readDirectoryLocation(source);
103
+ const dir = await readRange(source, offset, offset + size);
104
+ const entries = [];
105
+ let p = 0;
106
+ for (let i = 0; i < count; i++) {
107
+ if (p + 46 > dir.length || u32(dir, p) !== CENTRAL_HEADER_SIG) throw new Error('bad central-directory header');
108
+ const nameLength = u16(dir, p + 28), extraLength = u16(dir, p + 30), commentLength = u16(dir, p + 32);
109
+ const utf8 = (u16(dir, p + 8) & 0x800) !== 0;
110
+ const nameStart = p + 46, extraStart = nameStart + nameLength;
111
+ const fields = { uncompressed: u32(dir, p + 24), compressed: u32(dir, p + 20), offset: u32(dir, p + 42) };
112
+ applyZip64Extra(dir, extraStart, extraStart + extraLength, fields);
113
+ entries.push({
114
+ name: strFromU8(dir.subarray(nameStart, extraStart), !utf8),
115
+ compression: u16(dir, p + 10),
116
+ compressedSize: fields.compressed,
117
+ localHeaderOffset: fields.offset,
118
+ });
119
+ p = extraStart + extraLength + commentLength;
120
+ }
121
+ return entries;
122
+ }
123
+
124
+ /**
125
+ * Streams one entry's decompressed bytes to `onChunk`, reading `chunkSize`
126
+ * compressed bytes at a time and awaiting `onChunk` before reading more, so an
127
+ * async consumer (an off-thread zstd decoder) applies backpressure. `onRead`
128
+ * reports compressed bytes consumed, for progress.
129
+ */
130
+ export async function streamZipEntry(
131
+ source ,
132
+ entry ,
133
+ onChunk ,
134
+ { chunkSize, onRead } ,
135
+ ) {
136
+ const header = await readRange(source, entry.localHeaderOffset, entry.localHeaderOffset + 30);
137
+ if (u32(header, 0) !== LOCAL_HEADER_SIG) throw new Error(`bad local header for "${entry.name}"`);
138
+ const dataStart = entry.localHeaderOffset + 30 + u16(header, 26) + u16(header, 28);
139
+ const dataEnd = dataStart + entry.compressedSize;
140
+ if (dataEnd > source.size) throw new Error(`"${entry.name}" is truncated`);
141
+ if (entry.compression !== 0 && entry.compression !== 8) {
142
+ throw new Error(`"${entry.name}" uses unsupported zip compression method ${entry.compression}`);
143
+ }
144
+
145
+ const pending = [];
146
+ let inflateError = null;
147
+ const inflater = entry.compression === 8 ? new (UnzipInflate )() : null;
148
+ if (inflater) {
149
+ inflater.ondata = (err, data) => {
150
+ if (err) inflateError = err;
151
+ else if (data.length) pending.push(data);
152
+ };
153
+ }
154
+
155
+ for (let offset = dataStart; offset < dataEnd;) {
156
+ const end = Math.min(dataEnd, offset + chunkSize);
157
+ const slice = await readRange(source, offset, end);
158
+ offset = end;
159
+ if (inflater) inflater.push(slice, offset >= dataEnd);
160
+ else pending.push(slice);
161
+ if (inflateError) throw inflateError;
162
+ onRead?.(slice.length);
163
+ // Inflate output chunks are views that fflate may reuse on the next push,
164
+ // so each is fully consumed here before the loop reads further.
165
+ while (pending.length) await onChunk(pending.shift() );
166
+ }
167
+ }