sparkforensics-mcp 0.2.2 → 0.2.4
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/bin/sparkforensics-mcp.mjs +11 -5
- package/package.json +3 -3
- package/vendor-core/cli/collect-run.js +3 -2
- package/vendor-core/cli/native-zstd.js +351 -0
- package/vendor-core/detectors.js +243 -53
- package/vendor-core/docs-config.js +34 -8
- package/vendor-core/docs-content/chapters/03-memory-model.md +39 -0
- package/vendor-core/docs-content/chapters/11-cluster-config.md +40 -0
- package/vendor-core/docs-content/detection/gc.md +2 -0
- package/vendor-core/docs-content/detection/host.md +2 -1
- package/vendor-core/docs-content/detection/plan.md +3 -1
- package/vendor-core/docs-content/detection/shape.md +2 -1
- package/vendor-core/docs-content/detection/shfl.md +2 -1
- package/vendor-core/docs-content/detection/spill.md +1 -1
- package/vendor-core/docs-content/detection/strag.md +2 -1
- package/vendor-core/docs-content/detection/tiny.md +2 -1
- package/vendor-core/docs-content/tuning/failures.md +1 -1
- package/vendor-core/docs-content/tuning/gc.md +11 -4
- package/vendor-core/docs-content/tuning/shuffle.md +25 -5
- package/vendor-core/docs-content/tuning/skew.md +14 -6
- package/vendor-core/docs-content/tuning/small-files.md +12 -7
- package/vendor-core/docs-content/tuning/straggler.md +34 -0
- package/vendor-core/docs-content/tuning/tiny-tasks.md +1 -1
- package/vendor-core/docs-content/tuning/utilization.md +57 -7
- package/vendor-core/docs-content/upstream.json +4 -0
- package/vendor-core/event-handlers.js +321 -69
- package/vendor-core/event-schemas.js +8 -6
- package/vendor-core/evidence-report.js +3 -1
- package/vendor-core/impact-estimator.js +170 -34
- package/vendor-core/mcp-tools.js +20 -7
- package/vendor-core/occupancy.js +71 -2
- package/vendor-core/parser-worker.js +56 -24
- package/vendor-core/plan-summary.js +5 -1
- package/vendor-core/run-comparison.js +1 -15
- package/vendor-core/shs-fetch.js +18 -7
- package/vendor-core/shs-load.js +2 -1
- package/vendor-core/stage-quantiles.js +111 -3
- package/vendor-core/string-hash.js +15 -0
- package/vendor-core/types.js +14 -1
- package/vendor-core/vendor/fzstd.js +94 -18
- package/vendor-core/zstd-worker-client.js +180 -0
- package/vendor-core/zstd-worker.js +103 -0
|
@@ -2,9 +2,10 @@ import { Gunzip } from './vendor/fflate.js';
|
|
|
2
2
|
import { createLz4BlockDecoder } from './lz4-block.js';
|
|
3
3
|
import { Decompress as ZstdDecompress } from './vendor/fzstd.js';
|
|
4
4
|
import { createSnappyBlockDecoder } from './snappy-block.js';
|
|
5
|
-
import { createState, dispatchLine, buildChunkDecoder, emitParseCompletion,
|
|
5
|
+
import { createState, dispatchLine, buildChunkDecoder, emitParseCompletion, } from './event-handlers.js';
|
|
6
6
|
import { TASK_FIELD_NAMES } from './stage-quantiles.js';
|
|
7
7
|
import { runParseFromUrl, sniffCodec } from './shs-fetch.js';
|
|
8
|
+
import { createWorkerZstdDecoders } from './zstd-worker-client.js';
|
|
8
9
|
|
|
9
10
|
export {
|
|
10
11
|
buildChunkDecoder, createState, normalizeSparkProperties, parseSparkMemoryMB, extractResources,
|
|
@@ -34,7 +35,9 @@ const MIN_PROGRESS_STEPS = 100;
|
|
|
34
35
|
const PROGRESS_EMIT_LINES = 300;
|
|
35
36
|
|
|
36
37
|
|
|
37
|
-
|
|
38
|
+
// zstdDecoder replaces the vendored fzstd for zstd input; the Node CLI/MCP path passes
|
|
39
|
+
// cli/native-zstd.ts's native-zlib decoder, which a browser bundle can't import.
|
|
40
|
+
|
|
38
41
|
|
|
39
42
|
// Minimal shape streamFile/runParse/runParseFiles read off `file` (name, size,
|
|
40
43
|
// slice(start,end).arrayBuffer()): narrower than the full DOM `File`. A real
|
|
@@ -59,6 +62,12 @@ const PROGRESS_EMIT_LINES = 300;
|
|
|
59
62
|
// without touching the vendored files (mirrors shs-fetch.ts's shim).
|
|
60
63
|
|
|
61
64
|
|
|
65
|
+
// A Node decoder, or the browser's decompress worker (zstd-worker-client.ts), may decompress off
|
|
66
|
+
// the calling thread: streamFile awaits each push, and calls cancel() when it abandons the stream.
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
|
|
62
71
|
|
|
63
72
|
// Stream one File's (possibly compressed) bytes through the codec dispatch,
|
|
64
73
|
// in `chunkSize` slices, invoking `onChunk` with each decompressed buffer as
|
|
@@ -69,6 +78,7 @@ export async function streamFile(
|
|
|
69
78
|
file ,
|
|
70
79
|
onChunk ,
|
|
71
80
|
chunkSize ,
|
|
81
|
+
zstdDecoder ,
|
|
72
82
|
) {
|
|
73
83
|
const header = new Uint8Array(await file.slice(0, Math.min(8, file.size)).arrayBuffer());
|
|
74
84
|
const codec = sniffCodec(header);
|
|
@@ -81,7 +91,10 @@ export async function streamFile(
|
|
|
81
91
|
let currentPct = 0;
|
|
82
92
|
const gunzip = codec === 'gz' ? new (Gunzip )((inflated) => onChunk(inflated, currentPct)) : null;
|
|
83
93
|
const lz4 = codec === 'lz4' ? createLz4BlockDecoder((inflated) => onChunk(inflated, currentPct)) : null;
|
|
84
|
-
const
|
|
94
|
+
const onZstdChunk = (inflated ) => onChunk(inflated, currentPct);
|
|
95
|
+
const zstd = codec !== 'zstd' ? null
|
|
96
|
+
: zstdDecoder ? zstdDecoder(onZstdChunk)
|
|
97
|
+
: new (ZstdDecompress )(onZstdChunk);
|
|
85
98
|
const snappy = codec === 'snappy' ? createSnappyBlockDecoder((inflated) => onChunk(inflated, currentPct)) : null;
|
|
86
99
|
|
|
87
100
|
// A fixed read size gives too few progress checkpoints on smaller files
|
|
@@ -91,17 +104,22 @@ export async function streamFile(
|
|
|
91
104
|
const stepSize = Math.max(1, Math.min(chunkSize, Math.ceil(file.size / MIN_PROGRESS_STEPS)));
|
|
92
105
|
|
|
93
106
|
let offset = 0;
|
|
94
|
-
|
|
95
|
-
|
|
96
|
-
|
|
97
|
-
|
|
98
|
-
|
|
99
|
-
|
|
100
|
-
|
|
101
|
-
|
|
102
|
-
|
|
103
|
-
|
|
104
|
-
|
|
107
|
+
try {
|
|
108
|
+
while (offset < file.size) {
|
|
109
|
+
const start = offset;
|
|
110
|
+
const slice = new Uint8Array(await file.slice(start, start + stepSize).arrayBuffer());
|
|
111
|
+
offset += stepSize;
|
|
112
|
+
const final = offset >= file.size;
|
|
113
|
+
currentPct = start / file.size;
|
|
114
|
+
if (gunzip) gunzip.push(slice, final);
|
|
115
|
+
else if (lz4) lz4.push(slice);
|
|
116
|
+
else if (zstd) await zstd.push(slice, final);
|
|
117
|
+
else if (snappy) snappy.push(slice);
|
|
118
|
+
else onChunk(slice, currentPct);
|
|
119
|
+
}
|
|
120
|
+
} catch (e) {
|
|
121
|
+
zstd?.cancel?.();
|
|
122
|
+
throw e;
|
|
105
123
|
}
|
|
106
124
|
if (lz4) lz4.end();
|
|
107
125
|
if (snappy) snappy.end();
|
|
@@ -115,7 +133,7 @@ export async function streamFile(
|
|
|
115
133
|
export async function runParse(
|
|
116
134
|
file ,
|
|
117
135
|
state ,
|
|
118
|
-
{ emit = (msg ) => self.postMessage(msg), chunkSize = CHUNK_SIZE } = {},
|
|
136
|
+
{ emit = (msg ) => self.postMessage(msg), chunkSize = CHUNK_SIZE, zstdDecoder } = {},
|
|
119
137
|
) {
|
|
120
138
|
if (file.size === 0) {
|
|
121
139
|
emit({ type: 'error', message: 'File is empty.' });
|
|
@@ -123,10 +141,13 @@ export async function runParse(
|
|
|
123
141
|
}
|
|
124
142
|
|
|
125
143
|
const decoder = buildChunkDecoder();
|
|
144
|
+
const joined = [];
|
|
126
145
|
let linesProcessed = 0;
|
|
127
146
|
const feed = (bytes , pct ) => {
|
|
128
|
-
|
|
129
|
-
|
|
147
|
+
joined.length = 0;
|
|
148
|
+
const lines = decoder.decode(bytes, joined);
|
|
149
|
+
for (let i = 0, j = 0; i < lines.length; i++) {
|
|
150
|
+
dispatchLine(lines[i], state, emit, joined[j]?.index === i ? joined[j++] : undefined);
|
|
130
151
|
linesProcessed++;
|
|
131
152
|
if (linesProcessed % PROGRESS_EMIT_LINES === 0) {
|
|
132
153
|
emit({ type: 'progress', pct: pct ?? null, linesProcessed });
|
|
@@ -135,7 +156,7 @@ export async function runParse(
|
|
|
135
156
|
};
|
|
136
157
|
|
|
137
158
|
try {
|
|
138
|
-
await streamFile(file, feed, chunkSize);
|
|
159
|
+
await streamFile(file, feed, chunkSize, zstdDecoder);
|
|
139
160
|
} catch (e) {
|
|
140
161
|
const message = e instanceof Error ? e.message : String(e);
|
|
141
162
|
emit({ type: 'error', message: `Could not decompress "${file.name}": ${message}` });
|
|
@@ -163,7 +184,7 @@ export async function runParse(
|
|
|
163
184
|
export async function runParseFiles(
|
|
164
185
|
files ,
|
|
165
186
|
state ,
|
|
166
|
-
{ emit = (msg ) => self.postMessage(msg), chunkSize = CHUNK_SIZE } = {},
|
|
187
|
+
{ emit = (msg ) => self.postMessage(msg), chunkSize = CHUNK_SIZE, zstdDecoder } = {},
|
|
167
188
|
) {
|
|
168
189
|
if (files.length === 0) {
|
|
169
190
|
emit({ type: 'error', message: 'Rolling event-log directory contained no event files.' });
|
|
@@ -171,13 +192,16 @@ export async function runParseFiles(
|
|
|
171
192
|
}
|
|
172
193
|
|
|
173
194
|
const decoder = buildChunkDecoder();
|
|
195
|
+
const joined = [];
|
|
174
196
|
let linesProcessed = 0;
|
|
175
197
|
const totalSize = files.reduce((sum, f) => sum + f.size, 0);
|
|
176
198
|
let bytesBeforeCurrentFile = 0;
|
|
177
199
|
let currentFileSize = 0;
|
|
178
200
|
const feed = (bytes , pct ) => {
|
|
179
|
-
|
|
180
|
-
|
|
201
|
+
joined.length = 0;
|
|
202
|
+
const lines = decoder.decode(bytes, joined);
|
|
203
|
+
for (let i = 0, j = 0; i < lines.length; i++) {
|
|
204
|
+
dispatchLine(lines[i], state, emit, joined[j]?.index === i ? joined[j++] : undefined);
|
|
181
205
|
linesProcessed++;
|
|
182
206
|
if (linesProcessed % PROGRESS_EMIT_LINES === 0) {
|
|
183
207
|
const overallPct = totalSize > 0 ? (bytesBeforeCurrentFile + (pct ?? 0) * currentFileSize) / totalSize : null;
|
|
@@ -189,7 +213,7 @@ export async function runParseFiles(
|
|
|
189
213
|
for (const file of files) {
|
|
190
214
|
currentFileSize = file.size;
|
|
191
215
|
try {
|
|
192
|
-
await streamFile(file, feed, chunkSize);
|
|
216
|
+
await streamFile(file, feed, chunkSize, zstdDecoder);
|
|
193
217
|
} catch (e) {
|
|
194
218
|
const message = e instanceof Error ? e.message : String(e);
|
|
195
219
|
emit({ type: 'error', message: `Could not decompress "${file.name}": ${message}` });
|
|
@@ -216,17 +240,25 @@ const isWorker = typeof WorkerGlobalScope !== 'undefined' && self instanceof Wor
|
|
|
216
240
|
|
|
217
241
|
if (isWorker) {
|
|
218
242
|
let workerState = null;
|
|
243
|
+
// Dropped zstd files decompress in a second worker, overlapping with parsing here. The
|
|
244
|
+
// `new Worker(new URL(...))` stays inline for Vite's worker detection (see ingest.ts). The SHS
|
|
245
|
+
// path (runParseFromUrl) decodes whole zip entries synchronously and keeps in-thread fzstd.
|
|
246
|
+
const zstdDecoder = createWorkerZstdDecoders(
|
|
247
|
+
() => new Worker(new URL('./zstd-worker.js', import.meta.url), { type: 'module' }),
|
|
248
|
+
(onChunk) => new (ZstdDecompress )(onChunk),
|
|
249
|
+
{ onFallback: (reason) => console.warn(`zstd: decompressing on the parse worker: ${reason}`) },
|
|
250
|
+
);
|
|
219
251
|
|
|
220
252
|
self.onmessage = async ({ data } ) => {
|
|
221
253
|
if (data.type === 'parse') {
|
|
222
254
|
workerState = createState();
|
|
223
|
-
await runParse(data.file, workerState);
|
|
255
|
+
await runParse(data.file, workerState, { zstdDecoder });
|
|
224
256
|
} else if (data.type === 'parseFromUrl') {
|
|
225
257
|
workerState = createState();
|
|
226
258
|
await runParseFromUrl(data.request, workerState);
|
|
227
259
|
} else if (data.type === 'parseFiles') {
|
|
228
260
|
workerState = createState();
|
|
229
|
-
await runParseFiles(data.files, workerState);
|
|
261
|
+
await runParseFiles(data.files, workerState, { zstdDecoder });
|
|
230
262
|
} else if (data.type === 'getTaskData') {
|
|
231
263
|
const { stageId, reqId } = data;
|
|
232
264
|
const stored = workerState?.taskStore.get(stageId) ?? new Float64Array(0);
|
|
@@ -45,7 +45,7 @@ function pushLongFilterWarning(result , len ) {
|
|
|
45
45
|
}
|
|
46
46
|
|
|
47
47
|
// Stable relation-identity key for a scan node: "<format>:<relation>" (e.g.
|
|
48
|
-
// "delta:mx.
|
|
48
|
+
// "delta:mx.store_map", "parquet:warehouse.sales", "jdbc:dw.dim_product")
|
|
49
49
|
// or null when the node is internal Delta metadata / a non-scan / un-nameable.
|
|
50
50
|
// The single source of truth for scan identity, shared by `visitScan` (summary)
|
|
51
51
|
// and the `cachingOpportunity` detector so the regexes live in one place.
|
|
@@ -55,6 +55,10 @@ export function scanRelationId(name , detail ) {
|
|
|
55
55
|
// Delta tables). All catalog tables surface as `spark_catalog.<db>.<table>`.
|
|
56
56
|
let format = null, table = null;
|
|
57
57
|
const nameM = name.match(/^Scan\s+(parquet|orc|csv|json)\s+(spark_catalog\.\S+)$/i);
|
|
58
|
+
// Fast reject: every branch below that returns a key needs this name match, a FileScan
|
|
59
|
+
// detail or a JDBCRelation detail. Most plan nodes (Project, Filter, joins) have none, and
|
|
60
|
+
// their long details otherwise pay every regex below.
|
|
61
|
+
if (!nameM && !/FileScan/i.test(detail) && !detail.includes('JDBCRelation')) return null;
|
|
58
62
|
if (nameM) { format = nameM[1].toLowerCase(); table = nameM[2]; }
|
|
59
63
|
else {
|
|
60
64
|
const detM = detail.match(/FileScan\s+(parquet|orc|csv|json)\s+(spark_catalog\.[^[\s]+)\[/i);
|
|
@@ -1,5 +1,6 @@
|
|
|
1
1
|
import { computeWallClock } from './wall-clock.js';
|
|
2
2
|
import { normalizeDetail } from './detectors.js';
|
|
3
|
+
import { cyrb53 } from './string-hash.js';
|
|
3
4
|
import { captureSnapshot } from './session-snapshot.js';
|
|
4
5
|
|
|
5
6
|
|
|
@@ -35,21 +36,6 @@ export function normalizeStageName(name ) {
|
|
|
35
36
|
.trim();
|
|
36
37
|
}
|
|
37
38
|
|
|
38
|
-
// cyrb53: fast, non-cryptographic, deterministic 53-bit string hash. Not a
|
|
39
|
-
// security boundary; 53 bits keeps collision risk negligible at the per-node,
|
|
40
|
-
// per-stage call volume planTreeIdentity/sqlNodeIdentity put through it.
|
|
41
|
-
function cyrb53(str , seed = 0) {
|
|
42
|
-
let h1 = 0xdeadbeef ^ seed, h2 = 0x41c6ce57 ^ seed;
|
|
43
|
-
for (let i = 0; i < str.length; i++) {
|
|
44
|
-
const ch = str.charCodeAt(i);
|
|
45
|
-
h1 = Math.imul(h1 ^ ch, 2654435761);
|
|
46
|
-
h2 = Math.imul(h2 ^ ch, 1597334677);
|
|
47
|
-
}
|
|
48
|
-
h1 = Math.imul(h1 ^ (h1 >>> 16), 2246822507) ^ Math.imul(h2 ^ (h2 >>> 13), 3266489909);
|
|
49
|
-
h2 = Math.imul(h2 ^ (h2 >>> 16), 2246822507) ^ Math.imul(h1 ^ (h1 >>> 13), 3266489909);
|
|
50
|
-
return (4294967296 * (2097151 & h2) + (h1 >>> 0)).toString(16);
|
|
51
|
-
}
|
|
52
|
-
|
|
53
39
|
// Bottom-up, order-independent structural identity of a resolved plan tree:
|
|
54
40
|
// each node folds its normalized name/detail with its children's digests
|
|
55
41
|
// (children sorted, so AQE picking a different broadcast side still matches),
|
package/vendor-core/shs-fetch.js
CHANGED
|
@@ -3,7 +3,7 @@ import { createLz4BlockDecoder } from './lz4-block.js';
|
|
|
3
3
|
import { Decompress as ZstdDecompress } from './vendor/fzstd.js';
|
|
4
4
|
import { createSnappyBlockDecoder } from './snappy-block.js';
|
|
5
5
|
import { buildProxyRequestUrl, isShsErrorCode } from './shs-request.js';
|
|
6
|
-
import { dispatchLine, buildChunkDecoder, emitParseCompletion,
|
|
6
|
+
import { dispatchLine, buildChunkDecoder, emitParseCompletion, } from './event-handlers.js';
|
|
7
7
|
import { ShsProxyErrorBodySchema } from './shs-schemas.js';
|
|
8
8
|
import { naturalCompare, reassembleRollingEntries } from './rolling-log-reassembly.js';
|
|
9
9
|
|
|
@@ -36,8 +36,14 @@ export function sniffCodec(bytes )
|
|
|
36
36
|
// without touching the vendored files.
|
|
37
37
|
|
|
38
38
|
|
|
39
|
+
// Like parser-worker.ts's RunOpts.zstdDecoder, but synchronous: decodeEntry never awaits push(),
|
|
40
|
+
// so its result is typed `undefined` (not `void`, which would also accept an async decoder's
|
|
41
|
+
// Promise). Node callers pass cli/native-zstd.ts's createNativeZstdDecoder (nodeArchiveCodecs).
|
|
42
|
+
|
|
39
43
|
|
|
40
|
-
function decodeEntry(
|
|
44
|
+
function decodeEntry(
|
|
45
|
+
name , raw , onChunk , zstdDecoder ,
|
|
46
|
+
) {
|
|
41
47
|
const codec = sniffCodec(raw);
|
|
42
48
|
if (codec === 'lz4' || name.endsWith('.lz4')) {
|
|
43
49
|
const lz4 = createLz4BlockDecoder(onChunk);
|
|
@@ -46,7 +52,7 @@ function decodeEntry(name , raw , onChunk
|
|
|
46
52
|
} else if (codec === 'gz' || name.endsWith('.gz')) {
|
|
47
53
|
new (Gunzip )(onChunk).push(raw, true);
|
|
48
54
|
} else if (codec === 'zstd' || name.endsWith('.zstd') || name.endsWith('.zst')) {
|
|
49
|
-
new (ZstdDecompress )(onChunk).push(raw, true);
|
|
55
|
+
(zstdDecoder ? zstdDecoder(onChunk) : new (ZstdDecompress )(onChunk)).push(raw, true);
|
|
50
56
|
} else if (codec === 'snappy' || name.endsWith('.snappy')) {
|
|
51
57
|
const snappy = createSnappyBlockDecoder(onChunk);
|
|
52
58
|
snappy.push(raw);
|
|
@@ -138,7 +144,9 @@ export async function runParseFromUrl(
|
|
|
138
144
|
decodeShsArchive(zipBytes, state, emit);
|
|
139
145
|
}
|
|
140
146
|
|
|
141
|
-
export function decodeShsArchive(
|
|
147
|
+
export function decodeShsArchive(
|
|
148
|
+
zipBytes , state , emit , { zstdDecoder } = {},
|
|
149
|
+
) {
|
|
142
150
|
let entries ;
|
|
143
151
|
try {
|
|
144
152
|
entries = unzipSync(zipBytes);
|
|
@@ -165,18 +173,21 @@ export function decodeShsArchive(zipBytes , state , emit
|
|
|
165
173
|
}
|
|
166
174
|
|
|
167
175
|
const decoder = buildChunkDecoder();
|
|
176
|
+
const joined = [];
|
|
168
177
|
let linesProcessed = 0;
|
|
169
178
|
for (const name of names) {
|
|
170
179
|
try {
|
|
171
180
|
decodeEntry(name, entries[name], (bytes) => {
|
|
172
|
-
|
|
173
|
-
|
|
181
|
+
joined.length = 0;
|
|
182
|
+
const lines = decoder.decode(bytes, joined);
|
|
183
|
+
for (let i = 0, j = 0; i < lines.length; i++) {
|
|
184
|
+
dispatchLine(lines[i], state, emit, joined[j]?.index === i ? joined[j++] : undefined);
|
|
174
185
|
linesProcessed++;
|
|
175
186
|
if (linesProcessed % 2000 === 0) {
|
|
176
187
|
emit({ type: 'progress', pct: null, linesProcessed });
|
|
177
188
|
}
|
|
178
189
|
}
|
|
179
|
-
});
|
|
190
|
+
}, zstdDecoder);
|
|
180
191
|
} catch {
|
|
181
192
|
emitShsError(emit, 'invalid-event-log');
|
|
182
193
|
return;
|
package/vendor-core/shs-load.js
CHANGED
|
@@ -2,6 +2,7 @@ import { validateShsRequest } from './shs-request.js';
|
|
|
2
2
|
import { fetchShsEventLog } from './proxy.js';
|
|
3
3
|
import { decodeShsArchive } from './parser-worker.js';
|
|
4
4
|
import { collectViaDispatch } from './cli/collect-run.js';
|
|
5
|
+
import { nodeArchiveCodecs } from './cli/native-zstd.js';
|
|
5
6
|
import { deriveEvidenceAvailability } from './evidence-availability.js';
|
|
6
7
|
import { mcpError } from './mcp-error.js';
|
|
7
8
|
|
|
@@ -18,7 +19,7 @@ export const DEFAULT_IDLE_TIMEOUT_MS = envInt('SPARKFORENSICS_SHS_TIMEOUT_MS', 3
|
|
|
18
19
|
|
|
19
20
|
function collectShsAppModel(zipBytes ) {
|
|
20
21
|
return collectViaDispatch(
|
|
21
|
-
(state, emit) => decodeShsArchive(zipBytes, state, emit),
|
|
22
|
+
(state, emit) => decodeShsArchive(zipBytes, state, emit, nodeArchiveCodecs),
|
|
22
23
|
(msg) => {
|
|
23
24
|
const m = msg ;
|
|
24
25
|
return mcpError(m?.code ?? 'invalid-event-log', m?.message ?? 'Failed to decode SHS archive.');
|
|
@@ -135,16 +135,26 @@ export function finalizeStage(
|
|
|
135
135
|
const { p50: spillMemP50, p95: spillMemP95, max: spillMemMax } = computeFieldQuantiles(arr, FIELDS.MEM_SPILLED);
|
|
136
136
|
const { p50: spillDiskP50, p95: spillDiskP95, max: spillDiskMax } = computeFieldQuantiles(arr, FIELDS.DISK_SPILLED);
|
|
137
137
|
|
|
138
|
-
// Straggler count: tasks with duration > 4 * P50
|
|
138
|
+
// Straggler count: tasks with duration > 4 * P50, their summed excess over P50, and the longest
|
|
139
|
+
// task that isn't one.
|
|
139
140
|
const stragglerThreshold = 4 * p50;
|
|
140
141
|
let stragglerCount = 0;
|
|
142
|
+
let stragglerExcessMs = 0;
|
|
143
|
+
let longestNonStragglerMs = 0;
|
|
141
144
|
const taskArrCount = arr.length / FIELDS.STRIDE;
|
|
142
145
|
if (p50 > 0) {
|
|
143
146
|
for (let i = 0; i < taskArrCount; i++) {
|
|
144
|
-
|
|
147
|
+
const duration = arr[i * FIELDS.STRIDE + FIELDS.DURATION];
|
|
148
|
+
if (duration > stragglerThreshold) { stragglerCount++; stragglerExcessMs += duration - p50; }
|
|
149
|
+
else if (duration > longestNonStragglerMs) longestNonStragglerMs = duration;
|
|
145
150
|
}
|
|
146
151
|
}
|
|
147
152
|
|
|
153
|
+
const peakConcurrentTasks = computePeakConcurrentTasks(arr);
|
|
154
|
+
const tailReplayRecoveryMs = stragglerCount > 0
|
|
155
|
+
? computeTailReplayRecoveryMs(arr, p50, peakConcurrentTasks)
|
|
156
|
+
: 0; // no task over 4x P50: both replays schedule the same durations
|
|
157
|
+
|
|
148
158
|
const hostStatsArr = [...hostStats.entries()].map(
|
|
149
159
|
([host, s]) => ({ host, taskCount: s.taskCount, totalDuration: s.totalDuration })
|
|
150
160
|
);
|
|
@@ -160,9 +170,12 @@ export function finalizeStage(
|
|
|
160
170
|
|
|
161
171
|
const data = {
|
|
162
172
|
...stage,
|
|
163
|
-
hostStats: hostStatsArr, executorStats: executorStatsArr, failureReasons: failureReasonsArr, localityStats: localityStatsArr, stragglerCount,
|
|
173
|
+
hostStats: hostStatsArr, executorStats: executorStatsArr, failureReasons: failureReasonsArr, localityStats: localityStatsArr, stragglerCount, stragglerExcessMs, longestNonStragglerMs,
|
|
174
|
+
tailReplayRecoveryMs,
|
|
164
175
|
failedTaskSamples,
|
|
165
176
|
peakExecutionMemoryMax,
|
|
177
|
+
taskActiveMs: computeTaskActiveMs(arr),
|
|
178
|
+
peakConcurrentTasks,
|
|
166
179
|
taskDurationP50: p50,
|
|
167
180
|
taskDurationP95: p95,
|
|
168
181
|
taskDurationMax: max,
|
|
@@ -178,6 +191,101 @@ export function finalizeStage(
|
|
|
178
191
|
return { type: 'stage', data };
|
|
179
192
|
}
|
|
180
193
|
|
|
194
|
+
// Wall-clock time during which at least one of the stage's tasks was running: the union of its
|
|
195
|
+
// [launch, finish) intervals. A stage's submittedAt..completedAt window also covers time it sat
|
|
196
|
+
// open with no task running (waiting for a free slot, or between retried tasks), which no
|
|
197
|
+
// task-level fix can compress. Tasks missing either timestamp are skipped.
|
|
198
|
+
export function computeTaskActiveMs(arr ) {
|
|
199
|
+
const taskCount = arr.length / FIELDS.STRIDE;
|
|
200
|
+
const intervals = [];
|
|
201
|
+
for (let i = 0; i < taskCount; i++) {
|
|
202
|
+
const launch = arr[i * FIELDS.STRIDE + FIELDS.LAUNCH_TIME];
|
|
203
|
+
const finish = arr[i * FIELDS.STRIDE + FIELDS.FINISH_TIME];
|
|
204
|
+
if (launch > 0 && finish > launch) intervals.push([launch, finish]);
|
|
205
|
+
}
|
|
206
|
+
intervals.sort((a, b) => a[0] - b[0]);
|
|
207
|
+
let activeMs = 0, start = -Infinity, end = -Infinity;
|
|
208
|
+
for (const [launch, finish] of intervals) {
|
|
209
|
+
if (launch > end) {
|
|
210
|
+
if (end > start) activeMs += end - start;
|
|
211
|
+
start = launch;
|
|
212
|
+
end = finish;
|
|
213
|
+
} else if (finish > end) {
|
|
214
|
+
end = finish;
|
|
215
|
+
}
|
|
216
|
+
}
|
|
217
|
+
if (end > start) activeMs += end - start;
|
|
218
|
+
return activeMs;
|
|
219
|
+
}
|
|
220
|
+
|
|
221
|
+
// Most tasks running at once, over the same [launch, finish) intervals as computeTaskActiveMs: the
|
|
222
|
+
// slots the stage actually got. A task finishing at the instant another launches frees its slot.
|
|
223
|
+
export function computePeakConcurrentTasks(arr ) {
|
|
224
|
+
const taskCount = arr.length / FIELDS.STRIDE;
|
|
225
|
+
const launches = [];
|
|
226
|
+
const finishes = [];
|
|
227
|
+
for (let i = 0; i < taskCount; i++) {
|
|
228
|
+
const launch = arr[i * FIELDS.STRIDE + FIELDS.LAUNCH_TIME];
|
|
229
|
+
const finish = arr[i * FIELDS.STRIDE + FIELDS.FINISH_TIME];
|
|
230
|
+
if (launch > 0 && finish > launch) { launches.push(launch); finishes.push(finish); }
|
|
231
|
+
}
|
|
232
|
+
launches.sort((a, b) => a - b);
|
|
233
|
+
finishes.sort((a, b) => a - b);
|
|
234
|
+
let peak = 0, running = 0, finished = 0;
|
|
235
|
+
for (const launch of launches) {
|
|
236
|
+
while (finished < finishes.length && finishes[finished] <= launch) { running--; finished++; }
|
|
237
|
+
running++;
|
|
238
|
+
if (running > peak) peak = running;
|
|
239
|
+
}
|
|
240
|
+
return peak;
|
|
241
|
+
}
|
|
242
|
+
|
|
243
|
+
// Wall-clock a tail fix recovers, replayed from the stage's own tasks: list scheduling (tasks in
|
|
244
|
+
// launch order, each on the slot that frees first) over `slots` slots, once with the real
|
|
245
|
+
// durations and once with every task over 4x P50 (the straggler definition above) capped at P50.
|
|
246
|
+
// The difference is the claim; dev/eval-tail-replay.mjs keeps an independent copy as its ground
|
|
247
|
+
// truth. Equal free times are interchangeable slots, so which one a min-heap picks can't change
|
|
248
|
+
// the result. Equal launch times keep the task array's order.
|
|
249
|
+
export function computeTailReplayRecoveryMs(arr , p50 , slots ) {
|
|
250
|
+
const taskCount = arr.length / FIELDS.STRIDE;
|
|
251
|
+
if (taskCount < 2 || !(p50 > 0)) return 0;
|
|
252
|
+
const order = new Uint32Array(taskCount);
|
|
253
|
+
for (let i = 0; i < taskCount; i++) order[i] = i;
|
|
254
|
+
order.sort((a, b) => arr[a * FIELDS.STRIDE + FIELDS.LAUNCH_TIME] - arr[b * FIELDS.STRIDE + FIELDS.LAUNCH_TIME] || a - b);
|
|
255
|
+
const free = new Float64Array(Math.max(1, Math.min(slots, taskCount)));
|
|
256
|
+
const capAboveMs = 4 * p50;
|
|
257
|
+
const actualEndMs = listScheduleEndMs(arr, order, free, Infinity, p50);
|
|
258
|
+
const fixedEndMs = listScheduleEndMs(arr, order, free, capAboveMs, p50);
|
|
259
|
+
return Math.max(0, actualEndMs - fixedEndMs);
|
|
260
|
+
}
|
|
261
|
+
|
|
262
|
+
// End of a list schedule over `free` (a min-heap of slot free times, reset here), each task's
|
|
263
|
+
// duration replaced by `cappedMs` when over `capAboveMs`.
|
|
264
|
+
function listScheduleEndMs(
|
|
265
|
+
arr , order , free , capAboveMs , cappedMs ,
|
|
266
|
+
) {
|
|
267
|
+
free.fill(0);
|
|
268
|
+
const n = free.length;
|
|
269
|
+
let endMs = 0;
|
|
270
|
+
for (let t = 0; t < order.length; t++) {
|
|
271
|
+
const duration = arr[order[t] * FIELDS.STRIDE + FIELDS.DURATION];
|
|
272
|
+
const finish = free[0] + (duration > capAboveMs ? cappedMs : duration);
|
|
273
|
+
if (finish > endMs) endMs = finish;
|
|
274
|
+
// Replace the root (earliest free slot) and sift it down.
|
|
275
|
+
let i = 0;
|
|
276
|
+
for (;;) {
|
|
277
|
+
const left = 2 * i + 1;
|
|
278
|
+
if (left >= n) break;
|
|
279
|
+
const child = left + 1 < n && free[left + 1] < free[left] ? left + 1 : left;
|
|
280
|
+
if (free[child] >= finish) break;
|
|
281
|
+
free[i] = free[child];
|
|
282
|
+
i = child;
|
|
283
|
+
}
|
|
284
|
+
free[i] = finish;
|
|
285
|
+
}
|
|
286
|
+
return endMs;
|
|
287
|
+
}
|
|
288
|
+
|
|
181
289
|
export function computeFieldQuantiles(arr , fieldIndex ) {
|
|
182
290
|
const taskCount = arr.length / FIELDS.STRIDE;
|
|
183
291
|
if (taskCount === 0) return { p50: 0, p95: 0, max: 0 };
|
|
@@ -0,0 +1,15 @@
|
|
|
1
|
+
// cyrb53: fast, non-cryptographic, deterministic 53-bit string hash. Not a
|
|
2
|
+
// security boundary; 53 bits keeps collision risk negligible at the per-node
|
|
3
|
+
// call volume the plan-identity digests (run-comparison.ts, detectors.ts) put
|
|
4
|
+
// through it.
|
|
5
|
+
export function cyrb53(str , seed = 0) {
|
|
6
|
+
let h1 = 0xdeadbeef ^ seed, h2 = 0x41c6ce57 ^ seed;
|
|
7
|
+
for (let i = 0; i < str.length; i++) {
|
|
8
|
+
const ch = str.charCodeAt(i);
|
|
9
|
+
h1 = Math.imul(h1 ^ ch, 2654435761);
|
|
10
|
+
h2 = Math.imul(h2 ^ ch, 1597334677);
|
|
11
|
+
}
|
|
12
|
+
h1 = Math.imul(h1 ^ (h1 >>> 16), 2246822507) ^ Math.imul(h2 ^ (h2 >>> 13), 3266489909);
|
|
13
|
+
h2 = Math.imul(h2 ^ (h2 >>> 16), 2246822507) ^ Math.imul(h1 ^ (h1 >>> 13), 3266489909);
|
|
14
|
+
return (4294967296 * (2097151 & h2) + (h1 >>> 0)).toString(16);
|
|
15
|
+
}
|