sparkforensics-mcp 0.2.4 → 0.4.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -0
- package/README.md +7 -1
- package/bin/sparkforensics-mcp.mjs +41 -10
- package/package.json +1 -1
- package/vendor-core/analyzer.js +156 -48
- package/vendor-core/check-coverage.js +88 -0
- package/vendor-core/cli/budgets.js +31 -18
- package/vendor-core/cli/collect-run.js +76 -31
- package/vendor-core/cli/native-zstd.js +2 -2
- package/vendor-core/cli/threshold-config.js +28 -0
- package/vendor-core/comparison-verdict.js +177 -0
- package/vendor-core/core-source-hash.txt +1 -0
- package/vendor-core/core-usage-locality.js +56 -2
- package/vendor-core/detector-docs.js +58 -0
- package/vendor-core/detectors.js +933 -459
- package/vendor-core/docs-config.js +0 -36
- package/vendor-core/docs-content/chapters/nav-index.json +31 -0
- package/vendor-core/docs-content/detection/cstor.md +9 -0
- package/vendor-core/docs-content/detection/fail.md +6 -2
- package/vendor-core/docs-site-config.js +3 -0
- package/vendor-core/event-handlers.js +191 -6
- package/vendor-core/event-schemas.js +29 -0
- package/vendor-core/evidence-report.js +440 -112
- package/vendor-core/export-data.js +79 -6
- package/vendor-core/finding-action-label.js +9 -88
- package/vendor-core/finding-filter-predicate.js +9 -0
- package/vendor-core/finding-generic-recommendation.js +6 -104
- package/vendor-core/finding-names.js +21 -45
- package/vendor-core/finding-presentation.js +333 -0
- package/vendor-core/finding-tag-help.js +110 -0
- package/vendor-core/finding-types.js +361 -0
- package/vendor-core/findings-of-type.js +11 -0
- package/vendor-core/format-utils.js +92 -27
- package/vendor-core/html-export.js +51 -0
- package/vendor-core/impact-band.js +21 -8
- package/vendor-core/impact-estimator.js +8 -521
- package/vendor-core/impact-format.js +114 -0
- package/vendor-core/impact-model.js +175 -0
- package/vendor-core/ingest.js +2 -0
- package/vendor-core/intervals.js +13 -0
- package/vendor-core/list-runs.js +2 -3
- package/vendor-core/load-vendored.js +70 -5
- package/vendor-core/mcp-server-factory.js +14 -10
- package/vendor-core/mcp-tools.js +105 -45
- package/vendor-core/model-assembler.js +12 -0
- package/vendor-core/occupancy.js +1 -1
- package/vendor-core/parser-worker.js +22 -5
- package/vendor-core/plan-graph-model.js +3 -2
- package/vendor-core/plan-node-detail.js +1 -1
- package/vendor-core/recommendation-rollup.js +63 -3
- package/vendor-core/redact.js +68 -28
- package/vendor-core/run-comparison.js +40 -7
- package/vendor-core/run-interpretation.js +290 -0
- package/vendor-core/run-outcome.js +74 -0
- package/vendor-core/run-payload.js +17 -0
- package/vendor-core/run-shape.js +40 -0
- package/vendor-core/run-verdict.js +353 -0
- package/vendor-core/scaling-sim.js +4 -5
- package/vendor-core/scorecard-estimates.js +62 -0
- package/vendor-core/shs-fetch.js +175 -65
- package/vendor-core/shs-load.js +1 -1
- package/vendor-core/sql-stages.js +11 -0
- package/vendor-core/stage-quantiles.js +14 -0
- package/vendor-core/task-failure.js +151 -0
- package/vendor-core/threshold-overrides.js +160 -0
- package/vendor-core/threshold-summary.js +11 -33
- package/vendor-core/types.js +6 -42
- package/vendor-core/vendor/fflate.js +1 -1
- package/vendor-core/wall-clock.js +1 -12
- package/vendor-core/wasted-core-hours.js +2 -2
- package/vendor-core/zip-archive.js +167 -0
package/vendor-core/mcp-tools.js
CHANGED
|
@@ -6,16 +6,21 @@ import { collectRun } from './cli/collect-run.js';
|
|
|
6
6
|
import { deriveEvidenceAvailability } from './evidence-availability.js';
|
|
7
7
|
import { resolveFromShs, DEFAULT_MAX_ARCHIVE_BYTES, DEFAULT_IDLE_TIMEOUT_MS } from './shs-load.js';
|
|
8
8
|
import { mcpError } from './mcp-error.js';
|
|
9
|
-
import { buildEvidenceReport, toFindingsFilter,
|
|
10
|
-
import {
|
|
9
|
+
import { buildEvidenceReport, toFindingsFilter, } from './evidence-report.js';
|
|
10
|
+
import { redactComparison } from './redact.js';
|
|
11
11
|
import { computeWallClock } from './wall-clock.js';
|
|
12
12
|
import { analyze } from './analyzer.js';
|
|
13
13
|
import { buildComparison, renderComparisonMarkdown, } from './run-comparison.js';
|
|
14
|
+
import { comparisonVerdict, } from './comparison-verdict.js';
|
|
14
15
|
import { evaluateBudgets, } from './cli/budgets.js';
|
|
15
16
|
import { FINDING_NAMES, titleCase } from './finding-names.js';
|
|
16
|
-
import { docAnchorForType
|
|
17
|
+
import { docAnchorForType } from './detector-docs.js';
|
|
18
|
+
import { DETECTORS, } from './detectors.js';
|
|
19
|
+
import { tunedDetectors } from './threshold-overrides.js';
|
|
20
|
+
import { tuningDocSlugForAnchor, pageForAnchor } from './docs-config.js';
|
|
17
21
|
import { typeTag } from './format-utils.js';
|
|
18
|
-
|
|
22
|
+
|
|
23
|
+
|
|
19
24
|
|
|
20
25
|
|
|
21
26
|
// RunRef uses a nested `source` key, matching every real call site (resolveOrCreateRun's
|
|
@@ -33,13 +38,22 @@ import { typeTag } from './format-utils.js';
|
|
|
33
38
|
|
|
34
39
|
|
|
35
40
|
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
|
|
36
48
|
|
|
37
49
|
// compareRuns returns a smaller MCP-facing projection of CompareRunsResult
|
|
38
|
-
// (runIdA/runIdB/findingsDelta/metricDeltas/confidence/reason/matchedCoverage), not the full
|
|
39
|
-
// shape (no baselineLabel/stageSkew/baseStages/candStages).
|
|
50
|
+
// (runIdA/runIdB/verdict/findingsDelta/metricDeltas/confidence/reason/matchedCoverage), not the full
|
|
51
|
+
// raw shape (no baselineLabel/stageSkew/baseStages/candStages/jobOutcomes).
|
|
40
52
|
|
|
41
53
|
|
|
42
54
|
|
|
55
|
+
|
|
56
|
+
|
|
43
57
|
|
|
44
58
|
|
|
45
59
|
|
|
@@ -169,20 +183,29 @@ export async function resolveOrCreateRun(
|
|
|
169
183
|
return resolution;
|
|
170
184
|
}
|
|
171
185
|
|
|
186
|
+
// `thresholds` on every analyzing tool below: the server's --thresholds overrides, fixed for the
|
|
187
|
+
// process by createMcpServer(). A client can't set them per call; the dashboard never has them.
|
|
172
188
|
export function diagnoseRun(runId , opts
|
|
173
189
|
|
|
174
|
-
|
|
190
|
+
|
|
175
191
|
)
|
|
176
|
-
|
|
177
|
-
|
|
192
|
+
|
|
193
|
+
|
|
194
|
+
|
|
178
195
|
{
|
|
179
196
|
const appModel = getCachedAppModel(runId);
|
|
180
197
|
const findingsFilter = toFindingsFilter(opts?.impactBand, opts?.type, opts?.stageId);
|
|
181
|
-
const { json, markdown } = buildEvidenceReport(appModel, {
|
|
198
|
+
const { json, markdown } = buildEvidenceReport(appModel, {
|
|
199
|
+
redact: opts?.redact, markdown: opts?.markdown, findingsFilter, thresholds: opts?.thresholds,
|
|
200
|
+
});
|
|
182
201
|
const include = opts?.include ?? [];
|
|
202
|
+
const tuned = json.summary.tunedThresholds;
|
|
183
203
|
return {
|
|
184
|
-
runId, findings: json.findings, recommendations: json.recommendations, cleanChecks: json.cleanChecks,
|
|
204
|
+
runId, verdict: json.verdict, findings: json.findings, recommendations: json.recommendations, cleanChecks: json.cleanChecks,
|
|
205
|
+
notRunChecks: json.notRunChecks,
|
|
185
206
|
runComplete: appModel.app?.endTime != null,
|
|
207
|
+
// Top level too, so a client that never asks for `summary` still sees the run was tuned.
|
|
208
|
+
...(tuned ? { tunedThresholds: tuned } : {}),
|
|
186
209
|
...(include.includes('summary') ? { summary: json.summary } : {}),
|
|
187
210
|
...(include.includes('evidenceAvailability') ? { evidenceAvailability: json.evidenceAvailability } : {}),
|
|
188
211
|
...(include.includes('detectors') ? { detectors: json.detectors } : {}),
|
|
@@ -211,17 +234,26 @@ function extractDocTitle(markdown ) {
|
|
|
211
234
|
|
|
212
235
|
|
|
213
236
|
|
|
237
|
+
// A finding type documents itself; a detector-level type that never appears on a finding
|
|
238
|
+
// (broadcastSizing) documents the types its entry emits.
|
|
239
|
+
function documentedTypes(type ) {
|
|
240
|
+
if (FINDING_NAMES[type] !== undefined) return [type];
|
|
241
|
+
return DETECTORS.find((entry) => entry.type === type)?.emits ?? [];
|
|
242
|
+
}
|
|
243
|
+
|
|
214
244
|
/** Detection + tuning reference documentation for one finding `type`, independent of any run
|
|
215
|
-
* (documentation is a property of the type: a client fetches it once per type and caches it).
|
|
245
|
+
* (documentation is a property of the type: a client fetches it once per type and caches it).
|
|
246
|
+
* A detector-level `type` resolves to the documentation of the finding types its entry emits. */
|
|
216
247
|
export function getFindingDocumentation(type ) {
|
|
217
|
-
const
|
|
218
|
-
|
|
248
|
+
const types = documentedTypes(type);
|
|
249
|
+
const [docType] = types;
|
|
250
|
+
if (docType === undefined) throw mcpError('invalid-type', `Unknown finding type: ${type}`);
|
|
219
251
|
|
|
220
|
-
const tag = typeTag(
|
|
252
|
+
const tag = typeTag(docType);
|
|
221
253
|
const detectionContent = readFileSync(join(DOCS_CONTENT_DIR, 'detection', `${tag.toLowerCase()}.md`), 'utf8');
|
|
222
254
|
const detectionDoc = { tag, title: extractDocTitle(detectionContent), content: detectionContent };
|
|
223
255
|
|
|
224
|
-
const anchor = docAnchorForType(
|
|
256
|
+
const anchor = docAnchorForType(docType);
|
|
225
257
|
const slug = anchor ? tuningDocSlugForAnchor(anchor) : null;
|
|
226
258
|
const tuningPath = slug ? join(DOCS_CONTENT_DIR, 'tuning', `${slug}.md`) : null;
|
|
227
259
|
let tuningDoc = null;
|
|
@@ -235,7 +267,8 @@ export function getFindingDocumentation(type ) {
|
|
|
235
267
|
if (entry) tuningDoc = { anchor, title: entry.title, content: readNavEntryContent(entry) };
|
|
236
268
|
}
|
|
237
269
|
|
|
238
|
-
|
|
270
|
+
const name = types.map((t) => titleCase(FINDING_NAMES[t] ?? t)).join(' / ');
|
|
271
|
+
return { type, name, detectionDoc, tuningDoc };
|
|
239
272
|
}
|
|
240
273
|
|
|
241
274
|
const CHAPTERS_NAV_FILE = join(DOCS_CONTENT_DIR, 'chapters', 'nav-index.json');
|
|
@@ -266,10 +299,10 @@ export function getReferenceDoc(anchor ) {
|
|
|
266
299
|
}
|
|
267
300
|
|
|
268
301
|
export function getFindingEvidence(
|
|
269
|
-
runId , findingId , opts
|
|
302
|
+
runId , findingId , opts ,
|
|
270
303
|
) {
|
|
271
304
|
const appModel = getCachedAppModel(runId);
|
|
272
|
-
const { json } = buildEvidenceReport(appModel, { redact: opts?.redact, markdown: false });
|
|
305
|
+
const { json } = buildEvidenceReport(appModel, { redact: opts?.redact, markdown: false, thresholds: opts?.thresholds });
|
|
273
306
|
const finding = json.findings.find((f) => f.id === findingId);
|
|
274
307
|
if (!finding) throw mcpError('finding-not-found', `No finding ${findingId} on run ${runId}.`);
|
|
275
308
|
return { runId, finding };
|
|
@@ -279,15 +312,19 @@ function hasCompleteInterval(app ) {
|
|
|
279
312
|
return Number.isFinite(app?.startTime) && Number.isFinite(app?.endTime) && (app?.endTime ?? 0) > (app?.startTime ?? 0);
|
|
280
313
|
}
|
|
281
314
|
|
|
282
|
-
export function getRunSummary(runId , opts
|
|
315
|
+
export function getRunSummary(runId , opts ) {
|
|
283
316
|
const appModel = getCachedAppModel(runId);
|
|
284
317
|
const { app, stages, jobs, sql, executors } = appModel;
|
|
285
318
|
const durationMs = hasCompleteInterval(app) ? computeWallClock(app, stages).total : null;
|
|
286
|
-
//
|
|
287
|
-
//
|
|
288
|
-
//
|
|
319
|
+
// The failure reason needs the stageFailed findings, so this reads the (cached) evidence report.
|
|
320
|
+
// With redact, the app identity comes from that same redacted report, so a host token in the app
|
|
321
|
+
// name and in Spark's failure reason get the same pseudonym.
|
|
322
|
+
const { summary } = buildEvidenceReport(appModel, { redact: opts?.redact, markdown: false, thresholds: opts?.thresholds }).json;
|
|
323
|
+
const { outcome } = summary;
|
|
289
324
|
const rawApp = { id: app?.id ?? null, name: app?.name ?? null, sparkVersion: app?.sparkVersion ?? null };
|
|
290
|
-
const redactedApp = opts?.redact
|
|
325
|
+
const redactedApp = opts?.redact
|
|
326
|
+
? { id: summary.app.id ?? null, name: summary.app.name ?? null, sparkVersion: summary.app.sparkVersion ?? null }
|
|
327
|
+
: rawApp;
|
|
291
328
|
return {
|
|
292
329
|
runId,
|
|
293
330
|
app: redactedApp,
|
|
@@ -297,27 +334,31 @@ export function getRunSummary(runId , opts )
|
|
|
297
334
|
executorCount: { added: executors.added.length, removed: executors.removed.length },
|
|
298
335
|
durationMs,
|
|
299
336
|
runComplete: app?.endTime != null,
|
|
337
|
+
...outcome,
|
|
338
|
+
runShape: summary.runShape,
|
|
300
339
|
};
|
|
301
340
|
}
|
|
302
341
|
|
|
303
342
|
// Shared by compareRuns and evaluateBudgetsForRun: both need a resolved run's finding catalog
|
|
304
343
|
// (same analyze() call shape) first.
|
|
305
|
-
async function resolveAndAnalyze(
|
|
344
|
+
async function resolveAndAnalyze(
|
|
345
|
+
ref , thresholds ,
|
|
346
|
+
) {
|
|
306
347
|
const { runId, appModel } = await resolveOrCreateRun(ref);
|
|
307
348
|
const catalog = analyze(
|
|
308
349
|
appModel.app, appModel.stages, appModel.executors.added, appModel.executors.removed,
|
|
309
|
-
appModel.jobs, appModel.sql, appModel.runAggregates,
|
|
350
|
+
appModel.jobs, appModel.sql, appModel.runAggregates, { thresholds },
|
|
310
351
|
);
|
|
311
352
|
return { runId, appModel, catalog };
|
|
312
353
|
}
|
|
313
354
|
|
|
314
355
|
export async function compareRuns(
|
|
315
|
-
a , b , opts
|
|
316
|
-
)
|
|
356
|
+
a , b , opts ,
|
|
357
|
+
) {
|
|
317
358
|
const [
|
|
318
359
|
{ runId: runIdA, appModel: appModelA, catalog: catalogA },
|
|
319
360
|
{ runId: runIdB, appModel: appModelB, catalog: catalogB },
|
|
320
|
-
] = await Promise.all([resolveAndAnalyze(a), resolveAndAnalyze(b)]);
|
|
361
|
+
] = await Promise.all([resolveAndAnalyze(a, opts?.thresholds), resolveAndAnalyze(b, opts?.thresholds)]);
|
|
321
362
|
|
|
322
363
|
// buildComparison's captureSnapshot uses an empty taskDataCache: that arg only feeds the
|
|
323
364
|
// interactive stage-detail drill-down, which none of compare/matchStages/metricDeltas/findingsDelta
|
|
@@ -330,44 +371,63 @@ export async function compareRuns(
|
|
|
330
371
|
// Stage names throughout `built` carry raw Spark stage text, which can embed a host/IP token as
|
|
331
372
|
// free text, the same residual redactReport() already scrubs from the evidence report.
|
|
332
373
|
const result = opts?.redact ? redactComparison(built) : built;
|
|
374
|
+
const verdict = comparisonVerdict(result);
|
|
333
375
|
|
|
334
376
|
return {
|
|
335
377
|
runIdA,
|
|
336
378
|
runIdB,
|
|
379
|
+
verdict,
|
|
337
380
|
findingsDelta: result.findings,
|
|
338
381
|
metricDeltas: result.metrics,
|
|
339
382
|
confidence: result.confidence,
|
|
340
383
|
reason: result.reason,
|
|
341
384
|
matchedCoverage: result.matchedCoverage,
|
|
342
|
-
...(opts?.
|
|
385
|
+
...tunedField(opts?.thresholds),
|
|
386
|
+
...(opts?.markdown ? { markdown: renderComparisonMarkdown(result, verdict, tunedDetectors(opts?.thresholds)) } : {}),
|
|
343
387
|
};
|
|
344
388
|
}
|
|
345
389
|
|
|
390
|
+
// Both runs of a comparison use the same overrides, so a tuned delta compares like with like.
|
|
391
|
+
function tunedField(thresholds ) {
|
|
392
|
+
const tuned = tunedDetectors(thresholds);
|
|
393
|
+
return tuned ? { tunedThresholds: tuned } : {};
|
|
394
|
+
}
|
|
395
|
+
|
|
346
396
|
export async function evaluateBudgetsForRun(
|
|
347
397
|
primary ,
|
|
348
398
|
budgets ,
|
|
349
399
|
secondary ,
|
|
350
|
-
|
|
400
|
+
opts ,
|
|
401
|
+
)
|
|
402
|
+
|
|
403
|
+
|
|
404
|
+
{
|
|
351
405
|
// Mirrors the CLI's --regression-metric/--max-regression-pct pairing guard: unlike the CLI, this
|
|
352
406
|
// tool never defaults regressionMetric, so seeing it set here means the caller asked for a
|
|
353
407
|
// regression check and forgot the threshold, evaluateBudgets() would otherwise skip it silently.
|
|
354
408
|
if (budgets.regressionMetric !== undefined && budgets.maxRegressionPct === undefined) {
|
|
355
409
|
throw mcpError('access-or-upstream-failure', 'regressionMetric requires maxRegressionPct.');
|
|
356
410
|
}
|
|
357
|
-
const [
|
|
358
|
-
resolveAndAnalyze(primary),
|
|
359
|
-
secondary ? resolveAndAnalyze(secondary) : Promise.resolve(undefined),
|
|
411
|
+
const [first, second] = await Promise.all([
|
|
412
|
+
resolveAndAnalyze(primary, opts?.thresholds),
|
|
413
|
+
secondary ? resolveAndAnalyze(secondary, opts?.thresholds) : Promise.resolve(undefined),
|
|
360
414
|
]);
|
|
361
415
|
|
|
362
|
-
|
|
363
|
-
|
|
364
|
-
|
|
365
|
-
|
|
366
|
-
|
|
367
|
-
|
|
368
|
-
|
|
369
|
-
|
|
370
|
-
|
|
371
|
-
|
|
372
|
-
|
|
416
|
+
// Same roles as the CLI's positional run + --baseline: with a second run, `primary` is the
|
|
417
|
+
// baseline and `secondary` the candidate, and absolute budgets plus the run-complete check
|
|
418
|
+
// apply to the candidate.
|
|
419
|
+
const candidate = second ?? first;
|
|
420
|
+
const baseline = second ? first : undefined;
|
|
421
|
+
const comparison = baseline
|
|
422
|
+
? buildComparison(
|
|
423
|
+
{ label: baseline.runId, appModel: baseline.appModel, catalog: baseline.catalog },
|
|
424
|
+
{ label: candidate.runId, appModel: candidate.appModel, catalog: candidate.catalog },
|
|
425
|
+
)
|
|
426
|
+
: undefined;
|
|
427
|
+
|
|
428
|
+
const { results, violated, inconclusive } = evaluateBudgets({
|
|
429
|
+
appModel: candidate.appModel, catalog: candidate.catalog, budgets, comparison, thresholds: opts?.thresholds,
|
|
430
|
+
});
|
|
431
|
+
|
|
432
|
+
return { runId: first.runId, results, violated, inconclusive, ...tunedField(opts?.thresholds) };
|
|
373
433
|
}
|
|
@@ -63,6 +63,18 @@ export function createModelCallbacks(
|
|
|
63
63
|
if (stage) stage.executorMetrics = execMetrics;
|
|
64
64
|
}
|
|
65
65
|
},
|
|
66
|
+
// Patch speculation totals that grew after StageCompleted: Spark kills a losing speculative
|
|
67
|
+
// copy only once its stage finishes. `data` is Map<stageId, { speculationWasteMs, speculationWastedAttempts }>.
|
|
68
|
+
onStageSpeculationWaste(data ) {
|
|
69
|
+
const totalsByStage = data ;
|
|
70
|
+
for (const [stageId, totals] of totalsByStage) {
|
|
71
|
+
const stage = appModel.stages.get(stageId);
|
|
72
|
+
if (stage) {
|
|
73
|
+
stage.speculationWasteMs = totals.speculationWasteMs;
|
|
74
|
+
stage.speculationWastedAttempts = totals.speculationWastedAttempts;
|
|
75
|
+
}
|
|
76
|
+
}
|
|
77
|
+
},
|
|
66
78
|
onDone, onError,
|
|
67
79
|
};
|
|
68
80
|
}
|
package/vendor-core/occupancy.js
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
// Occupancy-weighted wall-clock attribution for per-finding impact: every
|
|
2
2
|
// stage's wall-clock claim is apportioned purely from its own observed window
|
|
3
3
|
// and how much it overlapped with other stages. No graph, no parentIds traversal.
|
|
4
|
-
import { mergeIntervals } from './
|
|
4
|
+
import { mergeIntervals } from './intervals.js';
|
|
5
5
|
|
|
6
6
|
|
|
7
7
|
|
|
@@ -4,14 +4,15 @@ import { Decompress as ZstdDecompress } from './vendor/fzstd.js';
|
|
|
4
4
|
import { createSnappyBlockDecoder } from './snappy-block.js';
|
|
5
5
|
import { createState, dispatchLine, buildChunkDecoder, emitParseCompletion, } from './event-handlers.js';
|
|
6
6
|
import { TASK_FIELD_NAMES } from './stage-quantiles.js';
|
|
7
|
-
import { runParseFromUrl, sniffCodec } from './shs-fetch.js';
|
|
7
|
+
import { runParseFromUrl, sniffCodec, parseZipArchive } from './shs-fetch.js';
|
|
8
|
+
import { isZip } from './zip-archive.js';
|
|
8
9
|
import { createWorkerZstdDecoders } from './zstd-worker-client.js';
|
|
9
10
|
|
|
10
11
|
export {
|
|
11
12
|
buildChunkDecoder, createState, normalizeSparkProperties, parseSparkMemoryMB, extractResources,
|
|
12
13
|
accumulateTask, resolvePlanTree, startApplication, updateEnvironment, startJob, endJob, submitStage,
|
|
13
14
|
mergeStageRddInfo, recordStageExecutorMetrics, startSqlExecution, endSqlExecution,
|
|
14
|
-
applyDriverAccumUpdates, addExecutor, removeExecutor, processEvent, dispatchLine,
|
|
15
|
+
applyDriverAccumUpdates, addExecutor, removeExecutor, recordBlockUpdate, processEvent, dispatchLine,
|
|
15
16
|
collectStageExecutorMetrics,
|
|
16
17
|
} from './event-handlers.js';
|
|
17
18
|
|
|
@@ -35,6 +36,8 @@ const MIN_PROGRESS_STEPS = 100;
|
|
|
35
36
|
const PROGRESS_EMIT_LINES = 300;
|
|
36
37
|
|
|
37
38
|
|
|
39
|
+
const NOT_AN_EVENT_LOG = 'Not a Spark event log: no application-start event found. Choose a Spark event log file, or check the docs for supported formats.';
|
|
40
|
+
|
|
38
41
|
// zstdDecoder replaces the vendored fzstd for zstd input; the Node CLI/MCP path passes
|
|
39
42
|
// cli/native-zstd.ts's native-zlib decoder, which a browser bundle can't import.
|
|
40
43
|
|
|
@@ -140,6 +143,20 @@ export async function runParse(
|
|
|
140
143
|
return;
|
|
141
144
|
}
|
|
142
145
|
|
|
146
|
+
// A Spark History Server download (the UI's download link, or GET
|
|
147
|
+
// /api/v1/applications/<id>/logs) is a zip holding the log file, or a
|
|
148
|
+
// rolling log's parts: unwrap it through the same path the SHS fetch uses.
|
|
149
|
+
if (isZip(new Uint8Array(await file.slice(0, Math.min(4, file.size)).arrayBuffer()))) {
|
|
150
|
+
await parseZipArchive(file, state, emit, {
|
|
151
|
+
zstdDecoder,
|
|
152
|
+
chunkSize,
|
|
153
|
+
progressEvery: PROGRESS_EMIT_LINES,
|
|
154
|
+
reportPct: true,
|
|
155
|
+
onInvalid: (detail) => emit({ type: 'error', message: detail ?? NOT_AN_EVENT_LOG }),
|
|
156
|
+
});
|
|
157
|
+
return;
|
|
158
|
+
}
|
|
159
|
+
|
|
143
160
|
const decoder = buildChunkDecoder();
|
|
144
161
|
const joined = [];
|
|
145
162
|
let linesProcessed = 0;
|
|
@@ -168,7 +185,7 @@ export async function runParse(
|
|
|
168
185
|
}
|
|
169
186
|
|
|
170
187
|
if (!state.app) {
|
|
171
|
-
emit({ type: 'error', message:
|
|
188
|
+
emit({ type: 'error', message: NOT_AN_EVENT_LOG });
|
|
172
189
|
return;
|
|
173
190
|
}
|
|
174
191
|
|
|
@@ -227,7 +244,7 @@ export async function runParseFiles(
|
|
|
227
244
|
}
|
|
228
245
|
|
|
229
246
|
if (!state.app) {
|
|
230
|
-
emit({ type: 'error', message:
|
|
247
|
+
emit({ type: 'error', message: NOT_AN_EVENT_LOG });
|
|
231
248
|
return;
|
|
232
249
|
}
|
|
233
250
|
|
|
@@ -242,7 +259,7 @@ if (isWorker) {
|
|
|
242
259
|
let workerState = null;
|
|
243
260
|
// Dropped zstd files decompress in a second worker, overlapping with parsing here. The
|
|
244
261
|
// `new Worker(new URL(...))` stays inline for Vite's worker detection (see ingest.ts). The SHS
|
|
245
|
-
// path (runParseFromUrl)
|
|
262
|
+
// path (runParseFromUrl) keeps in-thread fzstd.
|
|
246
263
|
const zstdDecoder = createWorkerZstdDecoders(
|
|
247
264
|
() => new Worker(new URL('./zstd-worker.js', import.meta.url), { type: 'module' }),
|
|
248
265
|
(onChunk) => new (ZstdDecompress )(onChunk),
|
|
@@ -15,7 +15,7 @@ import {
|
|
|
15
15
|
formatPlanMetricValue,
|
|
16
16
|
} from './plan-node-detail.js';
|
|
17
17
|
import { computeSegments, mapSegmentsToStagesForDisplay } from './plan-duration-attribution.js';
|
|
18
|
-
import { stageIdsForSqlExec } from './
|
|
18
|
+
import { stageIdsForSqlExec } from './sql-stages.js';
|
|
19
19
|
|
|
20
20
|
|
|
21
21
|
|
|
@@ -121,7 +121,8 @@ export function buildPlanGraphModel(
|
|
|
121
121
|
const findingsByNodeId = new Map ();
|
|
122
122
|
if (sqlExecutionId != null) {
|
|
123
123
|
for (const finding of findings) {
|
|
124
|
-
if (finding.executionId !== sqlExecutionId) continue;
|
|
124
|
+
if (!('executionId' in finding) || finding.executionId !== sqlExecutionId) continue;
|
|
125
|
+
// `?? []`: a hand-built or foreign finding may still lack the field the type promises.
|
|
125
126
|
for (const nodeId of finding.planNodeIds ?? []) {
|
|
126
127
|
const list = findingsByNodeId.get(nodeId);
|
|
127
128
|
if (list) list.push(finding);
|
|
@@ -5,7 +5,7 @@
|
|
|
5
5
|
|
|
6
6
|
import { pathBasename, formatBytes, formatDuration } from './format-utils.js';
|
|
7
7
|
import { attributeStageDurationToPlan, attributeStageDurationToPlanInclusive } from './plan-duration-attribution.js';
|
|
8
|
-
import { stageIdsForSqlExec } from './
|
|
8
|
+
import { stageIdsForSqlExec } from './sql-stages.js';
|
|
9
9
|
|
|
10
10
|
|
|
11
11
|
const BOILERPLATE_PREFIXES = ['serializefromobject', 'deserializetoobject', 'mapelements',
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
// Explicit .ts extensions: plain Node's ESM resolver (the runtime CLI/MCP path
|
|
2
2
|
// runs under) requires the exact specifier, unlike a bundler.
|
|
3
|
-
import { mergeIntervals } from './
|
|
4
|
-
import { worstImpactBand, IMPACT_BAND_ORDER } from './format-utils.js';
|
|
3
|
+
import { mergeIntervals } from './intervals.js';
|
|
4
|
+
import { formatWallClockRange, worstImpactBand, IMPACT_BAND_ORDER } from './format-utils.js';
|
|
5
5
|
|
|
6
6
|
|
|
7
7
|
|
|
@@ -48,7 +48,7 @@ function groupByType(findings ) {
|
|
|
48
48
|
}
|
|
49
49
|
|
|
50
50
|
function stageIdsOf(finding ) {
|
|
51
|
-
if (finding
|
|
51
|
+
if ('stageIds' in finding) return finding.stageIds;
|
|
52
52
|
if (finding.stageId != null) return [finding.stageId];
|
|
53
53
|
return [];
|
|
54
54
|
}
|
|
@@ -124,6 +124,31 @@ export function buildRecommendationRollup(
|
|
|
124
124
|
});
|
|
125
125
|
}
|
|
126
126
|
|
|
127
|
+
/** A group's trailing figure on the Findings board, and its tooltip. `time`
|
|
128
|
+
* and `resource` both lead with the "×N" finding count (the "worth
|
|
129
|
+
* expanding" signal); `count` skips it since the impact-band tally already
|
|
130
|
+
* implies N. The row stays terse ("×2 · 476ms recoverable") to fit a dense
|
|
131
|
+
* right-aligned column; the title spells the shorthand out. */
|
|
132
|
+
export function rollupGroupStat(group ) {
|
|
133
|
+
if (group.kind === 'time') {
|
|
134
|
+
const recoverable = formatWallClockRange(group.recoverableMsHigh, group.recoverableMsHigh);
|
|
135
|
+
return {
|
|
136
|
+
stat: `×${group.findingCount} · ${recoverable} recoverable`,
|
|
137
|
+
statTitle: `${group.findingCount} findings of this type; up to ${recoverable} of run time could be recovered by fixing them`,
|
|
138
|
+
};
|
|
139
|
+
}
|
|
140
|
+
if (group.kind === 'resource') {
|
|
141
|
+
return {
|
|
142
|
+
stat: `×${group.findingCount} · resource-cost projection`,
|
|
143
|
+
statTitle: `${group.findingCount} findings of this type; a resource-cost estimate (not run time) is projected for fixing them`,
|
|
144
|
+
};
|
|
145
|
+
}
|
|
146
|
+
return {
|
|
147
|
+
stat: Object.entries(group.byImpactBand).map(([impactBand, count]) => `${count} ${impactBand}`).join(', '),
|
|
148
|
+
statTitle: `${group.findingCount} findings of this type, by impact`,
|
|
149
|
+
};
|
|
150
|
+
}
|
|
151
|
+
|
|
127
152
|
/** True when a finding is real evidence of an issue, as opposed to a mere
|
|
128
153
|
* evidence-unavailable caveat. Shared by `isEligible` below and by anything
|
|
129
154
|
* else that decides whether a REGISTRY type has "something to show" (an
|
|
@@ -150,6 +175,10 @@ export function isEligible(finding ) {
|
|
|
150
175
|
// `isRealFinding`, this exclusion doesn't apply beyond the rollup: an
|
|
151
176
|
// incompleteRun finding still backs its own ordinary active widget card.
|
|
152
177
|
if (finding.type === 'incompleteRun') return false;
|
|
178
|
+
// A missing-evidence caveat (memoryUtilization's is already excluded by isRealFinding;
|
|
179
|
+
// cacheUtilization's storageUnobserved keeps its widget card, so a run with no block updates
|
|
180
|
+
// never lists Cache Storage as a passed check) is never a fix to rank.
|
|
181
|
+
if ('dataUnavailable' in finding && finding.dataUnavailable) return false;
|
|
153
182
|
return isRealFinding(finding);
|
|
154
183
|
}
|
|
155
184
|
|
|
@@ -191,3 +220,34 @@ export function rankFindings(findings ) {
|
|
|
191
220
|
return (IMPACT_BAND_ORDER[a.impactBand] ?? 9) - (IMPACT_BAND_ORDER[b.impactBand] ?? 9);
|
|
192
221
|
});
|
|
193
222
|
}
|
|
223
|
+
|
|
224
|
+
/** A board group ready to render: members ranked representative first, the representative's
|
|
225
|
+
* band (the heading the group sits under) and the group's trailing figure. */
|
|
226
|
+
|
|
227
|
+
|
|
228
|
+
|
|
229
|
+
|
|
230
|
+
|
|
231
|
+
|
|
232
|
+
|
|
233
|
+
|
|
234
|
+
|
|
235
|
+
|
|
236
|
+
/** The Findings board's groups over `eligible`, in fix-first order. The run's interpretation
|
|
237
|
+
* carries this for every eligible finding; the board recomputes it only over a filtered subset. */
|
|
238
|
+
export function rankedRollup(
|
|
239
|
+
eligible ,
|
|
240
|
+
stages ,
|
|
241
|
+
) {
|
|
242
|
+
return buildRecommendationRollup(eligible, stages).map((group) => {
|
|
243
|
+
const members = rankFindings(group.findings);
|
|
244
|
+
return {
|
|
245
|
+
kind: group.kind,
|
|
246
|
+
type: group.type,
|
|
247
|
+
unit: group.kind === 'resource' ? group.unit : null,
|
|
248
|
+
band: members[0].impactBand,
|
|
249
|
+
members,
|
|
250
|
+
...rollupGroupStat(group),
|
|
251
|
+
};
|
|
252
|
+
});
|
|
253
|
+
}
|