sparkforensics-mcp 0.3.0 → 0.4.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/bin/sparkforensics-mcp.mjs +41 -10
- package/package.json +1 -1
- package/vendor-core/analyzer.js +156 -48
- package/vendor-core/check-coverage.js +88 -0
- package/vendor-core/cli/budgets.js +31 -18
- package/vendor-core/cli/collect-run.js +76 -31
- package/vendor-core/cli/threshold-config.js +28 -0
- package/vendor-core/comparison-verdict.js +177 -0
- package/vendor-core/core-source-hash.txt +1 -0
- package/vendor-core/core-usage-locality.js +56 -2
- package/vendor-core/detector-docs.js +58 -0
- package/vendor-core/detectors.js +918 -458
- package/vendor-core/docs-config.js +0 -36
- package/vendor-core/docs-content/chapters/nav-index.json +31 -0
- package/vendor-core/docs-content/detection/cstor.md +9 -0
- package/vendor-core/docs-site-config.js +3 -0
- package/vendor-core/event-handlers.js +170 -6
- package/vendor-core/event-schemas.js +21 -0
- package/vendor-core/evidence-report.js +421 -112
- package/vendor-core/export-data.js +79 -6
- package/vendor-core/finding-action-label.js +9 -88
- package/vendor-core/finding-filter-predicate.js +9 -0
- package/vendor-core/finding-generic-recommendation.js +6 -104
- package/vendor-core/finding-names.js +21 -45
- package/vendor-core/finding-presentation.js +333 -0
- package/vendor-core/finding-tag-help.js +110 -0
- package/vendor-core/finding-types.js +361 -0
- package/vendor-core/findings-of-type.js +11 -0
- package/vendor-core/format-utils.js +92 -27
- package/vendor-core/html-export.js +51 -0
- package/vendor-core/impact-band.js +21 -8
- package/vendor-core/impact-estimator.js +8 -521
- package/vendor-core/impact-format.js +114 -0
- package/vendor-core/impact-model.js +175 -0
- package/vendor-core/ingest.js +2 -0
- package/vendor-core/intervals.js +13 -0
- package/vendor-core/list-runs.js +2 -3
- package/vendor-core/load-vendored.js +70 -5
- package/vendor-core/mcp-server-factory.js +14 -10
- package/vendor-core/mcp-tools.js +105 -45
- package/vendor-core/model-assembler.js +12 -0
- package/vendor-core/occupancy.js +1 -1
- package/vendor-core/parser-worker.js +1 -1
- package/vendor-core/plan-graph-model.js +3 -2
- package/vendor-core/plan-node-detail.js +1 -1
- package/vendor-core/recommendation-rollup.js +63 -3
- package/vendor-core/redact.js +45 -27
- package/vendor-core/run-comparison.js +32 -7
- package/vendor-core/run-interpretation.js +290 -0
- package/vendor-core/run-outcome.js +74 -0
- package/vendor-core/run-payload.js +17 -0
- package/vendor-core/run-shape.js +40 -0
- package/vendor-core/run-verdict.js +353 -0
- package/vendor-core/scaling-sim.js +4 -5
- package/vendor-core/scorecard-estimates.js +62 -0
- package/vendor-core/sql-stages.js +11 -0
- package/vendor-core/stage-quantiles.js +2 -0
- package/vendor-core/threshold-overrides.js +160 -0
- package/vendor-core/threshold-summary.js +11 -33
- package/vendor-core/types.js +6 -42
- package/vendor-core/wall-clock.js +1 -12
- package/vendor-core/wasted-core-hours.js +2 -2
package/vendor-core/mcp-tools.js
CHANGED
|
@@ -6,16 +6,21 @@ import { collectRun } from './cli/collect-run.js';
|
|
|
6
6
|
import { deriveEvidenceAvailability } from './evidence-availability.js';
|
|
7
7
|
import { resolveFromShs, DEFAULT_MAX_ARCHIVE_BYTES, DEFAULT_IDLE_TIMEOUT_MS } from './shs-load.js';
|
|
8
8
|
import { mcpError } from './mcp-error.js';
|
|
9
|
-
import { buildEvidenceReport, toFindingsFilter,
|
|
10
|
-
import {
|
|
9
|
+
import { buildEvidenceReport, toFindingsFilter, } from './evidence-report.js';
|
|
10
|
+
import { redactComparison } from './redact.js';
|
|
11
11
|
import { computeWallClock } from './wall-clock.js';
|
|
12
12
|
import { analyze } from './analyzer.js';
|
|
13
13
|
import { buildComparison, renderComparisonMarkdown, } from './run-comparison.js';
|
|
14
|
+
import { comparisonVerdict, } from './comparison-verdict.js';
|
|
14
15
|
import { evaluateBudgets, } from './cli/budgets.js';
|
|
15
16
|
import { FINDING_NAMES, titleCase } from './finding-names.js';
|
|
16
|
-
import { docAnchorForType
|
|
17
|
+
import { docAnchorForType } from './detector-docs.js';
|
|
18
|
+
import { DETECTORS, } from './detectors.js';
|
|
19
|
+
import { tunedDetectors } from './threshold-overrides.js';
|
|
20
|
+
import { tuningDocSlugForAnchor, pageForAnchor } from './docs-config.js';
|
|
17
21
|
import { typeTag } from './format-utils.js';
|
|
18
|
-
|
|
22
|
+
|
|
23
|
+
|
|
19
24
|
|
|
20
25
|
|
|
21
26
|
// RunRef uses a nested `source` key, matching every real call site (resolveOrCreateRun's
|
|
@@ -33,13 +38,22 @@ import { typeTag } from './format-utils.js';
|
|
|
33
38
|
|
|
34
39
|
|
|
35
40
|
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
|
|
36
48
|
|
|
37
49
|
// compareRuns returns a smaller MCP-facing projection of CompareRunsResult
|
|
38
|
-
// (runIdA/runIdB/findingsDelta/metricDeltas/confidence/reason/matchedCoverage), not the full
|
|
39
|
-
// shape (no baselineLabel/stageSkew/baseStages/candStages).
|
|
50
|
+
// (runIdA/runIdB/verdict/findingsDelta/metricDeltas/confidence/reason/matchedCoverage), not the full
|
|
51
|
+
// raw shape (no baselineLabel/stageSkew/baseStages/candStages/jobOutcomes).
|
|
40
52
|
|
|
41
53
|
|
|
42
54
|
|
|
55
|
+
|
|
56
|
+
|
|
43
57
|
|
|
44
58
|
|
|
45
59
|
|
|
@@ -169,20 +183,29 @@ export async function resolveOrCreateRun(
|
|
|
169
183
|
return resolution;
|
|
170
184
|
}
|
|
171
185
|
|
|
186
|
+
// `thresholds` on every analyzing tool below: the server's --thresholds overrides, fixed for the
|
|
187
|
+
// process by createMcpServer(). A client can't set them per call; the dashboard never has them.
|
|
172
188
|
export function diagnoseRun(runId , opts
|
|
173
189
|
|
|
174
|
-
|
|
190
|
+
|
|
175
191
|
)
|
|
176
|
-
|
|
177
|
-
|
|
192
|
+
|
|
193
|
+
|
|
194
|
+
|
|
178
195
|
{
|
|
179
196
|
const appModel = getCachedAppModel(runId);
|
|
180
197
|
const findingsFilter = toFindingsFilter(opts?.impactBand, opts?.type, opts?.stageId);
|
|
181
|
-
const { json, markdown } = buildEvidenceReport(appModel, {
|
|
198
|
+
const { json, markdown } = buildEvidenceReport(appModel, {
|
|
199
|
+
redact: opts?.redact, markdown: opts?.markdown, findingsFilter, thresholds: opts?.thresholds,
|
|
200
|
+
});
|
|
182
201
|
const include = opts?.include ?? [];
|
|
202
|
+
const tuned = json.summary.tunedThresholds;
|
|
183
203
|
return {
|
|
184
|
-
runId, findings: json.findings, recommendations: json.recommendations, cleanChecks: json.cleanChecks,
|
|
204
|
+
runId, verdict: json.verdict, findings: json.findings, recommendations: json.recommendations, cleanChecks: json.cleanChecks,
|
|
205
|
+
notRunChecks: json.notRunChecks,
|
|
185
206
|
runComplete: appModel.app?.endTime != null,
|
|
207
|
+
// Top level too, so a client that never asks for `summary` still sees the run was tuned.
|
|
208
|
+
...(tuned ? { tunedThresholds: tuned } : {}),
|
|
186
209
|
...(include.includes('summary') ? { summary: json.summary } : {}),
|
|
187
210
|
...(include.includes('evidenceAvailability') ? { evidenceAvailability: json.evidenceAvailability } : {}),
|
|
188
211
|
...(include.includes('detectors') ? { detectors: json.detectors } : {}),
|
|
@@ -211,17 +234,26 @@ function extractDocTitle(markdown ) {
|
|
|
211
234
|
|
|
212
235
|
|
|
213
236
|
|
|
237
|
+
// A finding type documents itself; a detector-level type that never appears on a finding
|
|
238
|
+
// (broadcastSizing) documents the types its entry emits.
|
|
239
|
+
function documentedTypes(type ) {
|
|
240
|
+
if (FINDING_NAMES[type] !== undefined) return [type];
|
|
241
|
+
return DETECTORS.find((entry) => entry.type === type)?.emits ?? [];
|
|
242
|
+
}
|
|
243
|
+
|
|
214
244
|
/** Detection + tuning reference documentation for one finding `type`, independent of any run
|
|
215
|
-
* (documentation is a property of the type: a client fetches it once per type and caches it).
|
|
245
|
+
* (documentation is a property of the type: a client fetches it once per type and caches it).
|
|
246
|
+
* A detector-level `type` resolves to the documentation of the finding types its entry emits. */
|
|
216
247
|
export function getFindingDocumentation(type ) {
|
|
217
|
-
const
|
|
218
|
-
|
|
248
|
+
const types = documentedTypes(type);
|
|
249
|
+
const [docType] = types;
|
|
250
|
+
if (docType === undefined) throw mcpError('invalid-type', `Unknown finding type: ${type}`);
|
|
219
251
|
|
|
220
|
-
const tag = typeTag(
|
|
252
|
+
const tag = typeTag(docType);
|
|
221
253
|
const detectionContent = readFileSync(join(DOCS_CONTENT_DIR, 'detection', `${tag.toLowerCase()}.md`), 'utf8');
|
|
222
254
|
const detectionDoc = { tag, title: extractDocTitle(detectionContent), content: detectionContent };
|
|
223
255
|
|
|
224
|
-
const anchor = docAnchorForType(
|
|
256
|
+
const anchor = docAnchorForType(docType);
|
|
225
257
|
const slug = anchor ? tuningDocSlugForAnchor(anchor) : null;
|
|
226
258
|
const tuningPath = slug ? join(DOCS_CONTENT_DIR, 'tuning', `${slug}.md`) : null;
|
|
227
259
|
let tuningDoc = null;
|
|
@@ -235,7 +267,8 @@ export function getFindingDocumentation(type ) {
|
|
|
235
267
|
if (entry) tuningDoc = { anchor, title: entry.title, content: readNavEntryContent(entry) };
|
|
236
268
|
}
|
|
237
269
|
|
|
238
|
-
|
|
270
|
+
const name = types.map((t) => titleCase(FINDING_NAMES[t] ?? t)).join(' / ');
|
|
271
|
+
return { type, name, detectionDoc, tuningDoc };
|
|
239
272
|
}
|
|
240
273
|
|
|
241
274
|
const CHAPTERS_NAV_FILE = join(DOCS_CONTENT_DIR, 'chapters', 'nav-index.json');
|
|
@@ -266,10 +299,10 @@ export function getReferenceDoc(anchor ) {
|
|
|
266
299
|
}
|
|
267
300
|
|
|
268
301
|
export function getFindingEvidence(
|
|
269
|
-
runId , findingId , opts
|
|
302
|
+
runId , findingId , opts ,
|
|
270
303
|
) {
|
|
271
304
|
const appModel = getCachedAppModel(runId);
|
|
272
|
-
const { json } = buildEvidenceReport(appModel, { redact: opts?.redact, markdown: false });
|
|
305
|
+
const { json } = buildEvidenceReport(appModel, { redact: opts?.redact, markdown: false, thresholds: opts?.thresholds });
|
|
273
306
|
const finding = json.findings.find((f) => f.id === findingId);
|
|
274
307
|
if (!finding) throw mcpError('finding-not-found', `No finding ${findingId} on run ${runId}.`);
|
|
275
308
|
return { runId, finding };
|
|
@@ -279,15 +312,19 @@ function hasCompleteInterval(app ) {
|
|
|
279
312
|
return Number.isFinite(app?.startTime) && Number.isFinite(app?.endTime) && (app?.endTime ?? 0) > (app?.startTime ?? 0);
|
|
280
313
|
}
|
|
281
314
|
|
|
282
|
-
export function getRunSummary(runId , opts
|
|
315
|
+
export function getRunSummary(runId , opts ) {
|
|
283
316
|
const appModel = getCachedAppModel(runId);
|
|
284
317
|
const { app, stages, jobs, sql, executors } = appModel;
|
|
285
318
|
const durationMs = hasCompleteInterval(app) ? computeWallClock(app, stages).total : null;
|
|
286
|
-
//
|
|
287
|
-
//
|
|
288
|
-
//
|
|
319
|
+
// The failure reason needs the stageFailed findings, so this reads the (cached) evidence report.
|
|
320
|
+
// With redact, the app identity comes from that same redacted report, so a host token in the app
|
|
321
|
+
// name and in Spark's failure reason get the same pseudonym.
|
|
322
|
+
const { summary } = buildEvidenceReport(appModel, { redact: opts?.redact, markdown: false, thresholds: opts?.thresholds }).json;
|
|
323
|
+
const { outcome } = summary;
|
|
289
324
|
const rawApp = { id: app?.id ?? null, name: app?.name ?? null, sparkVersion: app?.sparkVersion ?? null };
|
|
290
|
-
const redactedApp = opts?.redact
|
|
325
|
+
const redactedApp = opts?.redact
|
|
326
|
+
? { id: summary.app.id ?? null, name: summary.app.name ?? null, sparkVersion: summary.app.sparkVersion ?? null }
|
|
327
|
+
: rawApp;
|
|
291
328
|
return {
|
|
292
329
|
runId,
|
|
293
330
|
app: redactedApp,
|
|
@@ -297,27 +334,31 @@ export function getRunSummary(runId , opts )
|
|
|
297
334
|
executorCount: { added: executors.added.length, removed: executors.removed.length },
|
|
298
335
|
durationMs,
|
|
299
336
|
runComplete: app?.endTime != null,
|
|
337
|
+
...outcome,
|
|
338
|
+
runShape: summary.runShape,
|
|
300
339
|
};
|
|
301
340
|
}
|
|
302
341
|
|
|
303
342
|
// Shared by compareRuns and evaluateBudgetsForRun: both need a resolved run's finding catalog
|
|
304
343
|
// (same analyze() call shape) first.
|
|
305
|
-
async function resolveAndAnalyze(
|
|
344
|
+
async function resolveAndAnalyze(
|
|
345
|
+
ref , thresholds ,
|
|
346
|
+
) {
|
|
306
347
|
const { runId, appModel } = await resolveOrCreateRun(ref);
|
|
307
348
|
const catalog = analyze(
|
|
308
349
|
appModel.app, appModel.stages, appModel.executors.added, appModel.executors.removed,
|
|
309
|
-
appModel.jobs, appModel.sql, appModel.runAggregates,
|
|
350
|
+
appModel.jobs, appModel.sql, appModel.runAggregates, { thresholds },
|
|
310
351
|
);
|
|
311
352
|
return { runId, appModel, catalog };
|
|
312
353
|
}
|
|
313
354
|
|
|
314
355
|
export async function compareRuns(
|
|
315
|
-
a , b , opts
|
|
316
|
-
)
|
|
356
|
+
a , b , opts ,
|
|
357
|
+
) {
|
|
317
358
|
const [
|
|
318
359
|
{ runId: runIdA, appModel: appModelA, catalog: catalogA },
|
|
319
360
|
{ runId: runIdB, appModel: appModelB, catalog: catalogB },
|
|
320
|
-
] = await Promise.all([resolveAndAnalyze(a), resolveAndAnalyze(b)]);
|
|
361
|
+
] = await Promise.all([resolveAndAnalyze(a, opts?.thresholds), resolveAndAnalyze(b, opts?.thresholds)]);
|
|
321
362
|
|
|
322
363
|
// buildComparison's captureSnapshot uses an empty taskDataCache: that arg only feeds the
|
|
323
364
|
// interactive stage-detail drill-down, which none of compare/matchStages/metricDeltas/findingsDelta
|
|
@@ -330,44 +371,63 @@ export async function compareRuns(
|
|
|
330
371
|
// Stage names throughout `built` carry raw Spark stage text, which can embed a host/IP token as
|
|
331
372
|
// free text, the same residual redactReport() already scrubs from the evidence report.
|
|
332
373
|
const result = opts?.redact ? redactComparison(built) : built;
|
|
374
|
+
const verdict = comparisonVerdict(result);
|
|
333
375
|
|
|
334
376
|
return {
|
|
335
377
|
runIdA,
|
|
336
378
|
runIdB,
|
|
379
|
+
verdict,
|
|
337
380
|
findingsDelta: result.findings,
|
|
338
381
|
metricDeltas: result.metrics,
|
|
339
382
|
confidence: result.confidence,
|
|
340
383
|
reason: result.reason,
|
|
341
384
|
matchedCoverage: result.matchedCoverage,
|
|
342
|
-
...(opts?.
|
|
385
|
+
...tunedField(opts?.thresholds),
|
|
386
|
+
...(opts?.markdown ? { markdown: renderComparisonMarkdown(result, verdict, tunedDetectors(opts?.thresholds)) } : {}),
|
|
343
387
|
};
|
|
344
388
|
}
|
|
345
389
|
|
|
390
|
+
// Both runs of a comparison use the same overrides, so a tuned delta compares like with like.
|
|
391
|
+
function tunedField(thresholds ) {
|
|
392
|
+
const tuned = tunedDetectors(thresholds);
|
|
393
|
+
return tuned ? { tunedThresholds: tuned } : {};
|
|
394
|
+
}
|
|
395
|
+
|
|
346
396
|
export async function evaluateBudgetsForRun(
|
|
347
397
|
primary ,
|
|
348
398
|
budgets ,
|
|
349
399
|
secondary ,
|
|
350
|
-
|
|
400
|
+
opts ,
|
|
401
|
+
)
|
|
402
|
+
|
|
403
|
+
|
|
404
|
+
{
|
|
351
405
|
// Mirrors the CLI's --regression-metric/--max-regression-pct pairing guard: unlike the CLI, this
|
|
352
406
|
// tool never defaults regressionMetric, so seeing it set here means the caller asked for a
|
|
353
407
|
// regression check and forgot the threshold, evaluateBudgets() would otherwise skip it silently.
|
|
354
408
|
if (budgets.regressionMetric !== undefined && budgets.maxRegressionPct === undefined) {
|
|
355
409
|
throw mcpError('access-or-upstream-failure', 'regressionMetric requires maxRegressionPct.');
|
|
356
410
|
}
|
|
357
|
-
const [
|
|
358
|
-
resolveAndAnalyze(primary),
|
|
359
|
-
secondary ? resolveAndAnalyze(secondary) : Promise.resolve(undefined),
|
|
411
|
+
const [first, second] = await Promise.all([
|
|
412
|
+
resolveAndAnalyze(primary, opts?.thresholds),
|
|
413
|
+
secondary ? resolveAndAnalyze(secondary, opts?.thresholds) : Promise.resolve(undefined),
|
|
360
414
|
]);
|
|
361
415
|
|
|
362
|
-
|
|
363
|
-
|
|
364
|
-
|
|
365
|
-
|
|
366
|
-
|
|
367
|
-
|
|
368
|
-
|
|
369
|
-
|
|
370
|
-
|
|
371
|
-
|
|
372
|
-
|
|
416
|
+
// Same roles as the CLI's positional run + --baseline: with a second run, `primary` is the
|
|
417
|
+
// baseline and `secondary` the candidate, and absolute budgets plus the run-complete check
|
|
418
|
+
// apply to the candidate.
|
|
419
|
+
const candidate = second ?? first;
|
|
420
|
+
const baseline = second ? first : undefined;
|
|
421
|
+
const comparison = baseline
|
|
422
|
+
? buildComparison(
|
|
423
|
+
{ label: baseline.runId, appModel: baseline.appModel, catalog: baseline.catalog },
|
|
424
|
+
{ label: candidate.runId, appModel: candidate.appModel, catalog: candidate.catalog },
|
|
425
|
+
)
|
|
426
|
+
: undefined;
|
|
427
|
+
|
|
428
|
+
const { results, violated, inconclusive } = evaluateBudgets({
|
|
429
|
+
appModel: candidate.appModel, catalog: candidate.catalog, budgets, comparison, thresholds: opts?.thresholds,
|
|
430
|
+
});
|
|
431
|
+
|
|
432
|
+
return { runId: first.runId, results, violated, inconclusive, ...tunedField(opts?.thresholds) };
|
|
373
433
|
}
|
|
@@ -63,6 +63,18 @@ export function createModelCallbacks(
|
|
|
63
63
|
if (stage) stage.executorMetrics = execMetrics;
|
|
64
64
|
}
|
|
65
65
|
},
|
|
66
|
+
// Patch speculation totals that grew after StageCompleted: Spark kills a losing speculative
|
|
67
|
+
// copy only once its stage finishes. `data` is Map<stageId, { speculationWasteMs, speculationWastedAttempts }>.
|
|
68
|
+
onStageSpeculationWaste(data ) {
|
|
69
|
+
const totalsByStage = data ;
|
|
70
|
+
for (const [stageId, totals] of totalsByStage) {
|
|
71
|
+
const stage = appModel.stages.get(stageId);
|
|
72
|
+
if (stage) {
|
|
73
|
+
stage.speculationWasteMs = totals.speculationWasteMs;
|
|
74
|
+
stage.speculationWastedAttempts = totals.speculationWastedAttempts;
|
|
75
|
+
}
|
|
76
|
+
}
|
|
77
|
+
},
|
|
66
78
|
onDone, onError,
|
|
67
79
|
};
|
|
68
80
|
}
|
package/vendor-core/occupancy.js
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
// Occupancy-weighted wall-clock attribution for per-finding impact: every
|
|
2
2
|
// stage's wall-clock claim is apportioned purely from its own observed window
|
|
3
3
|
// and how much it overlapped with other stages. No graph, no parentIds traversal.
|
|
4
|
-
import { mergeIntervals } from './
|
|
4
|
+
import { mergeIntervals } from './intervals.js';
|
|
5
5
|
|
|
6
6
|
|
|
7
7
|
|
|
@@ -12,7 +12,7 @@ export {
|
|
|
12
12
|
buildChunkDecoder, createState, normalizeSparkProperties, parseSparkMemoryMB, extractResources,
|
|
13
13
|
accumulateTask, resolvePlanTree, startApplication, updateEnvironment, startJob, endJob, submitStage,
|
|
14
14
|
mergeStageRddInfo, recordStageExecutorMetrics, startSqlExecution, endSqlExecution,
|
|
15
|
-
applyDriverAccumUpdates, addExecutor, removeExecutor, processEvent, dispatchLine,
|
|
15
|
+
applyDriverAccumUpdates, addExecutor, removeExecutor, recordBlockUpdate, processEvent, dispatchLine,
|
|
16
16
|
collectStageExecutorMetrics,
|
|
17
17
|
} from './event-handlers.js';
|
|
18
18
|
|
|
@@ -15,7 +15,7 @@ import {
|
|
|
15
15
|
formatPlanMetricValue,
|
|
16
16
|
} from './plan-node-detail.js';
|
|
17
17
|
import { computeSegments, mapSegmentsToStagesForDisplay } from './plan-duration-attribution.js';
|
|
18
|
-
import { stageIdsForSqlExec } from './
|
|
18
|
+
import { stageIdsForSqlExec } from './sql-stages.js';
|
|
19
19
|
|
|
20
20
|
|
|
21
21
|
|
|
@@ -121,7 +121,8 @@ export function buildPlanGraphModel(
|
|
|
121
121
|
const findingsByNodeId = new Map ();
|
|
122
122
|
if (sqlExecutionId != null) {
|
|
123
123
|
for (const finding of findings) {
|
|
124
|
-
if (finding.executionId !== sqlExecutionId) continue;
|
|
124
|
+
if (!('executionId' in finding) || finding.executionId !== sqlExecutionId) continue;
|
|
125
|
+
// `?? []`: a hand-built or foreign finding may still lack the field the type promises.
|
|
125
126
|
for (const nodeId of finding.planNodeIds ?? []) {
|
|
126
127
|
const list = findingsByNodeId.get(nodeId);
|
|
127
128
|
if (list) list.push(finding);
|
|
@@ -5,7 +5,7 @@
|
|
|
5
5
|
|
|
6
6
|
import { pathBasename, formatBytes, formatDuration } from './format-utils.js';
|
|
7
7
|
import { attributeStageDurationToPlan, attributeStageDurationToPlanInclusive } from './plan-duration-attribution.js';
|
|
8
|
-
import { stageIdsForSqlExec } from './
|
|
8
|
+
import { stageIdsForSqlExec } from './sql-stages.js';
|
|
9
9
|
|
|
10
10
|
|
|
11
11
|
const BOILERPLATE_PREFIXES = ['serializefromobject', 'deserializetoobject', 'mapelements',
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
// Explicit .ts extensions: plain Node's ESM resolver (the runtime CLI/MCP path
|
|
2
2
|
// runs under) requires the exact specifier, unlike a bundler.
|
|
3
|
-
import { mergeIntervals } from './
|
|
4
|
-
import { worstImpactBand, IMPACT_BAND_ORDER } from './format-utils.js';
|
|
3
|
+
import { mergeIntervals } from './intervals.js';
|
|
4
|
+
import { formatWallClockRange, worstImpactBand, IMPACT_BAND_ORDER } from './format-utils.js';
|
|
5
5
|
|
|
6
6
|
|
|
7
7
|
|
|
@@ -48,7 +48,7 @@ function groupByType(findings ) {
|
|
|
48
48
|
}
|
|
49
49
|
|
|
50
50
|
function stageIdsOf(finding ) {
|
|
51
|
-
if (finding
|
|
51
|
+
if ('stageIds' in finding) return finding.stageIds;
|
|
52
52
|
if (finding.stageId != null) return [finding.stageId];
|
|
53
53
|
return [];
|
|
54
54
|
}
|
|
@@ -124,6 +124,31 @@ export function buildRecommendationRollup(
|
|
|
124
124
|
});
|
|
125
125
|
}
|
|
126
126
|
|
|
127
|
+
/** A group's trailing figure on the Findings board, and its tooltip. `time`
|
|
128
|
+
* and `resource` both lead with the "×N" finding count (the "worth
|
|
129
|
+
* expanding" signal); `count` skips it since the impact-band tally already
|
|
130
|
+
* implies N. The row stays terse ("×2 · 476ms recoverable") to fit a dense
|
|
131
|
+
* right-aligned column; the title spells the shorthand out. */
|
|
132
|
+
export function rollupGroupStat(group ) {
|
|
133
|
+
if (group.kind === 'time') {
|
|
134
|
+
const recoverable = formatWallClockRange(group.recoverableMsHigh, group.recoverableMsHigh);
|
|
135
|
+
return {
|
|
136
|
+
stat: `×${group.findingCount} · ${recoverable} recoverable`,
|
|
137
|
+
statTitle: `${group.findingCount} findings of this type; up to ${recoverable} of run time could be recovered by fixing them`,
|
|
138
|
+
};
|
|
139
|
+
}
|
|
140
|
+
if (group.kind === 'resource') {
|
|
141
|
+
return {
|
|
142
|
+
stat: `×${group.findingCount} · resource-cost projection`,
|
|
143
|
+
statTitle: `${group.findingCount} findings of this type; a resource-cost estimate (not run time) is projected for fixing them`,
|
|
144
|
+
};
|
|
145
|
+
}
|
|
146
|
+
return {
|
|
147
|
+
stat: Object.entries(group.byImpactBand).map(([impactBand, count]) => `${count} ${impactBand}`).join(', '),
|
|
148
|
+
statTitle: `${group.findingCount} findings of this type, by impact`,
|
|
149
|
+
};
|
|
150
|
+
}
|
|
151
|
+
|
|
127
152
|
/** True when a finding is real evidence of an issue, as opposed to a mere
|
|
128
153
|
* evidence-unavailable caveat. Shared by `isEligible` below and by anything
|
|
129
154
|
* else that decides whether a REGISTRY type has "something to show" (an
|
|
@@ -150,6 +175,10 @@ export function isEligible(finding ) {
|
|
|
150
175
|
// `isRealFinding`, this exclusion doesn't apply beyond the rollup: an
|
|
151
176
|
// incompleteRun finding still backs its own ordinary active widget card.
|
|
152
177
|
if (finding.type === 'incompleteRun') return false;
|
|
178
|
+
// A missing-evidence caveat (memoryUtilization's is already excluded by isRealFinding;
|
|
179
|
+
// cacheUtilization's storageUnobserved keeps its widget card, so a run with no block updates
|
|
180
|
+
// never lists Cache Storage as a passed check) is never a fix to rank.
|
|
181
|
+
if ('dataUnavailable' in finding && finding.dataUnavailable) return false;
|
|
153
182
|
return isRealFinding(finding);
|
|
154
183
|
}
|
|
155
184
|
|
|
@@ -191,3 +220,34 @@ export function rankFindings(findings ) {
|
|
|
191
220
|
return (IMPACT_BAND_ORDER[a.impactBand] ?? 9) - (IMPACT_BAND_ORDER[b.impactBand] ?? 9);
|
|
192
221
|
});
|
|
193
222
|
}
|
|
223
|
+
|
|
224
|
+
/** A board group ready to render: members ranked representative first, the representative's
|
|
225
|
+
* band (the heading the group sits under) and the group's trailing figure. */
|
|
226
|
+
|
|
227
|
+
|
|
228
|
+
|
|
229
|
+
|
|
230
|
+
|
|
231
|
+
|
|
232
|
+
|
|
233
|
+
|
|
234
|
+
|
|
235
|
+
|
|
236
|
+
/** The Findings board's groups over `eligible`, in fix-first order. The run's interpretation
|
|
237
|
+
* carries this for every eligible finding; the board recomputes it only over a filtered subset. */
|
|
238
|
+
export function rankedRollup(
|
|
239
|
+
eligible ,
|
|
240
|
+
stages ,
|
|
241
|
+
) {
|
|
242
|
+
return buildRecommendationRollup(eligible, stages).map((group) => {
|
|
243
|
+
const members = rankFindings(group.findings);
|
|
244
|
+
return {
|
|
245
|
+
kind: group.kind,
|
|
246
|
+
type: group.type,
|
|
247
|
+
unit: group.kind === 'resource' ? group.unit : null,
|
|
248
|
+
band: members[0].impactBand,
|
|
249
|
+
members,
|
|
250
|
+
...rollupGroupStat(group),
|
|
251
|
+
};
|
|
252
|
+
});
|
|
253
|
+
}
|
package/vendor-core/redact.js
CHANGED
|
@@ -7,7 +7,8 @@
|
|
|
7
7
|
// Deterministic (sorted assignment), idempotent (pseudonyms map to themselves),
|
|
8
8
|
// and non-mutating (returns a fresh, deep-copied tree).
|
|
9
9
|
|
|
10
|
-
|
|
10
|
+
import { decodeCollections, encodeCollections, } from './export-data.js';
|
|
11
|
+
|
|
11
12
|
import { redactTaskFailureGroup, } from './task-failure.js';
|
|
12
13
|
|
|
13
14
|
// Host / IP identifier patterns. Used to enumerate host names that surface only
|
|
@@ -35,7 +36,7 @@ const APP_ID_PATTERNS = [/\bapplication_\d{10,}_\d+\b/g];
|
|
|
35
36
|
// Walk every string in the tree once, collecting matches for each `{ patterns,
|
|
36
37
|
// out }` sink. One shared traversal for every token kind (instead of one
|
|
37
38
|
// traversal per kind) keeps redactComparison's dual host+app-id scan the same
|
|
38
|
-
// cost as the single-kind scan redactReport
|
|
39
|
+
// cost as the single-kind scan redactReport already does.
|
|
39
40
|
function scanTokens(node , sinks ) {
|
|
40
41
|
if (typeof node === 'string') {
|
|
41
42
|
for (const { patterns, out } of sinks) {
|
|
@@ -55,11 +56,6 @@ function scanTokens(node , sinks
|
|
|
55
56
|
}
|
|
56
57
|
}
|
|
57
58
|
|
|
58
|
-
// Walk every string in the tree, collecting host/IP tokens into `hosts`.
|
|
59
|
-
function scanHostTokens(node , hosts ) {
|
|
60
|
-
scanTokens(node, [{ patterns: HOST_PATTERNS, out: hosts }]);
|
|
61
|
-
}
|
|
62
|
-
|
|
63
59
|
// Recursively collects every string value found under a key literally named
|
|
64
60
|
// `host`, anywhere in the tree. Host names surface at several depths, a
|
|
65
61
|
// finding's own `host`, `evidence.host`, and now
|
|
@@ -104,7 +100,7 @@ function redactFailureGroups (node ) {
|
|
|
104
100
|
// HOST_PATTERNS matches (no IP/EC2 shape) nor collectHostFields's by-key-name
|
|
105
101
|
// walk catches (the literal key is the dotted Spark property name, never
|
|
106
102
|
// `host` itself). app.config is a flat Record<string, string> unique to
|
|
107
|
-
//
|
|
103
|
+
// redactRunModel: no other redact* export ships a raw Spark config dict.
|
|
108
104
|
// Known gap: a hostname value under a differently-named key isn't caught by
|
|
109
105
|
// this suffix check. Confirmed against a real cluster config: spark.master,
|
|
110
106
|
// spark.yarn.historyServer.address, and the plural YARN proxy/HA keys
|
|
@@ -198,24 +194,6 @@ export function redactReport (input ) {
|
|
|
198
194
|
return applyReplacements(report, { appIds, hosts });
|
|
199
195
|
}
|
|
200
196
|
|
|
201
|
-
// Narrow counterpart to redactReport(), for getRunSummary()'s standalone app
|
|
202
|
-
// object (no findings tree to walk). There's exactly one app id here, so no
|
|
203
|
-
// Set/Map/sort is needed for it; name/sparkVersion still go through the
|
|
204
|
-
// shared host/IP scan-and-replace since either can carry a host token as
|
|
205
|
-
// free text.
|
|
206
|
-
export function redactAppIdentity(
|
|
207
|
-
app ,
|
|
208
|
-
) {
|
|
209
|
-
const hosts = new Set ();
|
|
210
|
-
scanHostTokens(app.name, hosts);
|
|
211
|
-
scanHostTokens(app.sparkVersion, hosts);
|
|
212
|
-
return {
|
|
213
|
-
id: typeof app.id === 'string' && app.id.length > 0 ? 'app-1' : app.id,
|
|
214
|
-
name: applyReplacements(app.name, { hosts }),
|
|
215
|
-
sparkVersion: applyReplacements(app.sparkVersion, { hosts }),
|
|
216
|
-
};
|
|
217
|
-
}
|
|
218
|
-
|
|
219
197
|
// Run-comparison counterpart: no single app-id *field* to pseudonymize
|
|
220
198
|
// (baselineLabel/candidateLabel are caller-supplied labels, not Spark app
|
|
221
199
|
// ids), but stage names surface throughout the tree (FindingsDeltaRow.stages,
|
|
@@ -237,7 +215,9 @@ export function redactComparison (comparison ) {
|
|
|
237
215
|
// this also walks executors.added/removed for their literal `host` field
|
|
238
216
|
// (ExecutorAddedEvent.host), since raw executor records: not just findings
|
|
239
217
|
//: reach data.js.
|
|
240
|
-
|
|
218
|
+
|
|
219
|
+
|
|
220
|
+
function redactRunTree (input ) {
|
|
241
221
|
const data = redactFailureGroups(input);
|
|
242
222
|
const appIds = new Set ();
|
|
243
223
|
const hosts = new Set ();
|
|
@@ -251,3 +231,41 @@ export function redactExportData(input ) {
|
|
|
251
231
|
scanTokens(data, [{ patterns: HOST_PATTERNS, out: hosts }, { patterns: APP_ID_PATTERNS, out: appIds }]);
|
|
252
232
|
return applyReplacements(data, { appIds, hosts });
|
|
253
233
|
}
|
|
234
|
+
|
|
235
|
+
/** A run's model and findings with every identifier pseudonymized. Redact this before anything derives
|
|
236
|
+
* text from the run: the verdict truncates Spark's failure reason, and an
|
|
237
|
+
* identifier cut by that truncation is a fragment no later pass can match. */
|
|
238
|
+
export function redactRunModel(
|
|
239
|
+
appModel ,
|
|
240
|
+
catalog ,
|
|
241
|
+
configFindings ,
|
|
242
|
+
) {
|
|
243
|
+
// Maps and Sets would lose their entries in the deep copy, so they cross as
|
|
244
|
+
// tagged plain objects, the way the export payload carries them.
|
|
245
|
+
const tree = encodeCollections({
|
|
246
|
+
app: appModel.app,
|
|
247
|
+
stages: [...appModel.stages.values()],
|
|
248
|
+
jobs: [...appModel.jobs.values()],
|
|
249
|
+
sql: [...appModel.sql.values()],
|
|
250
|
+
executors: appModel.executors,
|
|
251
|
+
runAggregates: appModel.runAggregates,
|
|
252
|
+
evidenceAvailability: appModel.evidenceAvailability,
|
|
253
|
+
catalog,
|
|
254
|
+
configFindings,
|
|
255
|
+
}) ;
|
|
256
|
+
const redacted = decodeCollections(redactRunTree(tree)) ;
|
|
257
|
+
const byId = (items ) => new Map((items ).map((item) => [item.id, item]));
|
|
258
|
+
return {
|
|
259
|
+
appModel: {
|
|
260
|
+
app: redacted.app,
|
|
261
|
+
stages: byId(redacted.stages),
|
|
262
|
+
jobs: byId(redacted.jobs),
|
|
263
|
+
sql: byId(redacted.sql),
|
|
264
|
+
executors: redacted.executors,
|
|
265
|
+
runAggregates: redacted.runAggregates ,
|
|
266
|
+
evidenceAvailability: redacted.evidenceAvailability ,
|
|
267
|
+
} ,
|
|
268
|
+
catalog: redacted.catalog,
|
|
269
|
+
configFindings: redacted.configFindings,
|
|
270
|
+
};
|
|
271
|
+
}
|