sparkforensics-mcp 0.3.0 → 0.5.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/bin/sparkforensics-mcp.mjs +41 -10
- package/package.json +1 -1
- package/vendor-core/allocation.js +106 -0
- package/vendor-core/analyzer.js +168 -60
- package/vendor-core/check-coverage.js +88 -0
- package/vendor-core/cli/budgets.js +54 -27
- package/vendor-core/cli/collect-run.js +84 -32
- package/vendor-core/cli/regression-budgets.js +83 -0
- package/vendor-core/cli/threshold-config.js +28 -0
- package/vendor-core/comparison-verdict.js +177 -0
- package/vendor-core/core-source-hash.txt +1 -0
- package/vendor-core/core-usage-locality.js +56 -2
- package/vendor-core/detector-docs.js +58 -0
- package/vendor-core/detectors.js +1094 -500
- package/vendor-core/docs-config.js +0 -36
- package/vendor-core/docs-content/chapters/nav-index.json +31 -0
- package/vendor-core/docs-content/detection/cache.md +3 -2
- package/vendor-core/docs-content/detection/cfg.md +9 -8
- package/vendor-core/docs-content/detection/chrn.md +1 -2
- package/vendor-core/docs-content/detection/cold.md +4 -2
- package/vendor-core/docs-content/detection/cstor.md +9 -0
- package/vendor-core/docs-content/detection/fail.md +3 -2
- package/vendor-core/docs-content/detection/gc.md +3 -2
- package/vendor-core/docs-content/detection/host.md +2 -1
- package/vendor-core/docs-content/detection/local.md +1 -1
- package/vendor-core/docs-content/detection/mem.md +5 -2
- package/vendor-core/docs-content/detection/plan.md +2 -1
- package/vendor-core/docs-content/detection/sfail.md +2 -1
- package/vendor-core/docs-content/detection/shape.md +5 -4
- package/vendor-core/docs-content/detection/skew.md +3 -1
- package/vendor-core/docs-content/detection/slow.md +2 -2
- package/vendor-core/docs-content/detection/spec.md +2 -3
- package/vendor-core/docs-content/detection/spill.md +1 -1
- package/vendor-core/docs-site-config.js +3 -0
- package/vendor-core/effective-conf.js +107 -0
- package/vendor-core/efficiency-model.js +8 -6
- package/vendor-core/event-handlers.js +321 -44
- package/vendor-core/event-schemas.js +23 -0
- package/vendor-core/evidence-report.js +432 -115
- package/vendor-core/export-data.js +79 -6
- package/vendor-core/finding-action-label.js +9 -88
- package/vendor-core/finding-filter-predicate.js +9 -0
- package/vendor-core/finding-generic-recommendation.js +26 -105
- package/vendor-core/finding-names.js +28 -45
- package/vendor-core/finding-presentation.js +368 -0
- package/vendor-core/finding-tag-help.js +110 -0
- package/vendor-core/finding-types.js +373 -0
- package/vendor-core/findings-of-type.js +11 -0
- package/vendor-core/format-utils.js +96 -30
- package/vendor-core/html-export.js +51 -0
- package/vendor-core/impact-band.js +21 -8
- package/vendor-core/impact-estimator.js +25 -520
- package/vendor-core/impact-format.js +115 -0
- package/vendor-core/impact-model.js +197 -0
- package/vendor-core/ingest.js +6 -2
- package/vendor-core/intervals.js +13 -0
- package/vendor-core/list-runs.js +7 -5
- package/vendor-core/load-vendored.js +70 -5
- package/vendor-core/mcp-server-factory.js +14 -10
- package/vendor-core/mcp-tools.js +105 -45
- package/vendor-core/model-assembler.js +35 -1
- package/vendor-core/occupancy.js +1 -1
- package/vendor-core/parser-worker.js +2 -2
- package/vendor-core/plan-graph-model.js +3 -2
- package/vendor-core/plan-node-detail.js +1 -1
- package/vendor-core/proxy.js +3 -1
- package/vendor-core/python-stage.js +25 -0
- package/vendor-core/recommendation-rollup.js +70 -3
- package/vendor-core/redact.js +96 -37
- package/vendor-core/remediation.js +20 -0
- package/vendor-core/run-comparison.js +73 -29
- package/vendor-core/run-interpretation.js +291 -0
- package/vendor-core/run-metrics.js +198 -0
- package/vendor-core/run-outcome.js +74 -0
- package/vendor-core/run-payload.js +17 -0
- package/vendor-core/run-shape.js +40 -0
- package/vendor-core/run-totals.js +24 -0
- package/vendor-core/run-verdict.js +352 -0
- package/vendor-core/scaling-sim.js +4 -5
- package/vendor-core/scorecard-estimates.js +63 -0
- package/vendor-core/session-snapshot.js +7 -0
- package/vendor-core/shs-schemas.js +2 -2
- package/vendor-core/spark-memory.js +17 -0
- package/vendor-core/sql-stages.js +11 -0
- package/vendor-core/stage-plan-nodes.js +18 -0
- package/vendor-core/stage-quantiles.js +6 -0
- package/vendor-core/threshold-overrides.js +160 -0
- package/vendor-core/threshold-summary.js +11 -33
- package/vendor-core/types.js +54 -42
- package/vendor-core/wall-clock.js +1 -12
- package/vendor-core/wasted-core-hours.js +12 -9
- package/vendor-core/write-targets.js +312 -0
|
@@ -0,0 +1,352 @@
|
|
|
1
|
+
// The run verdict: where to start, a short summary, and the top places to look, ranked by
|
|
2
|
+
// potential savings (failures first on a failed run). Shared by the dashboard's verdict card and
|
|
3
|
+
// the CLI/MCP evidence report, so both paths name the same first step in the same words.
|
|
4
|
+
import { ENTRY_BY_TYPE } from './detectors.js';
|
|
5
|
+
import { hasFinishedStage, isCleanRun } from './check-coverage.js';
|
|
6
|
+
import { findingActionLabel } from './finding-action-label.js';
|
|
7
|
+
import { singleStageId } from './finding-filter-predicate.js';
|
|
8
|
+
import { recommendationText } from './finding-names.js';
|
|
9
|
+
import { formatDuration, IMPACT_BAND_ORDER } from './format-utils.js';
|
|
10
|
+
import { impactFigure, savingsMeaning } from './impact-format.js';
|
|
11
|
+
import { isEligible } from './recommendation-rollup.js';
|
|
12
|
+
import { FAILURE_TYPES, quotesReasonOf, summarizeRunOutcome, } from './run-outcome.js';
|
|
13
|
+
import { getScorecardEstimates, hasCompleteApplicationInterval } from './scorecard-estimates.js';
|
|
14
|
+
import { computeWallClock } from './wall-clock.js';
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
/** How many next steps the verdict lists before pointing at the full list. */
|
|
18
|
+
export const NEXT_STEP_LIMIT = 3;
|
|
19
|
+
|
|
20
|
+
/** Finding types whose widget sits in the board's reference region (listed after every action
|
|
21
|
+
* widget). The dashboard's registry must agree: tests/view/detector-registry.test.tsx checks it. */
|
|
22
|
+
export const REFERENCE_DISPLAY_TYPES = new Set([
|
|
23
|
+
'memoryUtilization', 'utilization', 'coreLocality', 'cacheUtilization',
|
|
24
|
+
]);
|
|
25
|
+
|
|
26
|
+
/** Every emitted finding type in the board's widget display order: action region before
|
|
27
|
+
* reference region, then ascending order of the type's `ENTRY_BY_TYPE` entry.
|
|
28
|
+
* The last tiebreak of the verdict ranking, so the CLI orders ties exactly as the dashboard. */
|
|
29
|
+
export const FINDING_DISPLAY_ORDER = (() => {
|
|
30
|
+
const region = (type ) => (REFERENCE_DISPLAY_TYPES.has(type) ? 1 : 0);
|
|
31
|
+
return [...ENTRY_BY_TYPE.entries()]
|
|
32
|
+
.sort(([a, entryA], [b, entryB]) => region(a) - region(b) || entryA.order - entryB.order)
|
|
33
|
+
.map(([type]) => type);
|
|
34
|
+
})();
|
|
35
|
+
|
|
36
|
+
const DISPLAY_INDEX = new Map(FINDING_DISPLAY_ORDER.map((type, index) => [type, index]));
|
|
37
|
+
|
|
38
|
+
// The high end of the finding's own occupancy-clipped wall-clock estimate, the figure the
|
|
39
|
+
// "Potential savings" line leads with. `null` with no quantified time claim
|
|
40
|
+
// (resourceOnly/informational basis): such a finding can never win on its own numbers.
|
|
41
|
+
function potentialSavingsMs(finding ) {
|
|
42
|
+
return finding.impactEstimate?.wallClock?.high ?? null;
|
|
43
|
+
}
|
|
44
|
+
|
|
45
|
+
/** True for a finding the verdict can route to: a known display type with a recommendation. */
|
|
46
|
+
export function isRankable(finding ) {
|
|
47
|
+
const recommendation = typeof finding.recommendation === 'string' ? finding.recommendation.trim() : '';
|
|
48
|
+
return DISPLAY_INDEX.has(finding.type) && recommendation.length > 0;
|
|
49
|
+
}
|
|
50
|
+
|
|
51
|
+
/** Every rankable finding, best first. Ranked by potential savings: a quantified estimate always
|
|
52
|
+
* outranks an unquantified one; ties (including "neither has one") fall back to impact band,
|
|
53
|
+
* then widget display order, then input order, so an unquantified warning still leads an info. */
|
|
54
|
+
export function rankBySavings(findings ) {
|
|
55
|
+
const candidates = findings
|
|
56
|
+
.map((finding, index) => ({ finding, index }))
|
|
57
|
+
.filter(({ finding }) => isRankable(finding));
|
|
58
|
+
candidates.sort((left, right) => {
|
|
59
|
+
const leftSavings = potentialSavingsMs(left.finding);
|
|
60
|
+
const rightSavings = potentialSavingsMs(right.finding);
|
|
61
|
+
if (leftSavings !== null && rightSavings !== null && leftSavings !== rightSavings) return rightSavings - leftSavings;
|
|
62
|
+
if ((leftSavings !== null) !== (rightSavings !== null)) return leftSavings !== null ? -1 : 1;
|
|
63
|
+
return (
|
|
64
|
+
(IMPACT_BAND_ORDER[left.finding.impactBand] ?? 9) - (IMPACT_BAND_ORDER[right.finding.impactBand] ?? 9)
|
|
65
|
+
|| (DISPLAY_INDEX.get(left.finding.type) ?? Number.MAX_SAFE_INTEGER) - (DISPLAY_INDEX.get(right.finding.type) ?? Number.MAX_SAFE_INTEGER)
|
|
66
|
+
|| left.index - right.index
|
|
67
|
+
);
|
|
68
|
+
});
|
|
69
|
+
return candidates.map(({ finding }) => finding);
|
|
70
|
+
}
|
|
71
|
+
|
|
72
|
+
/** One place worth a look: the highest-ranked finding at a location, plus the
|
|
73
|
+
* other finding types flagged at that same location. Findings that share a
|
|
74
|
+
* stage usually share one root cause, so they read as one step, not several. */
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
|
|
83
|
+
|
|
84
|
+
/** A finding's location identity for grouping. Per-stage findings (and
|
|
85
|
+
* sql-scope findings that touch exactly one stage) group by that stage; a
|
|
86
|
+
* multi-stage or app-level finding stands alone by its own type (and variant,
|
|
87
|
+
* for app-level ones), since two different app-level problems are not the
|
|
88
|
+
* same place. */
|
|
89
|
+
export function locationKey(finding ) {
|
|
90
|
+
const stageId = singleStageId(finding);
|
|
91
|
+
if (stageId != null) return { key: `stage:${stageId}`, stageId };
|
|
92
|
+
if ('stageIds' in finding && finding.stageIds.length > 1) {
|
|
93
|
+
return { key: `stages:${finding.type}:${[...finding.stageIds].sort((a, b) => a - b).join(',')}`, stageId: null };
|
|
94
|
+
}
|
|
95
|
+
const variant = 'variant' in finding ? finding.variant : undefined;
|
|
96
|
+
return { key: variant ? `app:${finding.type}:${variant}` : `app:${finding.type}`, stageId: null };
|
|
97
|
+
}
|
|
98
|
+
|
|
99
|
+
/** 0 for a failure at a stage of a failed job (what stopped the job), 1 for
|
|
100
|
+
* any other failure finding, 2 for everything else. */
|
|
101
|
+
function failureRank(finding , failedJobStageIds ) {
|
|
102
|
+
if (!FAILURE_TYPES.has(finding.type)) return 2;
|
|
103
|
+
return typeof finding.stageId === 'number' && failedJobStageIds.has(finding.stageId) ? 0 : 1;
|
|
104
|
+
}
|
|
105
|
+
|
|
106
|
+
/** Groups every rankable finding by location, ordered by the location's
|
|
107
|
+
* best-ranked finding, so step 1 is always the run's single biggest win. With
|
|
108
|
+
* `failedJobStageIds` (a run whose jobs failed), failure findings rank ahead
|
|
109
|
+
* of every savings figure and lead their location, those at a failed job's
|
|
110
|
+
* stage first: a speed-up is moot until the job finishes. */
|
|
111
|
+
export function buildNextSteps(findings , { failedJobStageIds } = {}) {
|
|
112
|
+
const ranked = rankBySavings(findings);
|
|
113
|
+
// Array.prototype.sort is stable, so savings order holds within each rank.
|
|
114
|
+
const ordered = failedJobStageIds
|
|
115
|
+
? [...ranked].sort((a, b) => failureRank(a, failedJobStageIds) - failureRank(b, failedJobStageIds))
|
|
116
|
+
: ranked;
|
|
117
|
+
const steps = new Map ();
|
|
118
|
+
for (const finding of ordered) {
|
|
119
|
+
const { key, stageId } = locationKey(finding);
|
|
120
|
+
const existing = steps.get(key);
|
|
121
|
+
if (!existing) {
|
|
122
|
+
steps.set(key, { key, lead: finding, related: [], stageId });
|
|
123
|
+
continue;
|
|
124
|
+
}
|
|
125
|
+
const seenTypes = new Set([existing.lead.type, ...existing.related.map((f) => f.type)]);
|
|
126
|
+
if (!seenTypes.has(finding.type)) existing.related.push(finding);
|
|
127
|
+
}
|
|
128
|
+
return [...steps.values()];
|
|
129
|
+
}
|
|
130
|
+
|
|
131
|
+
/** True for findings about executor capacity sitting idle. memoryUtilization
|
|
132
|
+
* also reports heap pressure and over-provisioning, which are not idle
|
|
133
|
+
* capacity, so only its idleCores variant counts. */
|
|
134
|
+
function isIdleCapacityFinding(finding ) {
|
|
135
|
+
return finding.type === 'utilization' || (finding.type === 'memoryUtilization' && finding.variant === 'idleCores');
|
|
136
|
+
}
|
|
137
|
+
|
|
138
|
+
/** Idle share at which the verdict notes, in its summary, that the cluster
|
|
139
|
+
* may be larger than the job needs. It never reorders the steps. */
|
|
140
|
+
export const IDLE_NOTABLE_PCT = 40;
|
|
141
|
+
|
|
142
|
+
export function isIdleCapacityStep(step ) {
|
|
143
|
+
return isIdleCapacityFinding(step.lead);
|
|
144
|
+
}
|
|
145
|
+
|
|
146
|
+
/** The idle share an idle-capacity finding itself reports: idleCores carries
|
|
147
|
+
* the idle rate, utilization the busy rate. Null for any other finding. */
|
|
148
|
+
function reportedIdlePct(finding ) {
|
|
149
|
+
if (typeof finding.value !== 'number') return null;
|
|
150
|
+
if (finding.type === 'memoryUtilization' && finding.variant === 'idleCores') return finding.value;
|
|
151
|
+
if (finding.type === 'utilization') return 100 - finding.value;
|
|
152
|
+
return null;
|
|
153
|
+
}
|
|
154
|
+
|
|
155
|
+
/** The run's idle share as the verdict states it: the figure the top-ranked
|
|
156
|
+
* idle-capacity step reports, so the verdict never disagrees with that step,
|
|
157
|
+
* or `fallbackPct` (the Scorecard's Unused core time) when no step reports one. */
|
|
158
|
+
export function verdictIdlePct(steps , fallbackPct ) {
|
|
159
|
+
const idleStep = steps.find(isIdleCapacityStep);
|
|
160
|
+
return (idleStep ? reportedIdlePct(idleStep.lead) : null) ?? fallbackPct;
|
|
161
|
+
}
|
|
162
|
+
|
|
163
|
+
function plural(count , noun ) {
|
|
164
|
+
return `${count} ${noun}${count === 1 ? '' : 's'}`;
|
|
165
|
+
}
|
|
166
|
+
|
|
167
|
+
/** The run facts the verdict's wording depends on, read once. */
|
|
168
|
+
|
|
169
|
+
|
|
170
|
+
|
|
171
|
+
|
|
172
|
+
|
|
173
|
+
|
|
174
|
+
|
|
175
|
+
|
|
176
|
+
|
|
177
|
+
|
|
178
|
+
|
|
179
|
+
|
|
180
|
+
|
|
181
|
+
|
|
182
|
+
|
|
183
|
+
|
|
184
|
+
function isFailedRun(facts ) {
|
|
185
|
+
return facts.outcome.failedJobs > 0;
|
|
186
|
+
}
|
|
187
|
+
|
|
188
|
+
/** The failed-run title: the one thing a newcomer must know before any
|
|
189
|
+
* tuning advice is that the job did not finish. */
|
|
190
|
+
function failedTitle({ failedJobs, totalJobs } ) {
|
|
191
|
+
if (failedJobs < totalJobs) return `${failedJobs} of ${totalJobs} jobs failed in this run`;
|
|
192
|
+
return totalJobs === 1 ? 'This run failed: its job did not finish' : `This run failed: all ${totalJobs} jobs did not finish`;
|
|
193
|
+
}
|
|
194
|
+
|
|
195
|
+
/** An action label as it reads after "Start here:": only its first letter
|
|
196
|
+
* drops to lower case, so a name inside it ("Switch to Kryo") keeps its
|
|
197
|
+
* capital, and a leading acronym ("GC", "OOM") is left alone. */
|
|
198
|
+
function lowerFirst(label ) {
|
|
199
|
+
if (/^[A-Z]{2}/.test(label)) return label;
|
|
200
|
+
return label.charAt(0).toLowerCase() + label.slice(1);
|
|
201
|
+
}
|
|
202
|
+
|
|
203
|
+
export function verdictTitle(eligible , steps , facts ) {
|
|
204
|
+
if (isFailedRun(facts)) return failedTitle(facts.outcome);
|
|
205
|
+
if (eligible.length === 0 && facts.incomplete) return 'This log looks incomplete, so results cover only part of the run';
|
|
206
|
+
if (eligible.length === 0 && facts.noFinishedStages) return 'This log has no finished stages to check';
|
|
207
|
+
if (eligible.length === 0 && !facts.clean) return 'Nothing to fix, but some checks could not run on this log';
|
|
208
|
+
if (eligible.length === 0) return 'No findings to fix right now.';
|
|
209
|
+
// Every real detector writes a recommendation, so an eligible finding with
|
|
210
|
+
// no route is a defensive case: still never call such a run clean.
|
|
211
|
+
if (steps.length === 0) return `${plural(eligible.length, 'finding')} to review`;
|
|
212
|
+
const lead = steps[0];
|
|
213
|
+
if (isIdleCapacityStep(lead) && facts.idlePct != null) return `Start with cluster size: ${facts.idlePct}% of executor capacity sat idle`;
|
|
214
|
+
if (lead.stageId != null) return `Start with Stage ${lead.stageId}`;
|
|
215
|
+
return `Start here: ${lowerFirst(findingActionLabel(lead.lead))}`;
|
|
216
|
+
}
|
|
217
|
+
|
|
218
|
+
/** The run-level summary under the title: how much was found and where, what
|
|
219
|
+
* the first fix is worth, and the run's idle capacity when that is large
|
|
220
|
+
* enough to matter but is not the first step (whose title already says it). */
|
|
221
|
+
export function verdictSummary(eligible , steps , facts ) {
|
|
222
|
+
const sentences = [];
|
|
223
|
+
const { failedJobs, totalJobs } = facts.outcome;
|
|
224
|
+
if (failedJobs > 0) {
|
|
225
|
+
if (eligible.some((finding) => !FAILURE_TYPES.has(finding.type))) {
|
|
226
|
+
sentences.push('Fix the failure before tuning: the other findings cover only the work that ran.');
|
|
227
|
+
}
|
|
228
|
+
} else if (totalJobs > 0 && !facts.incomplete) {
|
|
229
|
+
sentences.push(totalJobs === 1 ? 'Its one job succeeded.' : `All ${totalJobs} jobs succeeded.`);
|
|
230
|
+
}
|
|
231
|
+
if (eligible.length === 0) {
|
|
232
|
+
if (failedJobs > 0) return sentences;
|
|
233
|
+
if (facts.clean) sentences.push('Every check passed for this run.');
|
|
234
|
+
} else if (steps.length === 0) {
|
|
235
|
+
sentences.push('They are listed by impact under Findings.');
|
|
236
|
+
} else {
|
|
237
|
+
sentences.push(`${plural(eligible.length, 'finding')} in ${plural(steps.length, 'place')}.`);
|
|
238
|
+
const wallClock = steps[0].lead.impactEstimate?.wallClock;
|
|
239
|
+
// An idle-capacity lead needs no sentence here: the title gives the idle share and step 1 the fix.
|
|
240
|
+
if (!isIdleCapacityStep(steps[0]) && wallClock && facts.runMs != null) {
|
|
241
|
+
sentences.push(`The first fix could save up to ${formatDuration(wallClock.high)} of this ${formatDuration(facts.runMs)} run.`);
|
|
242
|
+
}
|
|
243
|
+
if (steps.some((step) => step.related.length > 0)) {
|
|
244
|
+
sentences.push('Findings on one stage are grouped, and their savings overlap.');
|
|
245
|
+
}
|
|
246
|
+
}
|
|
247
|
+
if (facts.incomplete) {
|
|
248
|
+
sentences.push('The log has no end-of-run record, so these figures cover only the part of the run it captured.');
|
|
249
|
+
}
|
|
250
|
+
const leadIsIdle = steps.length > 0 && isIdleCapacityStep(steps[0]);
|
|
251
|
+
if (!leadIsIdle && facts.idlePct != null && facts.idlePct >= IDLE_NOTABLE_PCT) {
|
|
252
|
+
sentences.push(
|
|
253
|
+
steps.some(isIdleCapacityStep)
|
|
254
|
+
? `${facts.idlePct}% of the executor capacity sat idle, so the cluster may be larger than this job needs.`
|
|
255
|
+
: `${facts.idlePct}% of the run's core time went unused, so the cluster may be larger than this job needs.`,
|
|
256
|
+
);
|
|
257
|
+
}
|
|
258
|
+
return sentences;
|
|
259
|
+
}
|
|
260
|
+
|
|
261
|
+
const STACK_TRACE_HINT = 'Open the driver log only if you need the full stack trace.';
|
|
262
|
+
|
|
263
|
+
/** What the failure step whose reason the verdict quotes tells the reader:
|
|
264
|
+
* the detector's "inspect the driver log for the reason" would send a
|
|
265
|
+
* newcomer looking for something already on screen. The copied text carries
|
|
266
|
+
* the reason itself, since "quoted above" means nothing once pasted. */
|
|
267
|
+
export function quotedReasonText(reason ) {
|
|
268
|
+
const sentence = /[.!?]$/.test(reason) ? reason : `${reason}.`;
|
|
269
|
+
return {
|
|
270
|
+
shown: `Spark's recorded reason is quoted above. ${STACK_TRACE_HINT}`,
|
|
271
|
+
copied: `Spark's recorded reason: ${sentence} ${STACK_TRACE_HINT}`,
|
|
272
|
+
};
|
|
273
|
+
}
|
|
274
|
+
|
|
275
|
+
/** One step as pasteable text: the action, what to try, and the savings. */
|
|
276
|
+
export function stepCopyText(finding , recommendation , stageId = null) {
|
|
277
|
+
const impact = impactFigure(finding);
|
|
278
|
+
const meaning = savingsMeaning(finding);
|
|
279
|
+
const savings = impact && meaning ? `${impact} ${meaning}` : impact;
|
|
280
|
+
const where = stageId != null ? ` in Stage ${stageId}` : '';
|
|
281
|
+
const headline = `${findingActionLabel(finding)}${where}: ${recommendation}`;
|
|
282
|
+
return [/[.!?]$/.test(headline) ? headline : `${headline}.`, savings ? `Potential savings: ${savings}` : null]
|
|
283
|
+
.filter(Boolean)
|
|
284
|
+
.join(' ');
|
|
285
|
+
}
|
|
286
|
+
|
|
287
|
+
/** The whole verdict as a pasteable checklist for a ticket or a message:
|
|
288
|
+
* run, verdict, numbered steps (with their stage), and how many more places
|
|
289
|
+
* the full list holds. */
|
|
290
|
+
export function planCopyText(input
|
|
291
|
+
|
|
292
|
+
|
|
293
|
+
|
|
294
|
+
|
|
295
|
+
) {
|
|
296
|
+
const lines = [input.runName ? `Spark run ${input.runName}: ${input.title}` : input.title, ''];
|
|
297
|
+
input.steps.forEach(({ step, recommendation }, index) => {
|
|
298
|
+
lines.push(`${index + 1}. ${stepCopyText(step.lead, recommendation, step.stageId)}`);
|
|
299
|
+
});
|
|
300
|
+
if (input.remaining > 0) lines.push('', `${plural(input.remaining, 'more place')} to look at in the full findings list.`);
|
|
301
|
+
return lines.join('\n');
|
|
302
|
+
}
|
|
303
|
+
|
|
304
|
+
/** A step's "What to try" text for copying: Spark's quoted reason when it is this step's own
|
|
305
|
+
* failure, else the finding's recommendation. */
|
|
306
|
+
export function stepCopyRecommendation(step , outcome ) {
|
|
307
|
+
return quotesReasonOf(step.lead, outcome) && outcome.reason
|
|
308
|
+
? quotedReasonText(outcome.reason).copied
|
|
309
|
+
: recommendationText(step.lead);
|
|
310
|
+
}
|
|
311
|
+
|
|
312
|
+
/** Everything the verdict card shows, computed once for a run. `eligible` is what the verdict
|
|
313
|
+
* counts and ranks: the rollup-eligible findings of a known display type. */
|
|
314
|
+
|
|
315
|
+
|
|
316
|
+
|
|
317
|
+
|
|
318
|
+
|
|
319
|
+
|
|
320
|
+
|
|
321
|
+
|
|
322
|
+
|
|
323
|
+
|
|
324
|
+
|
|
325
|
+
|
|
326
|
+
|
|
327
|
+
|
|
328
|
+
export function buildRunVerdict(appModel , allFindings ) {
|
|
329
|
+
const eligible = allFindings.filter((finding) => isEligible(finding) && DISPLAY_INDEX.has(finding.type));
|
|
330
|
+
const outcome = summarizeRunOutcome(appModel.jobs, allFindings);
|
|
331
|
+
const steps = buildNextSteps(eligible, outcome.failedJobs > 0 ? { failedJobStageIds: outcome.failedJobStageIds } : {});
|
|
332
|
+
const facts = {
|
|
333
|
+
runMs: hasCompleteApplicationInterval(appModel.app) ? computeWallClock(appModel.app, appModel.stages).total : null,
|
|
334
|
+
idlePct: verdictIdlePct(steps, getScorecardEstimates(appModel).wastage.value),
|
|
335
|
+
incomplete: allFindings.some((finding) => finding.type === 'incompleteRun'),
|
|
336
|
+
clean: isCleanRun(appModel, allFindings),
|
|
337
|
+
outcome,
|
|
338
|
+
noFinishedStages: !hasFinishedStage(appModel.stages),
|
|
339
|
+
};
|
|
340
|
+
const title = verdictTitle(eligible, steps, facts);
|
|
341
|
+
const shown = steps.slice(0, NEXT_STEP_LIMIT);
|
|
342
|
+
const remaining = steps.length - shown.length;
|
|
343
|
+
const copyText = shown.length > 0
|
|
344
|
+
? planCopyText({
|
|
345
|
+
runName: appModel.app?.name ?? null,
|
|
346
|
+
title,
|
|
347
|
+
steps: shown.map((step) => ({ step, recommendation: stepCopyRecommendation(step, outcome) })),
|
|
348
|
+
remaining,
|
|
349
|
+
})
|
|
350
|
+
: null;
|
|
351
|
+
return { eligible, outcome, steps, facts, title, summary: verdictSummary(eligible, steps, facts), shown, remaining, copyText };
|
|
352
|
+
}
|
|
@@ -6,7 +6,6 @@
|
|
|
6
6
|
// Builds on computeCoreTimeSeries' clamp conceptually, but operates on the
|
|
7
7
|
// worker-posted per-stage aggregates (runAggregates.perStage) so raw task data
|
|
8
8
|
// never reaches the main thread.
|
|
9
|
-
import { computeWallClock } from './wall-clock.js';
|
|
10
9
|
import { computeTotalCores } from './core-count.js';
|
|
11
10
|
|
|
12
11
|
|
|
@@ -30,9 +29,11 @@ function estimatedTotalAtCores(perStage
|
|
|
30
29
|
return sum;
|
|
31
30
|
}
|
|
32
31
|
|
|
33
|
-
|
|
32
|
+
/** `observedActiveMs` is the run's stages-active wall-clock (the interpretation's
|
|
33
|
+
* `wallClock.stagesActive`), the one observed makespan predictions are scaled against. */
|
|
34
|
+
export function simulateScaling({ app, observedActiveMs, runAggregates, executorsAdded }
|
|
35
|
+
|
|
34
36
|
|
|
35
|
-
|
|
36
37
|
|
|
37
38
|
|
|
38
39
|
)
|
|
@@ -48,8 +49,6 @@ export function simulateScaling({ app, stages, runAggregates, executorsAdded }
|
|
|
48
49
|
// `executorsAdded` cast: computeTotalCores only reads `totalCores`, present on
|
|
49
50
|
// ExecutorAddedEvent (the only kind passed here) but not on the union type.
|
|
50
51
|
const baselineCores = computeTotalCores(app ?? {}, executorsAdded );
|
|
51
|
-
const wc = computeWallClock(app, stages );
|
|
52
|
-
const observedActiveMs = wc.stagesActive;
|
|
53
52
|
|
|
54
53
|
// Model Error: predicted-at-baseline vs. observed stages-active wall-clock.
|
|
55
54
|
const predictedAtBaseline = baselineCores > 0 ? estimatedTotalAtCores(perStage, baselineCores) : 0;
|
|
@@ -0,0 +1,63 @@
|
|
|
1
|
+
import { computeEfficiencyModel } from './efficiency-model.js';
|
|
2
|
+
import { computeWallClock } from './wall-clock.js';
|
|
3
|
+
|
|
4
|
+
|
|
5
|
+
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
export function hasCompleteApplicationInterval(app ) {
|
|
16
|
+
return typeof app?.startTime === 'number'
|
|
17
|
+
&& Number.isFinite(app.startTime)
|
|
18
|
+
&& typeof app?.endTime === 'number'
|
|
19
|
+
&& Number.isFinite(app.endTime)
|
|
20
|
+
&& app.endTime > app.startTime;
|
|
21
|
+
}
|
|
22
|
+
|
|
23
|
+
export function getScorecardEstimates(appModel )
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
{
|
|
27
|
+
if (!hasCompleteApplicationInterval(appModel.app)) {
|
|
28
|
+
return {
|
|
29
|
+
efficiency: { value: null, unavailableReason: 'application-timing' },
|
|
30
|
+
wastage: { value: null, unavailableReason: 'application-timing' },
|
|
31
|
+
};
|
|
32
|
+
}
|
|
33
|
+
|
|
34
|
+
const wallClock = computeWallClock(appModel.app, appModel.stages);
|
|
35
|
+
const efficiency = Math.min(100, Math.round((wallClock.stagesActive / wallClock.total) * 100));
|
|
36
|
+
|
|
37
|
+
if (!appModel.runAggregates) {
|
|
38
|
+
return {
|
|
39
|
+
efficiency: { value: efficiency, unavailableReason: null },
|
|
40
|
+
wastage: { value: null, unavailableReason: 'core-usage-summary' },
|
|
41
|
+
};
|
|
42
|
+
}
|
|
43
|
+
|
|
44
|
+
const efficiencyModel = computeEfficiencyModel({
|
|
45
|
+
app: appModel.app,
|
|
46
|
+
stages: appModel.stages,
|
|
47
|
+
executorsAdded: appModel.executors.added,
|
|
48
|
+
executorsRemoved: appModel.executors.removed,
|
|
49
|
+
runAggregates: appModel.runAggregates,
|
|
50
|
+
});
|
|
51
|
+
|
|
52
|
+
if (!Number.isFinite(efficiencyModel.availableComputeHours) || efficiencyModel.availableComputeHours <= 0) {
|
|
53
|
+
return {
|
|
54
|
+
efficiency: { value: efficiency, unavailableReason: null },
|
|
55
|
+
wastage: { value: null, unavailableReason: 'executor-capacity' },
|
|
56
|
+
};
|
|
57
|
+
}
|
|
58
|
+
|
|
59
|
+
return {
|
|
60
|
+
efficiency: { value: efficiency, unavailableReason: null },
|
|
61
|
+
wastage: { value: efficiencyModel.wastagePct, unavailableReason: null },
|
|
62
|
+
};
|
|
63
|
+
}
|
|
@@ -24,6 +24,8 @@ import { isSupportedEvidenceAvailability } from './evidence-availability.js';
|
|
|
24
24
|
|
|
25
25
|
|
|
26
26
|
|
|
27
|
+
|
|
28
|
+
|
|
27
29
|
|
|
28
30
|
|
|
29
31
|
|
|
@@ -44,6 +46,8 @@ export function captureSnapshot(
|
|
|
44
46
|
jobs: new Map(appModel.jobs),
|
|
45
47
|
runAggregates: appModel.runAggregates,
|
|
46
48
|
evidenceAvailability: appModel.evidenceAvailability,
|
|
49
|
+
skippedLines: appModel.skippedLines,
|
|
50
|
+
unreadableSqlExecutions: appModel.unreadableSqlExecutions && [...appModel.unreadableSqlExecutions],
|
|
47
51
|
catalog: [...catalog],
|
|
48
52
|
taskData: new Map(taskDataCache),
|
|
49
53
|
};
|
|
@@ -71,6 +75,9 @@ export function applySnapshot(
|
|
|
71
75
|
appModel.evidenceAvailability = isSupportedEvidenceAvailability(snapshot.evidenceAvailability)
|
|
72
76
|
? snapshot.evidenceAvailability
|
|
73
77
|
: null;
|
|
78
|
+
appModel.skippedLines = snapshot.skippedLines;
|
|
79
|
+
if (snapshot.unreadableSqlExecutions) appModel.unreadableSqlExecutions = [...snapshot.unreadableSqlExecutions];
|
|
80
|
+
else delete appModel.unreadableSqlExecutions;
|
|
74
81
|
|
|
75
82
|
taskDataCache.clear();
|
|
76
83
|
for (const [k, v] of snapshot.taskData) taskDataCache.set(k, v);
|
|
@@ -1,8 +1,8 @@
|
|
|
1
1
|
import { z } from 'zod';
|
|
2
2
|
|
|
3
3
|
// Proxy-level error envelope: the JSON body the local server's SHS proxy
|
|
4
|
-
// (./
|
|
5
|
-
// `{ code: '
|
|
4
|
+
// (./proxy.js) sends back on a non-OK upstream response, e.g.
|
|
5
|
+
// `{ code: 'upstream-unreachable' }`. This is NOT a SparkListener* event shape, so
|
|
6
6
|
// it lives here rather than in event-schemas.ts. `.passthrough()` since the
|
|
7
7
|
// proxy may attach extra debugging fields the consumer doesn't care about;
|
|
8
8
|
// only `code` is read.
|
|
@@ -0,0 +1,17 @@
|
|
|
1
|
+
// Parse a Spark memory-size string to MiB. Spark's JVM-memory configs use bytesConf(ByteUnit.MiB),
|
|
2
|
+
// so a bare number means MiB. A k/m/g/t suffix sets the unit (trailing "b" redundant); a lone "b"
|
|
3
|
+
// ("10b") means bytes.
|
|
4
|
+
export function parseSparkMemoryMB(value ) {
|
|
5
|
+
if (value == null) return null;
|
|
6
|
+
const m = String(value).trim().toLowerCase().match(/^([\d.]+)\s*([kmgt]?)(b?)$/);
|
|
7
|
+
if (!m) return null;
|
|
8
|
+
const n = parseFloat(m[1]);
|
|
9
|
+
if (!Number.isFinite(n)) return null;
|
|
10
|
+
switch (m[2]) {
|
|
11
|
+
case 'k': return Math.round(n / 1024);
|
|
12
|
+
case 'g': return Math.round(n * 1024);
|
|
13
|
+
case 't': return Math.round(n * 1024 * 1024);
|
|
14
|
+
case 'm': return Math.round(n);
|
|
15
|
+
default: return m[3] === 'b' ? Math.round(n / (1024 * 1024)) : Math.round(n);
|
|
16
|
+
}
|
|
17
|
+
}
|
|
@@ -0,0 +1,11 @@
|
|
|
1
|
+
// Shared by every scope:'sql' detector and the plan views. sql.get(id).stageIds is always empty (parser-worker
|
|
2
|
+
// never populates it), so stage linkage is derived from each stage's own sqlExecutionId.
|
|
3
|
+
// Minimal param shape (not DetectorStage): external callers pass Map<StageId, Stage>.
|
|
4
|
+
export function stageIdsForSqlExec(
|
|
5
|
+
executionId ,
|
|
6
|
+
stages ,
|
|
7
|
+
) {
|
|
8
|
+
const out = [];
|
|
9
|
+
for (const s of stages.values()) if (s.sqlExecutionId === executionId) out.push(s.id);
|
|
10
|
+
return out;
|
|
11
|
+
}
|
|
@@ -0,0 +1,18 @@
|
|
|
1
|
+
|
|
2
|
+
|
|
3
|
+
/** The plan nodes a stage actually ran: its SQL execution's resolved plan tree, filtered to the
|
|
4
|
+
* nodes attributed to it (`node.stageIds`). Empty when the stage has no SQL execution, the
|
|
5
|
+
* execution has no resolved plan, or no node is attributed to it. The one stage-to-plan mapping:
|
|
6
|
+
* stageIdentity's fingerprint and the Python-stage check both read it. */
|
|
7
|
+
export function planNodesOfStage(stage , sql ) {
|
|
8
|
+
const execId = stage.sqlExecutionId;
|
|
9
|
+
if (execId == null) return [];
|
|
10
|
+
const root = sql.get(execId)?.planTree ?? null;
|
|
11
|
+
if (!root) return [];
|
|
12
|
+
const nodes = [];
|
|
13
|
+
(function collect(node ) {
|
|
14
|
+
if (node.stageIds?.includes(stage.id)) nodes.push(node);
|
|
15
|
+
for (const child of node.children ?? []) collect(child);
|
|
16
|
+
})(root);
|
|
17
|
+
return nodes;
|
|
18
|
+
}
|
|
@@ -56,6 +56,7 @@ const TASK_FIELD_PROPS = TASK_FIELD_DESCRIPTORS.map((d) => d.prop);
|
|
|
56
56
|
|
|
57
57
|
|
|
58
58
|
|
|
59
|
+
|
|
59
60
|
|
|
60
61
|
|
|
61
62
|
export function finalizeStage(
|
|
@@ -124,9 +125,11 @@ export function finalizeStage(
|
|
|
124
125
|
acc.executorCpuTime += t.executorCpuTime;
|
|
125
126
|
acc.inputBytes += t.inputBytes;
|
|
126
127
|
acc.outputBytes += t.outputBytes;
|
|
128
|
+
if (t.outputRecords != null) acc.outputRecords = (acc.outputRecords ?? 0) + t.outputRecords;
|
|
127
129
|
for (let i = 0; i < TASK_FIELD_PROPS.length; i++) buf.push(t[TASK_FIELD_PROPS[i]]);
|
|
128
130
|
}
|
|
129
131
|
stage.taskCount = taskCount;
|
|
132
|
+
stage.peakExecutionMemoryMax = peakExecutionMemoryMax;
|
|
130
133
|
stage.failedTasks = failedTasks;
|
|
131
134
|
stage.speculativeTasks = speculativeTasks;
|
|
132
135
|
stage.taskAttempts = null; // no longer needed after finalize, freeing memory
|
|
@@ -199,6 +202,9 @@ export function finalizeStage(
|
|
|
199
202
|
};
|
|
200
203
|
delete data.taskAttempts; // internal-only field, already nulled above; never part of the public message
|
|
201
204
|
delete data.failureDetails; // internal-only intern table, summarized by failureGroups
|
|
205
|
+
delete data.speculativeWinners; // internal-only late-TaskEnd pairing state, kept worker-side
|
|
206
|
+
delete data.lateSpeculationWaste;
|
|
207
|
+
delete data.stageAttemptId;
|
|
202
208
|
|
|
203
209
|
return { type: 'stage', data };
|
|
204
210
|
}
|