sparkforensics-mcp 0.2.4 → 0.4.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -0
- package/README.md +7 -1
- package/bin/sparkforensics-mcp.mjs +41 -10
- package/package.json +1 -1
- package/vendor-core/analyzer.js +156 -48
- package/vendor-core/check-coverage.js +88 -0
- package/vendor-core/cli/budgets.js +31 -18
- package/vendor-core/cli/collect-run.js +76 -31
- package/vendor-core/cli/native-zstd.js +2 -2
- package/vendor-core/cli/threshold-config.js +28 -0
- package/vendor-core/comparison-verdict.js +177 -0
- package/vendor-core/core-source-hash.txt +1 -0
- package/vendor-core/core-usage-locality.js +56 -2
- package/vendor-core/detector-docs.js +58 -0
- package/vendor-core/detectors.js +933 -459
- package/vendor-core/docs-config.js +0 -36
- package/vendor-core/docs-content/chapters/nav-index.json +31 -0
- package/vendor-core/docs-content/detection/cstor.md +9 -0
- package/vendor-core/docs-content/detection/fail.md +6 -2
- package/vendor-core/docs-site-config.js +3 -0
- package/vendor-core/event-handlers.js +191 -6
- package/vendor-core/event-schemas.js +29 -0
- package/vendor-core/evidence-report.js +440 -112
- package/vendor-core/export-data.js +79 -6
- package/vendor-core/finding-action-label.js +9 -88
- package/vendor-core/finding-filter-predicate.js +9 -0
- package/vendor-core/finding-generic-recommendation.js +6 -104
- package/vendor-core/finding-names.js +21 -45
- package/vendor-core/finding-presentation.js +333 -0
- package/vendor-core/finding-tag-help.js +110 -0
- package/vendor-core/finding-types.js +361 -0
- package/vendor-core/findings-of-type.js +11 -0
- package/vendor-core/format-utils.js +92 -27
- package/vendor-core/html-export.js +51 -0
- package/vendor-core/impact-band.js +21 -8
- package/vendor-core/impact-estimator.js +8 -521
- package/vendor-core/impact-format.js +114 -0
- package/vendor-core/impact-model.js +175 -0
- package/vendor-core/ingest.js +2 -0
- package/vendor-core/intervals.js +13 -0
- package/vendor-core/list-runs.js +2 -3
- package/vendor-core/load-vendored.js +70 -5
- package/vendor-core/mcp-server-factory.js +14 -10
- package/vendor-core/mcp-tools.js +105 -45
- package/vendor-core/model-assembler.js +12 -0
- package/vendor-core/occupancy.js +1 -1
- package/vendor-core/parser-worker.js +22 -5
- package/vendor-core/plan-graph-model.js +3 -2
- package/vendor-core/plan-node-detail.js +1 -1
- package/vendor-core/recommendation-rollup.js +63 -3
- package/vendor-core/redact.js +68 -28
- package/vendor-core/run-comparison.js +40 -7
- package/vendor-core/run-interpretation.js +290 -0
- package/vendor-core/run-outcome.js +74 -0
- package/vendor-core/run-payload.js +17 -0
- package/vendor-core/run-shape.js +40 -0
- package/vendor-core/run-verdict.js +353 -0
- package/vendor-core/scaling-sim.js +4 -5
- package/vendor-core/scorecard-estimates.js +62 -0
- package/vendor-core/shs-fetch.js +175 -65
- package/vendor-core/shs-load.js +1 -1
- package/vendor-core/sql-stages.js +11 -0
- package/vendor-core/stage-quantiles.js +14 -0
- package/vendor-core/task-failure.js +151 -0
- package/vendor-core/threshold-overrides.js +160 -0
- package/vendor-core/threshold-summary.js +11 -33
- package/vendor-core/types.js +6 -42
- package/vendor-core/vendor/fflate.js +1 -1
- package/vendor-core/wall-clock.js +1 -12
- package/vendor-core/wasted-core-hours.js +2 -2
- package/vendor-core/zip-archive.js +167 -0
|
@@ -0,0 +1,40 @@
|
|
|
1
|
+
// The run's shape as the dashboard's Scorecard, ETL Phase Attribution and Core Usage by Locality
|
|
2
|
+
// cards show it, for the CLI/MCP paths: every figure comes from the same core function those cards
|
|
3
|
+
// read, and null means the card would show "Not measured" or "Unavailable".
|
|
4
|
+
import { hasFinishedStage } from './check-coverage.js';
|
|
5
|
+
import { buildLocalityChart } from './core-usage-locality.js';
|
|
6
|
+
import { attributeEtlPhases } from './etl-phases.js';
|
|
7
|
+
import { getScorecardEstimates, hasCompleteApplicationInterval } from './scorecard-estimates.js';
|
|
8
|
+
import { computeWallClock } from './wall-clock.js';
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
export function computeRunShape(appModel ) {
|
|
29
|
+
const timed = hasCompleteApplicationInterval(appModel.app);
|
|
30
|
+
const estimates = getScorecardEstimates(appModel);
|
|
31
|
+
const phases = attributeEtlPhases(appModel.stages);
|
|
32
|
+
const chart = buildLocalityChart([...appModel.stages.values()], appModel.app);
|
|
33
|
+
return {
|
|
34
|
+
wallClockMs: timed ? computeWallClock(appModel.app, appModel.stages).total : null,
|
|
35
|
+
efficiencyPct: hasFinishedStage(appModel.stages) ? estimates.efficiency.value : null,
|
|
36
|
+
unusedCoreTimePct: estimates.wastage.value,
|
|
37
|
+
etlPhasesMs: phases.extract + phases.transform + phases.load > 0 ? phases : null,
|
|
38
|
+
peakBusyCores: chart.hasActivity ? chart.peakCores : null,
|
|
39
|
+
};
|
|
40
|
+
}
|
|
@@ -0,0 +1,353 @@
|
|
|
1
|
+
// The run verdict: where to start, a short summary, and the top places to look, ranked by
|
|
2
|
+
// potential savings (failures first on a failed run). Shared by the dashboard's verdict card and
|
|
3
|
+
// the CLI/MCP evidence report, so both paths name the same first step in the same words.
|
|
4
|
+
import { ENTRY_BY_TYPE } from './detectors.js';
|
|
5
|
+
import { hasFinishedStage, isCleanRun } from './check-coverage.js';
|
|
6
|
+
import { findingActionLabel } from './finding-action-label.js';
|
|
7
|
+
import { singleStageId } from './finding-filter-predicate.js';
|
|
8
|
+
import { recommendationText } from './finding-names.js';
|
|
9
|
+
import { formatDuration, IMPACT_BAND_ORDER } from './format-utils.js';
|
|
10
|
+
import { impactFigure, savingsMeaning } from './impact-format.js';
|
|
11
|
+
import { isEligible } from './recommendation-rollup.js';
|
|
12
|
+
import { FAILURE_TYPES, quotesReasonOf, summarizeRunOutcome, } from './run-outcome.js';
|
|
13
|
+
import { getScorecardEstimates, hasCompleteApplicationInterval } from './scorecard-estimates.js';
|
|
14
|
+
import { computeWallClock } from './wall-clock.js';
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
/** How many next steps the verdict lists before pointing at the full list. */
|
|
18
|
+
export const NEXT_STEP_LIMIT = 3;
|
|
19
|
+
|
|
20
|
+
/** Finding types whose widget sits in the board's reference region (listed after every action
|
|
21
|
+
* widget). The dashboard's registry must agree: tests/view/detector-registry.test.tsx checks it. */
|
|
22
|
+
export const REFERENCE_DISPLAY_TYPES = new Set([
|
|
23
|
+
'memoryUtilization', 'utilization', 'coreLocality', 'cacheUtilization',
|
|
24
|
+
]);
|
|
25
|
+
|
|
26
|
+
/** Every emitted finding type in the board's widget display order: action region before
|
|
27
|
+
* reference region, then ascending order of the type's `ENTRY_BY_TYPE` entry.
|
|
28
|
+
* The last tiebreak of the verdict ranking, so the CLI orders ties exactly as the dashboard. */
|
|
29
|
+
export const FINDING_DISPLAY_ORDER = (() => {
|
|
30
|
+
const region = (type ) => (REFERENCE_DISPLAY_TYPES.has(type) ? 1 : 0);
|
|
31
|
+
return [...ENTRY_BY_TYPE.entries()]
|
|
32
|
+
.sort(([a, entryA], [b, entryB]) => region(a) - region(b) || entryA.order - entryB.order)
|
|
33
|
+
.map(([type]) => type);
|
|
34
|
+
})();
|
|
35
|
+
|
|
36
|
+
const DISPLAY_INDEX = new Map(FINDING_DISPLAY_ORDER.map((type, index) => [type, index]));
|
|
37
|
+
|
|
38
|
+
// The high end of the finding's own occupancy-clipped wall-clock estimate, the figure the
|
|
39
|
+
// "Potential savings" line leads with. `null` with no quantified time claim
|
|
40
|
+
// (resourceOnly/informational basis): such a finding can never win on its own numbers.
|
|
41
|
+
function potentialSavingsMs(finding ) {
|
|
42
|
+
return finding.impactEstimate?.wallClock?.high ?? null;
|
|
43
|
+
}
|
|
44
|
+
|
|
45
|
+
/** True for a finding the verdict can route to: a known display type with a recommendation. */
|
|
46
|
+
export function isRankable(finding ) {
|
|
47
|
+
const recommendation = typeof finding.recommendation === 'string' ? finding.recommendation.trim() : '';
|
|
48
|
+
return DISPLAY_INDEX.has(finding.type) && recommendation.length > 0;
|
|
49
|
+
}
|
|
50
|
+
|
|
51
|
+
/** Every rankable finding, best first. Ranked by potential savings: a quantified estimate always
|
|
52
|
+
* outranks an unquantified one; ties (including "neither has one") fall back to impact band,
|
|
53
|
+
* then widget display order, then input order, so an unquantified warning still leads an info. */
|
|
54
|
+
export function rankBySavings(findings ) {
|
|
55
|
+
const candidates = findings
|
|
56
|
+
.map((finding, index) => ({ finding, index }))
|
|
57
|
+
.filter(({ finding }) => isRankable(finding));
|
|
58
|
+
candidates.sort((left, right) => {
|
|
59
|
+
const leftSavings = potentialSavingsMs(left.finding);
|
|
60
|
+
const rightSavings = potentialSavingsMs(right.finding);
|
|
61
|
+
if (leftSavings !== null && rightSavings !== null && leftSavings !== rightSavings) return rightSavings - leftSavings;
|
|
62
|
+
if ((leftSavings !== null) !== (rightSavings !== null)) return leftSavings !== null ? -1 : 1;
|
|
63
|
+
return (
|
|
64
|
+
(IMPACT_BAND_ORDER[left.finding.impactBand] ?? 9) - (IMPACT_BAND_ORDER[right.finding.impactBand] ?? 9)
|
|
65
|
+
|| (DISPLAY_INDEX.get(left.finding.type) ?? Number.MAX_SAFE_INTEGER) - (DISPLAY_INDEX.get(right.finding.type) ?? Number.MAX_SAFE_INTEGER)
|
|
66
|
+
|| left.index - right.index
|
|
67
|
+
);
|
|
68
|
+
});
|
|
69
|
+
return candidates.map(({ finding }) => finding);
|
|
70
|
+
}
|
|
71
|
+
|
|
72
|
+
/** One place worth a look: the highest-ranked finding at a location, plus the
|
|
73
|
+
* other finding types flagged at that same location. Findings that share a
|
|
74
|
+
* stage usually share one root cause, so they read as one step, not several. */
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
|
|
83
|
+
|
|
84
|
+
/** A finding's location identity for grouping. Per-stage findings (and
|
|
85
|
+
* sql-scope findings that touch exactly one stage) group by that stage; a
|
|
86
|
+
* multi-stage or app-level finding stands alone by its own type (and variant,
|
|
87
|
+
* for app-level ones), since two different app-level problems are not the
|
|
88
|
+
* same place. */
|
|
89
|
+
export function locationKey(finding ) {
|
|
90
|
+
const stageId = singleStageId(finding);
|
|
91
|
+
if (stageId != null) return { key: `stage:${stageId}`, stageId };
|
|
92
|
+
if ('stageIds' in finding && finding.stageIds.length > 1) {
|
|
93
|
+
return { key: `stages:${finding.type}:${[...finding.stageIds].sort((a, b) => a - b).join(',')}`, stageId: null };
|
|
94
|
+
}
|
|
95
|
+
const variant = 'variant' in finding ? finding.variant : undefined;
|
|
96
|
+
return { key: variant ? `app:${finding.type}:${variant}` : `app:${finding.type}`, stageId: null };
|
|
97
|
+
}
|
|
98
|
+
|
|
99
|
+
/** 0 for a failure at a stage of a failed job (what stopped the job), 1 for
|
|
100
|
+
* any other failure finding, 2 for everything else. */
|
|
101
|
+
function failureRank(finding , failedJobStageIds ) {
|
|
102
|
+
if (!FAILURE_TYPES.has(finding.type)) return 2;
|
|
103
|
+
return typeof finding.stageId === 'number' && failedJobStageIds.has(finding.stageId) ? 0 : 1;
|
|
104
|
+
}
|
|
105
|
+
|
|
106
|
+
/** Groups every rankable finding by location, ordered by the location's
|
|
107
|
+
* best-ranked finding, so step 1 is always the run's single biggest win. With
|
|
108
|
+
* `failedJobStageIds` (a run whose jobs failed), failure findings rank ahead
|
|
109
|
+
* of every savings figure and lead their location, those at a failed job's
|
|
110
|
+
* stage first: a speed-up is moot until the job finishes. */
|
|
111
|
+
export function buildNextSteps(findings , { failedJobStageIds } = {}) {
|
|
112
|
+
const ranked = rankBySavings(findings);
|
|
113
|
+
// Array.prototype.sort is stable, so savings order holds within each rank.
|
|
114
|
+
const ordered = failedJobStageIds
|
|
115
|
+
? [...ranked].sort((a, b) => failureRank(a, failedJobStageIds) - failureRank(b, failedJobStageIds))
|
|
116
|
+
: ranked;
|
|
117
|
+
const steps = new Map ();
|
|
118
|
+
for (const finding of ordered) {
|
|
119
|
+
const { key, stageId } = locationKey(finding);
|
|
120
|
+
const existing = steps.get(key);
|
|
121
|
+
if (!existing) {
|
|
122
|
+
steps.set(key, { key, lead: finding, related: [], stageId });
|
|
123
|
+
continue;
|
|
124
|
+
}
|
|
125
|
+
const seenTypes = new Set([existing.lead.type, ...existing.related.map((f) => f.type)]);
|
|
126
|
+
if (!seenTypes.has(finding.type)) existing.related.push(finding);
|
|
127
|
+
}
|
|
128
|
+
return [...steps.values()];
|
|
129
|
+
}
|
|
130
|
+
|
|
131
|
+
/** True for findings about executor capacity sitting idle. memoryUtilization
|
|
132
|
+
* also reports heap pressure and over-provisioning, which are not idle
|
|
133
|
+
* capacity, so only its idleCores variant counts. */
|
|
134
|
+
function isIdleCapacityFinding(finding ) {
|
|
135
|
+
return finding.type === 'utilization' || (finding.type === 'memoryUtilization' && finding.variant === 'idleCores');
|
|
136
|
+
}
|
|
137
|
+
|
|
138
|
+
/** Idle share at which the verdict notes, in its summary, that the cluster
|
|
139
|
+
* may be larger than the job needs. It never reorders the steps. */
|
|
140
|
+
export const IDLE_NOTABLE_PCT = 40;
|
|
141
|
+
|
|
142
|
+
export function isIdleCapacityStep(step ) {
|
|
143
|
+
return isIdleCapacityFinding(step.lead);
|
|
144
|
+
}
|
|
145
|
+
|
|
146
|
+
/** The idle share an idle-capacity finding itself reports: idleCores carries
|
|
147
|
+
* the idle rate, utilization the busy rate. Null for any other finding. */
|
|
148
|
+
function reportedIdlePct(finding ) {
|
|
149
|
+
if (typeof finding.value !== 'number') return null;
|
|
150
|
+
if (finding.type === 'memoryUtilization' && finding.variant === 'idleCores') return finding.value;
|
|
151
|
+
if (finding.type === 'utilization') return 100 - finding.value;
|
|
152
|
+
return null;
|
|
153
|
+
}
|
|
154
|
+
|
|
155
|
+
/** The run's idle share as the verdict states it: the figure the top-ranked
|
|
156
|
+
* idle-capacity step reports, so the verdict never disagrees with that step,
|
|
157
|
+
* or `fallbackPct` (the Scorecard's Unused core time) when no step reports one. */
|
|
158
|
+
export function verdictIdlePct(steps , fallbackPct ) {
|
|
159
|
+
const idleStep = steps.find(isIdleCapacityStep);
|
|
160
|
+
return (idleStep ? reportedIdlePct(idleStep.lead) : null) ?? fallbackPct;
|
|
161
|
+
}
|
|
162
|
+
|
|
163
|
+
function plural(count , noun ) {
|
|
164
|
+
return `${count} ${noun}${count === 1 ? '' : 's'}`;
|
|
165
|
+
}
|
|
166
|
+
|
|
167
|
+
/** The run facts the verdict's wording depends on, read once. */
|
|
168
|
+
|
|
169
|
+
|
|
170
|
+
|
|
171
|
+
|
|
172
|
+
|
|
173
|
+
|
|
174
|
+
|
|
175
|
+
|
|
176
|
+
|
|
177
|
+
|
|
178
|
+
|
|
179
|
+
|
|
180
|
+
|
|
181
|
+
|
|
182
|
+
|
|
183
|
+
|
|
184
|
+
function isFailedRun(facts ) {
|
|
185
|
+
return facts.outcome.failedJobs > 0;
|
|
186
|
+
}
|
|
187
|
+
|
|
188
|
+
/** The failed-run title: the one thing a newcomer must know before any
|
|
189
|
+
* tuning advice is that the job did not finish. */
|
|
190
|
+
function failedTitle({ failedJobs, totalJobs } ) {
|
|
191
|
+
if (failedJobs < totalJobs) return `${failedJobs} of ${totalJobs} jobs failed in this run`;
|
|
192
|
+
return totalJobs === 1 ? 'This run failed: its job did not finish' : `This run failed: all ${totalJobs} jobs did not finish`;
|
|
193
|
+
}
|
|
194
|
+
|
|
195
|
+
/** An action label as it reads after "Start here:": only its first letter
|
|
196
|
+
* drops to lower case, so a name inside it ("Switch to Kryo") keeps its
|
|
197
|
+
* capital, and a leading acronym ("GC", "OOM") is left alone. */
|
|
198
|
+
function lowerFirst(label ) {
|
|
199
|
+
if (/^[A-Z]{2}/.test(label)) return label;
|
|
200
|
+
return label.charAt(0).toLowerCase() + label.slice(1);
|
|
201
|
+
}
|
|
202
|
+
|
|
203
|
+
export function verdictTitle(eligible , steps , facts ) {
|
|
204
|
+
if (isFailedRun(facts)) return failedTitle(facts.outcome);
|
|
205
|
+
if (eligible.length === 0 && facts.incomplete) return 'This log looks incomplete, so results cover only part of the run';
|
|
206
|
+
if (eligible.length === 0 && facts.noFinishedStages) return 'This log has no finished stages to check';
|
|
207
|
+
if (eligible.length === 0 && !facts.clean) return 'Nothing to fix, but some checks could not run on this log';
|
|
208
|
+
if (eligible.length === 0) return 'No findings to fix right now.';
|
|
209
|
+
// Every real detector writes a recommendation, so an eligible finding with
|
|
210
|
+
// no route is a defensive case: still never call such a run clean.
|
|
211
|
+
if (steps.length === 0) return `${plural(eligible.length, 'finding')} to review`;
|
|
212
|
+
const lead = steps[0];
|
|
213
|
+
if (isIdleCapacityStep(lead) && facts.idlePct != null) return `Start with cluster size: ${facts.idlePct}% of executor capacity sat idle`;
|
|
214
|
+
if (lead.stageId != null) return `Start with Stage ${lead.stageId}`;
|
|
215
|
+
return `Start here: ${lowerFirst(findingActionLabel(lead.lead))}`;
|
|
216
|
+
}
|
|
217
|
+
|
|
218
|
+
/** The run-level summary under the title: how much was found and where, what
|
|
219
|
+
* the first fix is worth, and the run's idle capacity when that is large
|
|
220
|
+
* enough to matter but is not the first step (whose title already says it). */
|
|
221
|
+
export function verdictSummary(eligible , steps , facts ) {
|
|
222
|
+
const sentences = [];
|
|
223
|
+
const { failedJobs, totalJobs } = facts.outcome;
|
|
224
|
+
if (failedJobs > 0) {
|
|
225
|
+
if (eligible.some((finding) => !FAILURE_TYPES.has(finding.type))) {
|
|
226
|
+
sentences.push('Fix the failure before tuning: the other findings cover only the work that ran.');
|
|
227
|
+
}
|
|
228
|
+
} else if (totalJobs > 0 && !facts.incomplete) {
|
|
229
|
+
sentences.push(totalJobs === 1 ? 'Its one job succeeded.' : `All ${totalJobs} jobs succeeded.`);
|
|
230
|
+
}
|
|
231
|
+
if (eligible.length === 0) {
|
|
232
|
+
if (failedJobs > 0) return sentences;
|
|
233
|
+
if (facts.clean) sentences.push('Every check passed for this run.');
|
|
234
|
+
} else if (steps.length === 0) {
|
|
235
|
+
sentences.push('They are listed by impact under Findings.');
|
|
236
|
+
} else {
|
|
237
|
+
sentences.push(`${plural(eligible.length, 'finding')} in ${plural(steps.length, 'place')}.`);
|
|
238
|
+
const wallClock = steps[0].lead.impactEstimate?.wallClock;
|
|
239
|
+
if (isIdleCapacityStep(steps[0])) {
|
|
240
|
+
sentences.push('A smaller cluster or dynamic allocation would free the idle cores for other jobs.');
|
|
241
|
+
} else if (wallClock && facts.runMs != null) {
|
|
242
|
+
sentences.push(`The first fix could save up to ${formatDuration(wallClock.high)} of this ${formatDuration(facts.runMs)} run.`);
|
|
243
|
+
}
|
|
244
|
+
if (steps.some((step) => step.related.length > 0)) {
|
|
245
|
+
sentences.push('Findings in the same stage usually share one cause, so they are grouped together and their savings overlap rather than add up.');
|
|
246
|
+
}
|
|
247
|
+
}
|
|
248
|
+
if (facts.incomplete) {
|
|
249
|
+
sentences.push('The log has no end-of-run record, so these figures cover only the part of the run it captured.');
|
|
250
|
+
}
|
|
251
|
+
const leadIsIdle = steps.length > 0 && isIdleCapacityStep(steps[0]);
|
|
252
|
+
if (!leadIsIdle && facts.idlePct != null && facts.idlePct >= IDLE_NOTABLE_PCT) {
|
|
253
|
+
sentences.push(
|
|
254
|
+
steps.some(isIdleCapacityStep)
|
|
255
|
+
? `${facts.idlePct}% of the executor capacity sat idle, so the cluster may be larger than this job needs.`
|
|
256
|
+
: `${facts.idlePct}% of the run's core time went unused, so the cluster may be larger than this job needs.`,
|
|
257
|
+
);
|
|
258
|
+
}
|
|
259
|
+
return sentences;
|
|
260
|
+
}
|
|
261
|
+
|
|
262
|
+
const STACK_TRACE_HINT = 'Open the driver log only if you need the full stack trace.';
|
|
263
|
+
|
|
264
|
+
/** What the failure step whose reason the verdict quotes tells the reader:
|
|
265
|
+
* the detector's "inspect the driver log for the reason" would send a
|
|
266
|
+
* newcomer looking for something already on screen. The copied text carries
|
|
267
|
+
* the reason itself, since "quoted above" means nothing once pasted. */
|
|
268
|
+
export function quotedReasonText(reason ) {
|
|
269
|
+
const sentence = /[.!?]$/.test(reason) ? reason : `${reason}.`;
|
|
270
|
+
return {
|
|
271
|
+
shown: `Spark's recorded reason is quoted above. ${STACK_TRACE_HINT}`,
|
|
272
|
+
copied: `Spark's recorded reason: ${sentence} ${STACK_TRACE_HINT}`,
|
|
273
|
+
};
|
|
274
|
+
}
|
|
275
|
+
|
|
276
|
+
/** One step as pasteable text: the action, what to try, and the savings. */
|
|
277
|
+
export function stepCopyText(finding , recommendation , stageId = null) {
|
|
278
|
+
const impact = impactFigure(finding);
|
|
279
|
+
const meaning = savingsMeaning(finding);
|
|
280
|
+
const savings = impact && meaning ? `${impact} ${meaning}` : impact;
|
|
281
|
+
const where = stageId != null ? ` in Stage ${stageId}` : '';
|
|
282
|
+
const headline = `${findingActionLabel(finding)}${where}: ${recommendation}`;
|
|
283
|
+
return [/[.!?]$/.test(headline) ? headline : `${headline}.`, savings ? `Potential savings: ${savings}` : null]
|
|
284
|
+
.filter(Boolean)
|
|
285
|
+
.join(' ');
|
|
286
|
+
}
|
|
287
|
+
|
|
288
|
+
/** The whole verdict as a pasteable checklist for a ticket or a message:
|
|
289
|
+
* run, verdict, numbered steps (with their stage), and how many more places
|
|
290
|
+
* the full list holds. */
|
|
291
|
+
export function planCopyText(input
|
|
292
|
+
|
|
293
|
+
|
|
294
|
+
|
|
295
|
+
|
|
296
|
+
) {
|
|
297
|
+
const lines = [input.runName ? `Spark run ${input.runName}: ${input.title}` : input.title, ''];
|
|
298
|
+
input.steps.forEach(({ step, recommendation }, index) => {
|
|
299
|
+
lines.push(`${index + 1}. ${stepCopyText(step.lead, recommendation, step.stageId)}`);
|
|
300
|
+
});
|
|
301
|
+
if (input.remaining > 0) lines.push('', `${plural(input.remaining, 'more place')} to look at in the full findings list.`);
|
|
302
|
+
return lines.join('\n');
|
|
303
|
+
}
|
|
304
|
+
|
|
305
|
+
/** A step's "What to try" text for copying: Spark's quoted reason when it is this step's own
|
|
306
|
+
* failure, else the finding's recommendation. */
|
|
307
|
+
export function stepCopyRecommendation(step , outcome ) {
|
|
308
|
+
return quotesReasonOf(step.lead, outcome) && outcome.reason
|
|
309
|
+
? quotedReasonText(outcome.reason).copied
|
|
310
|
+
: recommendationText(step.lead);
|
|
311
|
+
}
|
|
312
|
+
|
|
313
|
+
/** Everything the verdict card shows, computed once for a run. `eligible` is what the verdict
|
|
314
|
+
* counts and ranks: the rollup-eligible findings of a known display type. */
|
|
315
|
+
|
|
316
|
+
|
|
317
|
+
|
|
318
|
+
|
|
319
|
+
|
|
320
|
+
|
|
321
|
+
|
|
322
|
+
|
|
323
|
+
|
|
324
|
+
|
|
325
|
+
|
|
326
|
+
|
|
327
|
+
|
|
328
|
+
|
|
329
|
+
export function buildRunVerdict(appModel , allFindings ) {
|
|
330
|
+
const eligible = allFindings.filter((finding) => isEligible(finding) && DISPLAY_INDEX.has(finding.type));
|
|
331
|
+
const outcome = summarizeRunOutcome(appModel.jobs, allFindings);
|
|
332
|
+
const steps = buildNextSteps(eligible, outcome.failedJobs > 0 ? { failedJobStageIds: outcome.failedJobStageIds } : {});
|
|
333
|
+
const facts = {
|
|
334
|
+
runMs: hasCompleteApplicationInterval(appModel.app) ? computeWallClock(appModel.app, appModel.stages).total : null,
|
|
335
|
+
idlePct: verdictIdlePct(steps, getScorecardEstimates(appModel).wastage.value),
|
|
336
|
+
incomplete: allFindings.some((finding) => finding.type === 'incompleteRun'),
|
|
337
|
+
clean: isCleanRun(appModel, allFindings),
|
|
338
|
+
outcome,
|
|
339
|
+
noFinishedStages: !hasFinishedStage(appModel.stages),
|
|
340
|
+
};
|
|
341
|
+
const title = verdictTitle(eligible, steps, facts);
|
|
342
|
+
const shown = steps.slice(0, NEXT_STEP_LIMIT);
|
|
343
|
+
const remaining = steps.length - shown.length;
|
|
344
|
+
const copyText = shown.length > 0
|
|
345
|
+
? planCopyText({
|
|
346
|
+
runName: appModel.app?.name ?? null,
|
|
347
|
+
title,
|
|
348
|
+
steps: shown.map((step) => ({ step, recommendation: stepCopyRecommendation(step, outcome) })),
|
|
349
|
+
remaining,
|
|
350
|
+
})
|
|
351
|
+
: null;
|
|
352
|
+
return { eligible, outcome, steps, facts, title, summary: verdictSummary(eligible, steps, facts), shown, remaining, copyText };
|
|
353
|
+
}
|
|
@@ -6,7 +6,6 @@
|
|
|
6
6
|
// Builds on computeCoreTimeSeries' clamp conceptually, but operates on the
|
|
7
7
|
// worker-posted per-stage aggregates (runAggregates.perStage) so raw task data
|
|
8
8
|
// never reaches the main thread.
|
|
9
|
-
import { computeWallClock } from './wall-clock.js';
|
|
10
9
|
import { computeTotalCores } from './core-count.js';
|
|
11
10
|
|
|
12
11
|
|
|
@@ -30,9 +29,11 @@ function estimatedTotalAtCores(perStage
|
|
|
30
29
|
return sum;
|
|
31
30
|
}
|
|
32
31
|
|
|
33
|
-
|
|
32
|
+
/** `observedActiveMs` is the run's stages-active wall-clock (the interpretation's
|
|
33
|
+
* `wallClock.stagesActive`), the one observed makespan predictions are scaled against. */
|
|
34
|
+
export function simulateScaling({ app, observedActiveMs, runAggregates, executorsAdded }
|
|
35
|
+
|
|
34
36
|
|
|
35
|
-
|
|
36
37
|
|
|
37
38
|
|
|
38
39
|
)
|
|
@@ -48,8 +49,6 @@ export function simulateScaling({ app, stages, runAggregates, executorsAdded }
|
|
|
48
49
|
// `executorsAdded` cast: computeTotalCores only reads `totalCores`, present on
|
|
49
50
|
// ExecutorAddedEvent (the only kind passed here) but not on the union type.
|
|
50
51
|
const baselineCores = computeTotalCores(app ?? {}, executorsAdded );
|
|
51
|
-
const wc = computeWallClock(app, stages );
|
|
52
|
-
const observedActiveMs = wc.stagesActive;
|
|
53
52
|
|
|
54
53
|
// Model Error: predicted-at-baseline vs. observed stages-active wall-clock.
|
|
55
54
|
const predictedAtBaseline = baselineCores > 0 ? estimatedTotalAtCores(perStage, baselineCores) : 0;
|
|
@@ -0,0 +1,62 @@
|
|
|
1
|
+
import { computeEfficiencyModel } from './efficiency-model.js';
|
|
2
|
+
import { computeWallClock } from './wall-clock.js';
|
|
3
|
+
|
|
4
|
+
|
|
5
|
+
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
export function hasCompleteApplicationInterval(app ) {
|
|
16
|
+
return typeof app?.startTime === 'number'
|
|
17
|
+
&& Number.isFinite(app.startTime)
|
|
18
|
+
&& typeof app?.endTime === 'number'
|
|
19
|
+
&& Number.isFinite(app.endTime)
|
|
20
|
+
&& app.endTime > app.startTime;
|
|
21
|
+
}
|
|
22
|
+
|
|
23
|
+
export function getScorecardEstimates(appModel )
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
{
|
|
27
|
+
if (!hasCompleteApplicationInterval(appModel.app)) {
|
|
28
|
+
return {
|
|
29
|
+
efficiency: { value: null, unavailableReason: 'application-timing' },
|
|
30
|
+
wastage: { value: null, unavailableReason: 'application-timing' },
|
|
31
|
+
};
|
|
32
|
+
}
|
|
33
|
+
|
|
34
|
+
const wallClock = computeWallClock(appModel.app, appModel.stages);
|
|
35
|
+
const efficiency = Math.min(100, Math.round((wallClock.stagesActive / wallClock.total) * 100));
|
|
36
|
+
|
|
37
|
+
if (!appModel.runAggregates) {
|
|
38
|
+
return {
|
|
39
|
+
efficiency: { value: efficiency, unavailableReason: null },
|
|
40
|
+
wastage: { value: null, unavailableReason: 'core-usage-summary' },
|
|
41
|
+
};
|
|
42
|
+
}
|
|
43
|
+
|
|
44
|
+
const efficiencyModel = computeEfficiencyModel({
|
|
45
|
+
app: appModel.app,
|
|
46
|
+
stages: appModel.stages,
|
|
47
|
+
executorsAdded: appModel.executors.added,
|
|
48
|
+
runAggregates: appModel.runAggregates,
|
|
49
|
+
});
|
|
50
|
+
|
|
51
|
+
if (!Number.isFinite(efficiencyModel.availableComputeHours) || efficiencyModel.availableComputeHours <= 0) {
|
|
52
|
+
return {
|
|
53
|
+
efficiency: { value: efficiency, unavailableReason: null },
|
|
54
|
+
wastage: { value: null, unavailableReason: 'executor-capacity' },
|
|
55
|
+
};
|
|
56
|
+
}
|
|
57
|
+
|
|
58
|
+
return {
|
|
59
|
+
efficiency: { value: efficiency, unavailableReason: null },
|
|
60
|
+
wastage: { value: efficiencyModel.wastagePct, unavailableReason: null },
|
|
61
|
+
};
|
|
62
|
+
}
|