sparkforensics-mcp 0.3.0 → 0.5.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (92) hide show
  1. package/bin/sparkforensics-mcp.mjs +41 -10
  2. package/package.json +1 -1
  3. package/vendor-core/allocation.js +106 -0
  4. package/vendor-core/analyzer.js +168 -60
  5. package/vendor-core/check-coverage.js +88 -0
  6. package/vendor-core/cli/budgets.js +54 -27
  7. package/vendor-core/cli/collect-run.js +84 -32
  8. package/vendor-core/cli/regression-budgets.js +83 -0
  9. package/vendor-core/cli/threshold-config.js +28 -0
  10. package/vendor-core/comparison-verdict.js +177 -0
  11. package/vendor-core/core-source-hash.txt +1 -0
  12. package/vendor-core/core-usage-locality.js +56 -2
  13. package/vendor-core/detector-docs.js +58 -0
  14. package/vendor-core/detectors.js +1094 -500
  15. package/vendor-core/docs-config.js +0 -36
  16. package/vendor-core/docs-content/chapters/nav-index.json +31 -0
  17. package/vendor-core/docs-content/detection/cache.md +3 -2
  18. package/vendor-core/docs-content/detection/cfg.md +9 -8
  19. package/vendor-core/docs-content/detection/chrn.md +1 -2
  20. package/vendor-core/docs-content/detection/cold.md +4 -2
  21. package/vendor-core/docs-content/detection/cstor.md +9 -0
  22. package/vendor-core/docs-content/detection/fail.md +3 -2
  23. package/vendor-core/docs-content/detection/gc.md +3 -2
  24. package/vendor-core/docs-content/detection/host.md +2 -1
  25. package/vendor-core/docs-content/detection/local.md +1 -1
  26. package/vendor-core/docs-content/detection/mem.md +5 -2
  27. package/vendor-core/docs-content/detection/plan.md +2 -1
  28. package/vendor-core/docs-content/detection/sfail.md +2 -1
  29. package/vendor-core/docs-content/detection/shape.md +5 -4
  30. package/vendor-core/docs-content/detection/skew.md +3 -1
  31. package/vendor-core/docs-content/detection/slow.md +2 -2
  32. package/vendor-core/docs-content/detection/spec.md +2 -3
  33. package/vendor-core/docs-content/detection/spill.md +1 -1
  34. package/vendor-core/docs-site-config.js +3 -0
  35. package/vendor-core/effective-conf.js +107 -0
  36. package/vendor-core/efficiency-model.js +8 -6
  37. package/vendor-core/event-handlers.js +321 -44
  38. package/vendor-core/event-schemas.js +23 -0
  39. package/vendor-core/evidence-report.js +432 -115
  40. package/vendor-core/export-data.js +79 -6
  41. package/vendor-core/finding-action-label.js +9 -88
  42. package/vendor-core/finding-filter-predicate.js +9 -0
  43. package/vendor-core/finding-generic-recommendation.js +26 -105
  44. package/vendor-core/finding-names.js +28 -45
  45. package/vendor-core/finding-presentation.js +368 -0
  46. package/vendor-core/finding-tag-help.js +110 -0
  47. package/vendor-core/finding-types.js +373 -0
  48. package/vendor-core/findings-of-type.js +11 -0
  49. package/vendor-core/format-utils.js +96 -30
  50. package/vendor-core/html-export.js +51 -0
  51. package/vendor-core/impact-band.js +21 -8
  52. package/vendor-core/impact-estimator.js +25 -520
  53. package/vendor-core/impact-format.js +115 -0
  54. package/vendor-core/impact-model.js +197 -0
  55. package/vendor-core/ingest.js +6 -2
  56. package/vendor-core/intervals.js +13 -0
  57. package/vendor-core/list-runs.js +7 -5
  58. package/vendor-core/load-vendored.js +70 -5
  59. package/vendor-core/mcp-server-factory.js +14 -10
  60. package/vendor-core/mcp-tools.js +105 -45
  61. package/vendor-core/model-assembler.js +35 -1
  62. package/vendor-core/occupancy.js +1 -1
  63. package/vendor-core/parser-worker.js +2 -2
  64. package/vendor-core/plan-graph-model.js +3 -2
  65. package/vendor-core/plan-node-detail.js +1 -1
  66. package/vendor-core/proxy.js +3 -1
  67. package/vendor-core/python-stage.js +25 -0
  68. package/vendor-core/recommendation-rollup.js +70 -3
  69. package/vendor-core/redact.js +96 -37
  70. package/vendor-core/remediation.js +20 -0
  71. package/vendor-core/run-comparison.js +73 -29
  72. package/vendor-core/run-interpretation.js +291 -0
  73. package/vendor-core/run-metrics.js +198 -0
  74. package/vendor-core/run-outcome.js +74 -0
  75. package/vendor-core/run-payload.js +17 -0
  76. package/vendor-core/run-shape.js +40 -0
  77. package/vendor-core/run-totals.js +24 -0
  78. package/vendor-core/run-verdict.js +352 -0
  79. package/vendor-core/scaling-sim.js +4 -5
  80. package/vendor-core/scorecard-estimates.js +63 -0
  81. package/vendor-core/session-snapshot.js +7 -0
  82. package/vendor-core/shs-schemas.js +2 -2
  83. package/vendor-core/spark-memory.js +17 -0
  84. package/vendor-core/sql-stages.js +11 -0
  85. package/vendor-core/stage-plan-nodes.js +18 -0
  86. package/vendor-core/stage-quantiles.js +6 -0
  87. package/vendor-core/threshold-overrides.js +160 -0
  88. package/vendor-core/threshold-summary.js +11 -33
  89. package/vendor-core/types.js +54 -42
  90. package/vendor-core/wall-clock.js +1 -12
  91. package/vendor-core/wasted-core-hours.js +12 -9
  92. package/vendor-core/write-targets.js +312 -0
@@ -0,0 +1,352 @@
1
+ // The run verdict: where to start, a short summary, and the top places to look, ranked by
2
+ // potential savings (failures first on a failed run). Shared by the dashboard's verdict card and
3
+ // the CLI/MCP evidence report, so both paths name the same first step in the same words.
4
+ import { ENTRY_BY_TYPE } from './detectors.js';
5
+ import { hasFinishedStage, isCleanRun } from './check-coverage.js';
6
+ import { findingActionLabel } from './finding-action-label.js';
7
+ import { singleStageId } from './finding-filter-predicate.js';
8
+ import { recommendationText } from './finding-names.js';
9
+ import { formatDuration, IMPACT_BAND_ORDER } from './format-utils.js';
10
+ import { impactFigure, savingsMeaning } from './impact-format.js';
11
+ import { isEligible } from './recommendation-rollup.js';
12
+ import { FAILURE_TYPES, quotesReasonOf, summarizeRunOutcome, } from './run-outcome.js';
13
+ import { getScorecardEstimates, hasCompleteApplicationInterval } from './scorecard-estimates.js';
14
+ import { computeWallClock } from './wall-clock.js';
15
+
16
+
17
+ /** How many next steps the verdict lists before pointing at the full list. */
18
+ export const NEXT_STEP_LIMIT = 3;
19
+
20
+ /** Finding types whose widget sits in the board's reference region (listed after every action
21
+ * widget). The dashboard's registry must agree: tests/view/detector-registry.test.tsx checks it. */
22
+ export const REFERENCE_DISPLAY_TYPES = new Set([
23
+ 'memoryUtilization', 'utilization', 'coreLocality', 'cacheUtilization',
24
+ ]);
25
+
26
+ /** Every emitted finding type in the board's widget display order: action region before
27
+ * reference region, then ascending order of the type's `ENTRY_BY_TYPE` entry.
28
+ * The last tiebreak of the verdict ranking, so the CLI orders ties exactly as the dashboard. */
29
+ export const FINDING_DISPLAY_ORDER = (() => {
30
+ const region = (type ) => (REFERENCE_DISPLAY_TYPES.has(type) ? 1 : 0);
31
+ return [...ENTRY_BY_TYPE.entries()]
32
+ .sort(([a, entryA], [b, entryB]) => region(a) - region(b) || entryA.order - entryB.order)
33
+ .map(([type]) => type);
34
+ })();
35
+
36
+ const DISPLAY_INDEX = new Map(FINDING_DISPLAY_ORDER.map((type, index) => [type, index]));
37
+
38
+ // The high end of the finding's own occupancy-clipped wall-clock estimate, the figure the
39
+ // "Potential savings" line leads with. `null` with no quantified time claim
40
+ // (resourceOnly/informational basis): such a finding can never win on its own numbers.
41
+ function potentialSavingsMs(finding ) {
42
+ return finding.impactEstimate?.wallClock?.high ?? null;
43
+ }
44
+
45
+ /** True for a finding the verdict can route to: a known display type with a recommendation. */
46
+ export function isRankable(finding ) {
47
+ const recommendation = typeof finding.recommendation === 'string' ? finding.recommendation.trim() : '';
48
+ return DISPLAY_INDEX.has(finding.type) && recommendation.length > 0;
49
+ }
50
+
51
+ /** Every rankable finding, best first. Ranked by potential savings: a quantified estimate always
52
+ * outranks an unquantified one; ties (including "neither has one") fall back to impact band,
53
+ * then widget display order, then input order, so an unquantified warning still leads an info. */
54
+ export function rankBySavings(findings ) {
55
+ const candidates = findings
56
+ .map((finding, index) => ({ finding, index }))
57
+ .filter(({ finding }) => isRankable(finding));
58
+ candidates.sort((left, right) => {
59
+ const leftSavings = potentialSavingsMs(left.finding);
60
+ const rightSavings = potentialSavingsMs(right.finding);
61
+ if (leftSavings !== null && rightSavings !== null && leftSavings !== rightSavings) return rightSavings - leftSavings;
62
+ if ((leftSavings !== null) !== (rightSavings !== null)) return leftSavings !== null ? -1 : 1;
63
+ return (
64
+ (IMPACT_BAND_ORDER[left.finding.impactBand] ?? 9) - (IMPACT_BAND_ORDER[right.finding.impactBand] ?? 9)
65
+ || (DISPLAY_INDEX.get(left.finding.type) ?? Number.MAX_SAFE_INTEGER) - (DISPLAY_INDEX.get(right.finding.type) ?? Number.MAX_SAFE_INTEGER)
66
+ || left.index - right.index
67
+ );
68
+ });
69
+ return candidates.map(({ finding }) => finding);
70
+ }
71
+
72
+ /** One place worth a look: the highest-ranked finding at a location, plus the
73
+ * other finding types flagged at that same location. Findings that share a
74
+ * stage usually share one root cause, so they read as one step, not several. */
75
+
76
+
77
+
78
+
79
+
80
+
81
+
82
+
83
+
84
+ /** A finding's location identity for grouping. Per-stage findings (and
85
+ * sql-scope findings that touch exactly one stage) group by that stage; a
86
+ * multi-stage or app-level finding stands alone by its own type (and variant,
87
+ * for app-level ones), since two different app-level problems are not the
88
+ * same place. */
89
+ export function locationKey(finding ) {
90
+ const stageId = singleStageId(finding);
91
+ if (stageId != null) return { key: `stage:${stageId}`, stageId };
92
+ if ('stageIds' in finding && finding.stageIds.length > 1) {
93
+ return { key: `stages:${finding.type}:${[...finding.stageIds].sort((a, b) => a - b).join(',')}`, stageId: null };
94
+ }
95
+ const variant = 'variant' in finding ? finding.variant : undefined;
96
+ return { key: variant ? `app:${finding.type}:${variant}` : `app:${finding.type}`, stageId: null };
97
+ }
98
+
99
+ /** 0 for a failure at a stage of a failed job (what stopped the job), 1 for
100
+ * any other failure finding, 2 for everything else. */
101
+ function failureRank(finding , failedJobStageIds ) {
102
+ if (!FAILURE_TYPES.has(finding.type)) return 2;
103
+ return typeof finding.stageId === 'number' && failedJobStageIds.has(finding.stageId) ? 0 : 1;
104
+ }
105
+
106
+ /** Groups every rankable finding by location, ordered by the location's
107
+ * best-ranked finding, so step 1 is always the run's single biggest win. With
108
+ * `failedJobStageIds` (a run whose jobs failed), failure findings rank ahead
109
+ * of every savings figure and lead their location, those at a failed job's
110
+ * stage first: a speed-up is moot until the job finishes. */
111
+ export function buildNextSteps(findings , { failedJobStageIds } = {}) {
112
+ const ranked = rankBySavings(findings);
113
+ // Array.prototype.sort is stable, so savings order holds within each rank.
114
+ const ordered = failedJobStageIds
115
+ ? [...ranked].sort((a, b) => failureRank(a, failedJobStageIds) - failureRank(b, failedJobStageIds))
116
+ : ranked;
117
+ const steps = new Map ();
118
+ for (const finding of ordered) {
119
+ const { key, stageId } = locationKey(finding);
120
+ const existing = steps.get(key);
121
+ if (!existing) {
122
+ steps.set(key, { key, lead: finding, related: [], stageId });
123
+ continue;
124
+ }
125
+ const seenTypes = new Set([existing.lead.type, ...existing.related.map((f) => f.type)]);
126
+ if (!seenTypes.has(finding.type)) existing.related.push(finding);
127
+ }
128
+ return [...steps.values()];
129
+ }
130
+
131
+ /** True for findings about executor capacity sitting idle. memoryUtilization
132
+ * also reports heap pressure and over-provisioning, which are not idle
133
+ * capacity, so only its idleCores variant counts. */
134
+ function isIdleCapacityFinding(finding ) {
135
+ return finding.type === 'utilization' || (finding.type === 'memoryUtilization' && finding.variant === 'idleCores');
136
+ }
137
+
138
+ /** Idle share at which the verdict notes, in its summary, that the cluster
139
+ * may be larger than the job needs. It never reorders the steps. */
140
+ export const IDLE_NOTABLE_PCT = 40;
141
+
142
+ export function isIdleCapacityStep(step ) {
143
+ return isIdleCapacityFinding(step.lead);
144
+ }
145
+
146
+ /** The idle share an idle-capacity finding itself reports: idleCores carries
147
+ * the idle rate, utilization the busy rate. Null for any other finding. */
148
+ function reportedIdlePct(finding ) {
149
+ if (typeof finding.value !== 'number') return null;
150
+ if (finding.type === 'memoryUtilization' && finding.variant === 'idleCores') return finding.value;
151
+ if (finding.type === 'utilization') return 100 - finding.value;
152
+ return null;
153
+ }
154
+
155
+ /** The run's idle share as the verdict states it: the figure the top-ranked
156
+ * idle-capacity step reports, so the verdict never disagrees with that step,
157
+ * or `fallbackPct` (the Scorecard's Unused core time) when no step reports one. */
158
+ export function verdictIdlePct(steps , fallbackPct ) {
159
+ const idleStep = steps.find(isIdleCapacityStep);
160
+ return (idleStep ? reportedIdlePct(idleStep.lead) : null) ?? fallbackPct;
161
+ }
162
+
163
+ function plural(count , noun ) {
164
+ return `${count} ${noun}${count === 1 ? '' : 's'}`;
165
+ }
166
+
167
+ /** The run facts the verdict's wording depends on, read once. */
168
+
169
+
170
+
171
+
172
+
173
+
174
+
175
+
176
+
177
+
178
+
179
+
180
+
181
+
182
+
183
+
184
+ function isFailedRun(facts ) {
185
+ return facts.outcome.failedJobs > 0;
186
+ }
187
+
188
+ /** The failed-run title: the one thing a newcomer must know before any
189
+ * tuning advice is that the job did not finish. */
190
+ function failedTitle({ failedJobs, totalJobs } ) {
191
+ if (failedJobs < totalJobs) return `${failedJobs} of ${totalJobs} jobs failed in this run`;
192
+ return totalJobs === 1 ? 'This run failed: its job did not finish' : `This run failed: all ${totalJobs} jobs did not finish`;
193
+ }
194
+
195
+ /** An action label as it reads after "Start here:": only its first letter
196
+ * drops to lower case, so a name inside it ("Switch to Kryo") keeps its
197
+ * capital, and a leading acronym ("GC", "OOM") is left alone. */
198
+ function lowerFirst(label ) {
199
+ if (/^[A-Z]{2}/.test(label)) return label;
200
+ return label.charAt(0).toLowerCase() + label.slice(1);
201
+ }
202
+
203
+ export function verdictTitle(eligible , steps , facts ) {
204
+ if (isFailedRun(facts)) return failedTitle(facts.outcome);
205
+ if (eligible.length === 0 && facts.incomplete) return 'This log looks incomplete, so results cover only part of the run';
206
+ if (eligible.length === 0 && facts.noFinishedStages) return 'This log has no finished stages to check';
207
+ if (eligible.length === 0 && !facts.clean) return 'Nothing to fix, but some checks could not run on this log';
208
+ if (eligible.length === 0) return 'No findings to fix right now.';
209
+ // Every real detector writes a recommendation, so an eligible finding with
210
+ // no route is a defensive case: still never call such a run clean.
211
+ if (steps.length === 0) return `${plural(eligible.length, 'finding')} to review`;
212
+ const lead = steps[0];
213
+ if (isIdleCapacityStep(lead) && facts.idlePct != null) return `Start with cluster size: ${facts.idlePct}% of executor capacity sat idle`;
214
+ if (lead.stageId != null) return `Start with Stage ${lead.stageId}`;
215
+ return `Start here: ${lowerFirst(findingActionLabel(lead.lead))}`;
216
+ }
217
+
218
+ /** The run-level summary under the title: how much was found and where, what
219
+ * the first fix is worth, and the run's idle capacity when that is large
220
+ * enough to matter but is not the first step (whose title already says it). */
221
+ export function verdictSummary(eligible , steps , facts ) {
222
+ const sentences = [];
223
+ const { failedJobs, totalJobs } = facts.outcome;
224
+ if (failedJobs > 0) {
225
+ if (eligible.some((finding) => !FAILURE_TYPES.has(finding.type))) {
226
+ sentences.push('Fix the failure before tuning: the other findings cover only the work that ran.');
227
+ }
228
+ } else if (totalJobs > 0 && !facts.incomplete) {
229
+ sentences.push(totalJobs === 1 ? 'Its one job succeeded.' : `All ${totalJobs} jobs succeeded.`);
230
+ }
231
+ if (eligible.length === 0) {
232
+ if (failedJobs > 0) return sentences;
233
+ if (facts.clean) sentences.push('Every check passed for this run.');
234
+ } else if (steps.length === 0) {
235
+ sentences.push('They are listed by impact under Findings.');
236
+ } else {
237
+ sentences.push(`${plural(eligible.length, 'finding')} in ${plural(steps.length, 'place')}.`);
238
+ const wallClock = steps[0].lead.impactEstimate?.wallClock;
239
+ // An idle-capacity lead needs no sentence here: the title gives the idle share and step 1 the fix.
240
+ if (!isIdleCapacityStep(steps[0]) && wallClock && facts.runMs != null) {
241
+ sentences.push(`The first fix could save up to ${formatDuration(wallClock.high)} of this ${formatDuration(facts.runMs)} run.`);
242
+ }
243
+ if (steps.some((step) => step.related.length > 0)) {
244
+ sentences.push('Findings on one stage are grouped, and their savings overlap.');
245
+ }
246
+ }
247
+ if (facts.incomplete) {
248
+ sentences.push('The log has no end-of-run record, so these figures cover only the part of the run it captured.');
249
+ }
250
+ const leadIsIdle = steps.length > 0 && isIdleCapacityStep(steps[0]);
251
+ if (!leadIsIdle && facts.idlePct != null && facts.idlePct >= IDLE_NOTABLE_PCT) {
252
+ sentences.push(
253
+ steps.some(isIdleCapacityStep)
254
+ ? `${facts.idlePct}% of the executor capacity sat idle, so the cluster may be larger than this job needs.`
255
+ : `${facts.idlePct}% of the run's core time went unused, so the cluster may be larger than this job needs.`,
256
+ );
257
+ }
258
+ return sentences;
259
+ }
260
+
261
+ const STACK_TRACE_HINT = 'Open the driver log only if you need the full stack trace.';
262
+
263
+ /** What the failure step whose reason the verdict quotes tells the reader:
264
+ * the detector's "inspect the driver log for the reason" would send a
265
+ * newcomer looking for something already on screen. The copied text carries
266
+ * the reason itself, since "quoted above" means nothing once pasted. */
267
+ export function quotedReasonText(reason ) {
268
+ const sentence = /[.!?]$/.test(reason) ? reason : `${reason}.`;
269
+ return {
270
+ shown: `Spark's recorded reason is quoted above. ${STACK_TRACE_HINT}`,
271
+ copied: `Spark's recorded reason: ${sentence} ${STACK_TRACE_HINT}`,
272
+ };
273
+ }
274
+
275
+ /** One step as pasteable text: the action, what to try, and the savings. */
276
+ export function stepCopyText(finding , recommendation , stageId = null) {
277
+ const impact = impactFigure(finding);
278
+ const meaning = savingsMeaning(finding);
279
+ const savings = impact && meaning ? `${impact} ${meaning}` : impact;
280
+ const where = stageId != null ? ` in Stage ${stageId}` : '';
281
+ const headline = `${findingActionLabel(finding)}${where}: ${recommendation}`;
282
+ return [/[.!?]$/.test(headline) ? headline : `${headline}.`, savings ? `Potential savings: ${savings}` : null]
283
+ .filter(Boolean)
284
+ .join(' ');
285
+ }
286
+
287
+ /** The whole verdict as a pasteable checklist for a ticket or a message:
288
+ * run, verdict, numbered steps (with their stage), and how many more places
289
+ * the full list holds. */
290
+ export function planCopyText(input
291
+
292
+
293
+
294
+
295
+ ) {
296
+ const lines = [input.runName ? `Spark run ${input.runName}: ${input.title}` : input.title, ''];
297
+ input.steps.forEach(({ step, recommendation }, index) => {
298
+ lines.push(`${index + 1}. ${stepCopyText(step.lead, recommendation, step.stageId)}`);
299
+ });
300
+ if (input.remaining > 0) lines.push('', `${plural(input.remaining, 'more place')} to look at in the full findings list.`);
301
+ return lines.join('\n');
302
+ }
303
+
304
+ /** A step's "What to try" text for copying: Spark's quoted reason when it is this step's own
305
+ * failure, else the finding's recommendation. */
306
+ export function stepCopyRecommendation(step , outcome ) {
307
+ return quotesReasonOf(step.lead, outcome) && outcome.reason
308
+ ? quotedReasonText(outcome.reason).copied
309
+ : recommendationText(step.lead);
310
+ }
311
+
312
+ /** Everything the verdict card shows, computed once for a run. `eligible` is what the verdict
313
+ * counts and ranks: the rollup-eligible findings of a known display type. */
314
+
315
+
316
+
317
+
318
+
319
+
320
+
321
+
322
+
323
+
324
+
325
+
326
+
327
+
328
+ export function buildRunVerdict(appModel , allFindings ) {
329
+ const eligible = allFindings.filter((finding) => isEligible(finding) && DISPLAY_INDEX.has(finding.type));
330
+ const outcome = summarizeRunOutcome(appModel.jobs, allFindings);
331
+ const steps = buildNextSteps(eligible, outcome.failedJobs > 0 ? { failedJobStageIds: outcome.failedJobStageIds } : {});
332
+ const facts = {
333
+ runMs: hasCompleteApplicationInterval(appModel.app) ? computeWallClock(appModel.app, appModel.stages).total : null,
334
+ idlePct: verdictIdlePct(steps, getScorecardEstimates(appModel).wastage.value),
335
+ incomplete: allFindings.some((finding) => finding.type === 'incompleteRun'),
336
+ clean: isCleanRun(appModel, allFindings),
337
+ outcome,
338
+ noFinishedStages: !hasFinishedStage(appModel.stages),
339
+ };
340
+ const title = verdictTitle(eligible, steps, facts);
341
+ const shown = steps.slice(0, NEXT_STEP_LIMIT);
342
+ const remaining = steps.length - shown.length;
343
+ const copyText = shown.length > 0
344
+ ? planCopyText({
345
+ runName: appModel.app?.name ?? null,
346
+ title,
347
+ steps: shown.map((step) => ({ step, recommendation: stepCopyRecommendation(step, outcome) })),
348
+ remaining,
349
+ })
350
+ : null;
351
+ return { eligible, outcome, steps, facts, title, summary: verdictSummary(eligible, steps, facts), shown, remaining, copyText };
352
+ }
@@ -6,7 +6,6 @@
6
6
  // Builds on computeCoreTimeSeries' clamp conceptually, but operates on the
7
7
  // worker-posted per-stage aggregates (runAggregates.perStage) so raw task data
8
8
  // never reaches the main thread.
9
- import { computeWallClock } from './wall-clock.js';
10
9
  import { computeTotalCores } from './core-count.js';
11
10
 
12
11
 
@@ -30,9 +29,11 @@ function estimatedTotalAtCores(perStage
30
29
  return sum;
31
30
  }
32
31
 
33
- export function simulateScaling({ app, stages, runAggregates, executorsAdded }
32
+ /** `observedActiveMs` is the run's stages-active wall-clock (the interpretation's
33
+ * `wallClock.stagesActive`), the one observed makespan predictions are scaled against. */
34
+ export function simulateScaling({ app, observedActiveMs, runAggregates, executorsAdded }
35
+
34
36
 
35
-
36
37
 
37
38
 
38
39
  )
@@ -48,8 +49,6 @@ export function simulateScaling({ app, stages, runAggregates, executorsAdded }
48
49
  // `executorsAdded` cast: computeTotalCores only reads `totalCores`, present on
49
50
  // ExecutorAddedEvent (the only kind passed here) but not on the union type.
50
51
  const baselineCores = computeTotalCores(app ?? {}, executorsAdded );
51
- const wc = computeWallClock(app, stages );
52
- const observedActiveMs = wc.stagesActive;
53
52
 
54
53
  // Model Error: predicted-at-baseline vs. observed stages-active wall-clock.
55
54
  const predictedAtBaseline = baselineCores > 0 ? estimatedTotalAtCores(perStage, baselineCores) : 0;
@@ -0,0 +1,63 @@
1
+ import { computeEfficiencyModel } from './efficiency-model.js';
2
+ import { computeWallClock } from './wall-clock.js';
3
+
4
+
5
+
6
+
7
+
8
+
9
+
10
+
11
+
12
+
13
+
14
+
15
+ export function hasCompleteApplicationInterval(app ) {
16
+ return typeof app?.startTime === 'number'
17
+ && Number.isFinite(app.startTime)
18
+ && typeof app?.endTime === 'number'
19
+ && Number.isFinite(app.endTime)
20
+ && app.endTime > app.startTime;
21
+ }
22
+
23
+ export function getScorecardEstimates(appModel )
24
+
25
+
26
+ {
27
+ if (!hasCompleteApplicationInterval(appModel.app)) {
28
+ return {
29
+ efficiency: { value: null, unavailableReason: 'application-timing' },
30
+ wastage: { value: null, unavailableReason: 'application-timing' },
31
+ };
32
+ }
33
+
34
+ const wallClock = computeWallClock(appModel.app, appModel.stages);
35
+ const efficiency = Math.min(100, Math.round((wallClock.stagesActive / wallClock.total) * 100));
36
+
37
+ if (!appModel.runAggregates) {
38
+ return {
39
+ efficiency: { value: efficiency, unavailableReason: null },
40
+ wastage: { value: null, unavailableReason: 'core-usage-summary' },
41
+ };
42
+ }
43
+
44
+ const efficiencyModel = computeEfficiencyModel({
45
+ app: appModel.app,
46
+ stages: appModel.stages,
47
+ executorsAdded: appModel.executors.added,
48
+ executorsRemoved: appModel.executors.removed,
49
+ runAggregates: appModel.runAggregates,
50
+ });
51
+
52
+ if (!Number.isFinite(efficiencyModel.availableComputeHours) || efficiencyModel.availableComputeHours <= 0) {
53
+ return {
54
+ efficiency: { value: efficiency, unavailableReason: null },
55
+ wastage: { value: null, unavailableReason: 'executor-capacity' },
56
+ };
57
+ }
58
+
59
+ return {
60
+ efficiency: { value: efficiency, unavailableReason: null },
61
+ wastage: { value: efficiencyModel.wastagePct, unavailableReason: null },
62
+ };
63
+ }
@@ -24,6 +24,8 @@ import { isSupportedEvidenceAvailability } from './evidence-availability.js';
24
24
 
25
25
 
26
26
 
27
+
28
+
27
29
 
28
30
 
29
31
 
@@ -44,6 +46,8 @@ export function captureSnapshot(
44
46
  jobs: new Map(appModel.jobs),
45
47
  runAggregates: appModel.runAggregates,
46
48
  evidenceAvailability: appModel.evidenceAvailability,
49
+ skippedLines: appModel.skippedLines,
50
+ unreadableSqlExecutions: appModel.unreadableSqlExecutions && [...appModel.unreadableSqlExecutions],
47
51
  catalog: [...catalog],
48
52
  taskData: new Map(taskDataCache),
49
53
  };
@@ -71,6 +75,9 @@ export function applySnapshot(
71
75
  appModel.evidenceAvailability = isSupportedEvidenceAvailability(snapshot.evidenceAvailability)
72
76
  ? snapshot.evidenceAvailability
73
77
  : null;
78
+ appModel.skippedLines = snapshot.skippedLines;
79
+ if (snapshot.unreadableSqlExecutions) appModel.unreadableSqlExecutions = [...snapshot.unreadableSqlExecutions];
80
+ else delete appModel.unreadableSqlExecutions;
74
81
 
75
82
  taskDataCache.clear();
76
83
  for (const [k, v] of snapshot.taskData) taskDataCache.set(k, v);
@@ -1,8 +1,8 @@
1
1
  import { z } from 'zod';
2
2
 
3
3
  // Proxy-level error envelope: the JSON body the local server's SHS proxy
4
- // (./shs-request.js) sends back on a non-OK upstream response, e.g.
5
- // `{ code: 'shs-unreachable' }`. This is NOT a SparkListener* event shape, so
4
+ // (./proxy.js) sends back on a non-OK upstream response, e.g.
5
+ // `{ code: 'upstream-unreachable' }`. This is NOT a SparkListener* event shape, so
6
6
  // it lives here rather than in event-schemas.ts. `.passthrough()` since the
7
7
  // proxy may attach extra debugging fields the consumer doesn't care about;
8
8
  // only `code` is read.
@@ -0,0 +1,17 @@
1
+ // Parse a Spark memory-size string to MiB. Spark's JVM-memory configs use bytesConf(ByteUnit.MiB),
2
+ // so a bare number means MiB. A k/m/g/t suffix sets the unit (trailing "b" redundant); a lone "b"
3
+ // ("10b") means bytes.
4
+ export function parseSparkMemoryMB(value ) {
5
+ if (value == null) return null;
6
+ const m = String(value).trim().toLowerCase().match(/^([\d.]+)\s*([kmgt]?)(b?)$/);
7
+ if (!m) return null;
8
+ const n = parseFloat(m[1]);
9
+ if (!Number.isFinite(n)) return null;
10
+ switch (m[2]) {
11
+ case 'k': return Math.round(n / 1024);
12
+ case 'g': return Math.round(n * 1024);
13
+ case 't': return Math.round(n * 1024 * 1024);
14
+ case 'm': return Math.round(n);
15
+ default: return m[3] === 'b' ? Math.round(n / (1024 * 1024)) : Math.round(n);
16
+ }
17
+ }
@@ -0,0 +1,11 @@
1
+ // Shared by every scope:'sql' detector and the plan views. sql.get(id).stageIds is always empty (parser-worker
2
+ // never populates it), so stage linkage is derived from each stage's own sqlExecutionId.
3
+ // Minimal param shape (not DetectorStage): external callers pass Map<StageId, Stage>.
4
+ export function stageIdsForSqlExec(
5
+ executionId ,
6
+ stages ,
7
+ ) {
8
+ const out = [];
9
+ for (const s of stages.values()) if (s.sqlExecutionId === executionId) out.push(s.id);
10
+ return out;
11
+ }
@@ -0,0 +1,18 @@
1
+
2
+
3
+ /** The plan nodes a stage actually ran: its SQL execution's resolved plan tree, filtered to the
4
+ * nodes attributed to it (`node.stageIds`). Empty when the stage has no SQL execution, the
5
+ * execution has no resolved plan, or no node is attributed to it. The one stage-to-plan mapping:
6
+ * stageIdentity's fingerprint and the Python-stage check both read it. */
7
+ export function planNodesOfStage(stage , sql ) {
8
+ const execId = stage.sqlExecutionId;
9
+ if (execId == null) return [];
10
+ const root = sql.get(execId)?.planTree ?? null;
11
+ if (!root) return [];
12
+ const nodes = [];
13
+ (function collect(node ) {
14
+ if (node.stageIds?.includes(stage.id)) nodes.push(node);
15
+ for (const child of node.children ?? []) collect(child);
16
+ })(root);
17
+ return nodes;
18
+ }
@@ -56,6 +56,7 @@ const TASK_FIELD_PROPS = TASK_FIELD_DESCRIPTORS.map((d) => d.prop);
56
56
 
57
57
 
58
58
 
59
+
59
60
 
60
61
 
61
62
  export function finalizeStage(
@@ -124,9 +125,11 @@ export function finalizeStage(
124
125
  acc.executorCpuTime += t.executorCpuTime;
125
126
  acc.inputBytes += t.inputBytes;
126
127
  acc.outputBytes += t.outputBytes;
128
+ if (t.outputRecords != null) acc.outputRecords = (acc.outputRecords ?? 0) + t.outputRecords;
127
129
  for (let i = 0; i < TASK_FIELD_PROPS.length; i++) buf.push(t[TASK_FIELD_PROPS[i]]);
128
130
  }
129
131
  stage.taskCount = taskCount;
132
+ stage.peakExecutionMemoryMax = peakExecutionMemoryMax;
130
133
  stage.failedTasks = failedTasks;
131
134
  stage.speculativeTasks = speculativeTasks;
132
135
  stage.taskAttempts = null; // no longer needed after finalize, freeing memory
@@ -199,6 +202,9 @@ export function finalizeStage(
199
202
  };
200
203
  delete data.taskAttempts; // internal-only field, already nulled above; never part of the public message
201
204
  delete data.failureDetails; // internal-only intern table, summarized by failureGroups
205
+ delete data.speculativeWinners; // internal-only late-TaskEnd pairing state, kept worker-side
206
+ delete data.lateSpeculationWaste;
207
+ delete data.stageAttemptId;
202
208
 
203
209
  return { type: 'stage', data };
204
210
  }