sparkforensics-mcp 0.4.0 → 0.5.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (64) hide show
  1. package/package.json +1 -1
  2. package/vendor-core/allocation.js +106 -0
  3. package/vendor-core/analyzer.js +13 -13
  4. package/vendor-core/cli/budgets.js +23 -9
  5. package/vendor-core/cli/collect-run.js +11 -4
  6. package/vendor-core/cli/regression-budgets.js +83 -0
  7. package/vendor-core/comparison-verdict.js +22 -22
  8. package/vendor-core/core-source-hash.txt +1 -1
  9. package/vendor-core/detectors.js +202 -68
  10. package/vendor-core/docs-content/detection/cache.md +3 -2
  11. package/vendor-core/docs-content/detection/cfg.md +9 -8
  12. package/vendor-core/docs-content/detection/chrn.md +1 -2
  13. package/vendor-core/docs-content/detection/cold.md +4 -2
  14. package/vendor-core/docs-content/detection/fail.md +3 -2
  15. package/vendor-core/docs-content/detection/gc.md +3 -2
  16. package/vendor-core/docs-content/detection/host.md +2 -1
  17. package/vendor-core/docs-content/detection/local.md +1 -1
  18. package/vendor-core/docs-content/detection/mem.md +5 -2
  19. package/vendor-core/docs-content/detection/plan.md +2 -1
  20. package/vendor-core/docs-content/detection/sfail.md +2 -1
  21. package/vendor-core/docs-content/detection/shape.md +5 -4
  22. package/vendor-core/docs-content/detection/skew.md +3 -1
  23. package/vendor-core/docs-content/detection/slow.md +2 -2
  24. package/vendor-core/docs-content/detection/spec.md +2 -3
  25. package/vendor-core/docs-content/detection/spill.md +1 -1
  26. package/vendor-core/docs-site-config.js +1 -1
  27. package/vendor-core/effective-conf.js +107 -0
  28. package/vendor-core/efficiency-model.js +8 -6
  29. package/vendor-core/event-handlers.js +160 -47
  30. package/vendor-core/event-schemas.js +2 -0
  31. package/vendor-core/evidence-report.js +18 -10
  32. package/vendor-core/finding-generic-recommendation.js +20 -1
  33. package/vendor-core/finding-names.js +7 -0
  34. package/vendor-core/finding-presentation.js +61 -26
  35. package/vendor-core/finding-tag-help.js +1 -1
  36. package/vendor-core/finding-types.js +12 -0
  37. package/vendor-core/format-utils.js +4 -3
  38. package/vendor-core/impact-estimator.js +20 -2
  39. package/vendor-core/impact-format.js +14 -13
  40. package/vendor-core/impact-model.js +27 -5
  41. package/vendor-core/ingest.js +4 -2
  42. package/vendor-core/list-runs.js +5 -2
  43. package/vendor-core/mcp-tools.js +1 -1
  44. package/vendor-core/model-assembler.js +23 -1
  45. package/vendor-core/parser-worker.js +1 -1
  46. package/vendor-core/proxy.js +3 -1
  47. package/vendor-core/python-stage.js +25 -0
  48. package/vendor-core/recommendation-rollup.js +16 -9
  49. package/vendor-core/redact.js +51 -10
  50. package/vendor-core/remediation.js +20 -0
  51. package/vendor-core/run-comparison.js +43 -24
  52. package/vendor-core/run-interpretation.js +2 -1
  53. package/vendor-core/run-metrics.js +198 -0
  54. package/vendor-core/run-totals.js +24 -0
  55. package/vendor-core/run-verdict.js +3 -4
  56. package/vendor-core/scorecard-estimates.js +1 -0
  57. package/vendor-core/session-snapshot.js +7 -0
  58. package/vendor-core/shs-schemas.js +2 -2
  59. package/vendor-core/spark-memory.js +17 -0
  60. package/vendor-core/stage-plan-nodes.js +18 -0
  61. package/vendor-core/stage-quantiles.js +4 -0
  62. package/vendor-core/types.js +49 -1
  63. package/vendor-core/wasted-core-hours.js +10 -7
  64. package/vendor-core/write-targets.js +312 -0
@@ -1,7 +1,7 @@
1
1
  // Explicit .ts extensions: plain Node's ESM resolver (the runtime CLI/MCP path
2
2
  // runs under) requires the exact specifier, unlike a bundler.
3
3
  import { mergeIntervals } from './intervals.js';
4
- import { formatWallClockRange, worstImpactBand, IMPACT_BAND_ORDER } from './format-utils.js';
4
+ import { formatRawWaste, formatWallClockRange, readsAsZero, worstImpactBand, IMPACT_BAND_ORDER } from './format-utils.js';
5
5
 
6
6
 
7
7
 
@@ -124,11 +124,15 @@ export function buildRecommendationRollup(
124
124
  });
125
125
  }
126
126
 
127
- /** A group's trailing figure on the Findings board, and its tooltip. `time`
128
- * and `resource` both lead with the "×N" finding count (the "worth
129
- * expanding" signal); `count` skips it since the impact-band tally already
130
- * implies N. The row stays terse ("×2 · 476ms recoverable") to fit a dense
131
- * right-aligned column; the title spells the shorthand out. */
127
+ /** A resource group's summed waste as the board and CLI print it; null when it reads as zero. */
128
+ export function resourceGroupTotal(group ) {
129
+ const text = formatRawWaste({ value: group.total, unit: group.unit });
130
+ return readsAsZero(text) ? null : text;
131
+ }
132
+
133
+ /** A group's trailing figure on the Findings board, and its tooltip. Every kind leads with the
134
+ * "×N" finding count (the "worth expanding" signal). The row stays terse ("×2 · 476ms
135
+ * recoverable") to fit a dense right-aligned column; the title spells the shorthand out. */
132
136
  export function rollupGroupStat(group ) {
133
137
  if (group.kind === 'time') {
134
138
  const recoverable = formatWallClockRange(group.recoverableMsHigh, group.recoverableMsHigh);
@@ -138,13 +142,16 @@ export function rollupGroupStat(group )
138
142
  };
139
143
  }
140
144
  if (group.kind === 'resource') {
145
+ const total = resourceGroupTotal(group);
141
146
  return {
142
- stat: `×${group.findingCount} · resource-cost projection`,
143
- statTitle: `${group.findingCount} findings of this type; a resource-cost estimate (not run time) is projected for fixing them`,
147
+ stat: total ? `×${group.findingCount} · ${total}` : `×${group.findingCount}`,
148
+ statTitle: `${group.findingCount} findings of this type${total ? `; ${total} in total, a resource cost, not run time` : ''}`,
144
149
  };
145
150
  }
151
+ // The band heading above the row already names a one-band group's band.
152
+ const bands = Object.entries(group.byImpactBand);
146
153
  return {
147
- stat: Object.entries(group.byImpactBand).map(([impactBand, count]) => `${count} ${impactBand}`).join(', '),
154
+ stat: bands.length === 1 ? `×${group.findingCount}` : `×${group.findingCount} · ${bands.map(([impactBand, count]) => `${count} ${impactBand}`).join(', ')}`,
148
155
  statTitle: `${group.findingCount} findings of this type, by impact`,
149
156
  };
150
157
  }
@@ -9,13 +9,12 @@
9
9
 
10
10
  import { decodeCollections, encodeCollections, } from './export-data.js';
11
11
 
12
- import { redactTaskFailureGroup, } from './task-failure.js';
12
+ import { REDACTED_TEXT, redactTaskFailureGroup, } from './task-failure.js';
13
13
 
14
14
  // Host / IP identifier patterns. Used to enumerate host names that surface only
15
- // inside free text: recommendation strings, `stageFailed`'s failure-reason
16
- // value, SQL relation/node names, never as a structured `host` field, so
17
- // redaction reaches those residuals too. Pseudonyms (`host-1`) match neither
18
- // pattern, keeping the scan idempotent.
15
+ // inside free text: recommendation strings, SQL relation/node names, never as a
16
+ // structured `host` field, so redaction reaches those residuals too. Pseudonyms
17
+ // (`host-1`) match neither pattern, keeping the scan idempotent.
19
18
  const HOST_PATTERNS = [
20
19
  // EC2-style ip-10-1-2-3 with an optional dotted domain (ip-10-1-2-3.ec2.internal).
21
20
  // Each domain label must start with an alphanumeric, so a trailing sentence
@@ -95,6 +94,31 @@ function redactFailureGroups (node ) {
95
94
  return node;
96
95
  }
97
96
 
97
+ // A stage's failure reason is Spark's free-text message and can carry file paths and data values
98
+ // too, so it is replaced outright, like a failure group's message: a `stageFailed` finding's
99
+ // `valueText`, the `stageFailureReason` of a stage record, a job's `exception` (the run outcome's
100
+ // fallback reason), and the report's `failureReason` copy. Walking by key name, like
101
+ // collectHostFields, needs no path list. Returns a fresh tree.
102
+ const REASON_KEYS = new Set(['stageFailureReason', 'exception', 'failureReason']);
103
+ function redactStageFailureReasons (node ) {
104
+ if (Array.isArray(node)) return node.map((n) => redactStageFailureReasons(n)) ;
105
+ if (node && typeof node === 'object') {
106
+ const isStageFailed = (node ).type === 'stageFailed';
107
+ const out = {};
108
+ for (const [k, v] of Object.entries(node)) {
109
+ const isReason = REASON_KEYS.has(k) || (isStageFailed && k === 'valueText');
110
+ out[k] = isReason && typeof v === 'string' ? REDACTED_TEXT : redactStageFailureReasons(v);
111
+ }
112
+ return out ;
113
+ }
114
+ return node;
115
+ }
116
+
117
+ // Both passes for free text no host/app-id pattern recognizes.
118
+ function redactFailureText (node ) {
119
+ return redactStageFailureReasons(redactFailureGroups(node));
120
+ }
121
+
98
122
  // Spark config keys ending in `host`/`hostname` (e.g. spark.driver.host,
99
123
  // spark.yarn.am.hostname) carry plain FQDN host names that neither
100
124
  // HOST_PATTERNS matches (no IP/EC2 shape) nor collectHostFields's by-key-name
@@ -122,7 +146,7 @@ function collectConfigHostValues(config , hos
122
146
 
123
147
 
124
148
 
125
-
149
+
126
150
 
127
151
 
128
152
 
@@ -188,10 +212,19 @@ function applyReplacements (node , ids
188
212
  return deepReplace(node, merged) ;
189
213
  }
190
214
 
215
+ // The app name identifies a job as much as its id, so a redacted name becomes the id's
216
+ // pseudonym, as list_runs does. Only the structured field is replaced, never substrings: a
217
+ // short name ("t", "etl") would otherwise corrupt every string it happens to occur in.
218
+ function withRedactedName (app ) {
219
+ return app.name == null ? app : { ...app, name: app.id ?? null };
220
+ }
221
+
191
222
  export function redactReport (input ) {
192
- const report = redactFailureGroups(input);
223
+ const report = redactFailureText(input);
193
224
  const { appIds, hosts } = collectIds(report);
194
- return applyReplacements(report, { appIds, hosts });
225
+ const out = applyReplacements(report, { appIds, hosts });
226
+ const app = out.summary?.app;
227
+ return app ? { ...out, summary: { ...out.summary, app: withRedactedName(app) } } : out;
195
228
  }
196
229
 
197
230
  // Run-comparison counterpart: no single app-id *field* to pseudonymize
@@ -218,7 +251,7 @@ export function redactComparison (comparison ) {
218
251
 
219
252
 
220
253
  function redactRunTree (input ) {
221
- const data = redactFailureGroups(input);
254
+ const data = redactFailureText(input);
222
255
  const appIds = new Set ();
223
256
  const hosts = new Set ();
224
257
  const appId = data.app?.id;
@@ -229,7 +262,12 @@ function redactRunTree (input ) {
229
262
  collectHostFields(data.configFindings, hosts);
230
263
  collectConfigHostValues(data.app?.config, hosts);
231
264
  scanTokens(data, [{ patterns: HOST_PATTERNS, out: hosts }, { patterns: APP_ID_PATTERNS, out: appIds }]);
232
- return applyReplacements(data, { appIds, hosts });
265
+ const out = applyReplacements(data, { appIds, hosts });
266
+ if (!out.app) return out;
267
+ const app = withRedactedName(out.app);
268
+ // spark.app.name repeats the name in the config the HTML export ships.
269
+ if (app.config?.['spark.app.name'] == null) return { ...out, app };
270
+ return { ...out, app: { ...app, config: { ...app.config, 'spark.app.name': app.id ?? '' } } };
233
271
  }
234
272
 
235
273
  /** A run's model and findings with every identifier pseudonymized. Redact this before anything derives
@@ -264,6 +302,9 @@ export function redactRunModel(
264
302
  executors: redacted.executors,
265
303
  runAggregates: redacted.runAggregates ,
266
304
  evidenceAvailability: redacted.evidenceAvailability ,
305
+ // Counts and execution ids, nothing to pseudonymize.
306
+ skippedLines: appModel.skippedLines,
307
+ unreadableSqlExecutions: appModel.unreadableSqlExecutions,
267
308
  } ,
268
309
  catalog: redacted.catalog,
269
310
  configFindings: redacted.configFindings,
@@ -0,0 +1,20 @@
1
+ // Constructors for the structured `remediation` a detector attaches next to its prose
2
+ // `recommendation`: only where the detector already names the Spark property, never an invented one.
3
+
4
+
5
+
6
+
7
+ /** The property should be raised; `suggested` is the value when the detector computed one. */
8
+ export function increaseConf(key , suggested = null) {
9
+ return { kind: 'conf', key, direction: 'increase', suggested };
10
+ }
11
+
12
+ /** The property should be lowered; `suggested` is the value when the detector computed one. */
13
+ export function decreaseConf(key , suggested = null) {
14
+ return { kind: 'conf', key, direction: 'decrease', suggested };
15
+ }
16
+
17
+ /** The property should take a specific value (a switch or a class name), not move up or down. */
18
+ export function setConf(key , suggested = null) {
19
+ return { kind: 'conf', key, direction: 'set', suggested };
20
+ }
@@ -1,6 +1,9 @@
1
1
  import { computeWallClock } from './wall-clock.js';
2
2
  import { normalizeDetail } from './detectors.js';
3
3
  import { cyrb53 } from './string-hash.js';
4
+ import { computeAllocation } from './allocation.js';
5
+ import { totalExecutorCpuMs, withEarlierAttempts } from './run-totals.js';
6
+ import { planNodesOfStage } from './stage-plan-nodes.js';
4
7
  import { captureSnapshot } from './session-snapshot.js';
5
8
  import { tunedRunNote } from './threshold-overrides.js';
6
9
 
@@ -76,23 +79,18 @@ function planTreeIdentity(root ) {
76
79
  // Falls back to the coarser whole-tree identity when the stage has no
77
80
  // attributed nodes (hand-built snapshots without `stageIds`, or unmatched
78
81
  // accumulables).
79
- function sqlNodeIdentity(stage , snapshot ) {
82
+ function sqlNodeIdentity(stage , snapshot ) {
80
83
  const execId = stage.sqlExecutionId;
81
84
  if (execId == null) return '';
82
85
  const root = snapshot.sql.get(execId)?.planTree ?? null;
83
86
  if (!root) return '';
84
- const fingerprints = [];
85
- (function collect(node ) {
86
- if (node.stageIds?.includes(stage.id)) {
87
- fingerprints.push(JSON.stringify([normalizeStageName(node.name ?? ''), normalizeDetail(node.detail ?? '')]));
88
- }
89
- for (const child of node.children ?? []) collect(child);
90
- })(root);
87
+ const fingerprints = planNodesOfStage(stage, snapshot.sql)
88
+ .map((node) => JSON.stringify([normalizeStageName(node.name ?? ''), normalizeDetail(node.detail ?? '')]));
91
89
  if (fingerprints.length === 0) return planTreeIdentity(root) ?? '';
92
90
  return cyrb53(JSON.stringify(fingerprints.sort()));
93
91
  }
94
92
 
95
- export function stageIdentity(stage , snapshot ) {
93
+ export function stageIdentity(stage , snapshot ) {
96
94
  return normalizeStageName(stage.name ?? '') + '§' + sqlNodeIdentity(stage, snapshot);
97
95
  }
98
96
 
@@ -199,6 +197,7 @@ function skewRatios(stages ) {
199
197
  export const COMPARISON_METRIC_KEYS = [
200
198
  'wallClock', 'shuffleSpill', 'taskSkew', 'failedTaskRate', 'diskSpill', 'gcTime',
201
199
  'inputBytes', 'outputBytes', 'executorRunTime', 'taskCount', 'executorsAdded',
200
+ 'executorCpuTime', 'allocatedCoreHours',
202
201
  ];
203
202
 
204
203
  // Volume/count metrics, not cost metrics: more or less input/output data, or
@@ -228,20 +227,24 @@ function metric(
228
227
 
229
228
  export function metricDeltas(baseSnap , candSnap ) {
230
229
  const out = [];
230
+ // Run totals count every task attempt, failed and speculative ones too (run-totals.ts), the
231
+ // same sums the CLI metrics block reports; skew stays on each stage's latest attempt.
232
+ const attemptsOf = (snap ) => withEarlierAttempts([...snap.stages.values()]);
231
233
 
232
234
  // Wall-clock: always computable (computeWallClock tolerates a null app).
233
235
  out.push(metric('wallClock', 'Wall-clock duration',
234
236
  computeWallClock(baseSnap.app, baseSnap.stages).total,
235
237
  computeWallClock(candSnap.app, candSnap.stages).total));
236
238
 
237
- // Shuffle spill over the whole run: a sum needs no stage matching, and
238
- // matching is unreliable on real logs (see plan rationale), so scope it to
239
- // all stages exactly like task-skew and failed-rate below.
240
- const bSpill = sumField([...baseSnap.stages.values()], 'memoryBytesSpilled');
241
- const cSpill = sumField([...candSnap.stages.values()], 'memoryBytesSpilled');
242
- out.push(metric('shuffleSpill', 'Shuffle spill',
239
+ // Memory spill (Spark's memoryBytesSpilled) over the whole run: a sum needs no
240
+ // stage matching, and matching is unreliable on real logs, so scope it to all
241
+ // stages exactly like task-skew and failed-rate below. The key stays
242
+ // `shuffleSpill` so existing --regression-metric callers keep working.
243
+ const bSpill = sumField(attemptsOf(baseSnap), 'memoryBytesSpilled');
244
+ const cSpill = sumField(attemptsOf(candSnap), 'memoryBytesSpilled');
245
+ out.push(metric('shuffleSpill', 'Memory spill',
243
246
  bSpill.present ? bSpill.sum : null, cSpill.present ? cSpill.sum : null,
244
- { unavailableReason: bSpill.present && cSpill.present ? undefined : 'No shuffle-spill data recorded for a run' }));
247
+ { unavailableReason: bSpill.present && cSpill.present ? undefined : 'No memory-spill data recorded for a run' }));
245
248
 
246
249
  // Task skew: p95 of per-stage ratio across the whole run.
247
250
  const bSkew = p95(skewRatios([...baseSnap.stages.values()]));
@@ -250,10 +253,10 @@ export function metricDeltas(baseSnap , candSnap
250
253
  { unavailableReason: bSkew != null && cSkew != null ? undefined : 'No stage had measurable duration for a run' }));
251
254
 
252
255
  // Failed-task rate: Σ failedTasks / Σ taskCount.
253
- const bTasks = sumField([...baseSnap.stages.values()], 'taskCount');
254
- const cTasks = sumField([...candSnap.stages.values()], 'taskCount');
255
- const bFailed = sumField([...baseSnap.stages.values()], 'failedTasks');
256
- const cFailed = sumField([...candSnap.stages.values()], 'failedTasks');
256
+ const bTasks = sumField(attemptsOf(baseSnap), 'taskCount');
257
+ const cTasks = sumField(attemptsOf(candSnap), 'taskCount');
258
+ const bFailed = sumField(attemptsOf(baseSnap), 'failedTasks');
259
+ const cFailed = sumField(attemptsOf(candSnap), 'failedTasks');
257
260
  // Guard on BOTH inputs: a missing `failedTasks` field must render Unavailable,
258
261
  // not a false 0% rate (dividing an absent-and-therefore-0 numerator).
259
262
  const bRate = bTasks.present && bTasks.sum > 0 && bFailed.present ? bFailed.sum / bTasks.sum : null;
@@ -265,8 +268,8 @@ export function metricDeltas(baseSnap , candSnap
265
268
  // carries (set in finalizeStage). Correct at any match coverage, like the
266
269
  // sums above; no parser or detector change.
267
270
  const sumMetric = (key , label , field , reason ) => {
268
- const b = sumField([...baseSnap.stages.values()], field);
269
- const c = sumField([...candSnap.stages.values()], field);
271
+ const b = sumField(attemptsOf(baseSnap), field);
272
+ const c = sumField(attemptsOf(candSnap), field);
270
273
  out.push(metric(key, label, b.present ? b.sum : null, c.present ? c.sum : null,
271
274
  { unavailableReason: b.present && c.present ? undefined : reason }));
272
275
  };
@@ -285,6 +288,18 @@ export function metricDeltas(baseSnap , candSnap
285
288
  out.push(metric('executorsAdded', 'Executors added', bExec, cExec,
286
289
  { unavailableReason: bExec != null && cExec != null ? undefined : 'No executor events recorded for a run' }));
287
290
 
291
+ // Executor CPU time (ms) and allocated core-hours cost resources, so less is better. CPU time is
292
+ // null, not 0, on a run whose log never recorded it (older Spark).
293
+ const cpuMs = (snap ) => totalExecutorCpuMs(attemptsOf(snap));
294
+ const bCpu = cpuMs(baseSnap), cCpu = cpuMs(candSnap);
295
+ out.push(metric('executorCpuTime', 'Executor CPU time', bCpu, cCpu,
296
+ { unavailableReason: bCpu != null && cCpu != null ? undefined : 'No executor CPU time recorded for a run' }));
297
+ const coreHours = (snap ) =>
298
+ (snap.executors ? computeAllocation(snap).coreHours : null);
299
+ const bCore = coreHours(baseSnap), cCore = coreHours(candSnap);
300
+ out.push(metric('allocatedCoreHours', 'Allocated core-hours', bCore, cCore,
301
+ { unavailableReason: bCore != null && cCore != null ? undefined : 'No executor lifecycle or core count recorded for a run' }));
302
+
288
303
  return out;
289
304
  }
290
305
 
@@ -467,8 +482,12 @@ export function renderComparisonMarkdown(
467
482
  const lines = ['', '## Comparison to baseline', ''];
468
483
  if (tuned) lines.push(`- Tuned thresholds (both runs): ${tunedRunNote(tuned)}`, '');
469
484
  if (verdict) {
470
- // The dashboard comparison page's headline: run A is the baseline, run B the candidate.
471
- lines.push(`Run A: ${comparison.baselineLabel} · Run B: ${comparison.candidateLabel}`, '', verdict.title);
485
+ // Names each run by its label (MCP's run IDs), unless the labels are just the role names the
486
+ // CLI passes, where the verdict below already says baseline and candidate.
487
+ if (comparison.baselineLabel !== 'baseline' || comparison.candidateLabel !== 'candidate') {
488
+ lines.push(`Baseline: ${comparison.baselineLabel} · Candidate: ${comparison.candidateLabel}`, '');
489
+ }
490
+ lines.push(verdict.title);
472
491
  if (verdict.sentences.length > 0) lines.push('', verdict.sentences.join(' '));
473
492
  lines.push('');
474
493
  }
@@ -192,6 +192,7 @@ function interpretEfficiency(appModel ) {
192
192
  app: appModel.app,
193
193
  stages: appModel.stages,
194
194
  executorsAdded: appModel.executors.added,
195
+ executorsRemoved: appModel.executors.removed,
195
196
  runAggregates: appModel.runAggregates,
196
197
  });
197
198
  }
@@ -270,7 +271,7 @@ export function interpretRun(appModel , catalog , configFindi
270
271
  coverage: interpretCoverage(appModel, allFindings),
271
272
  runShape: interpretRunShape(appModel),
272
273
  wallClock: computeWallClock(appModel.app, appModel.stages),
273
- wastedCoreHours: computeWastedCoreHours(appModel.app, appModel.executors.added, appModel.runAggregates),
274
+ wastedCoreHours: computeWastedCoreHours(appModel.app, appModel.executors.added, appModel.runAggregates, appModel.executors.removed),
274
275
  efficiency: interpretEfficiency(appModel),
275
276
  wallClockReliable: checkConcurrentJobGroups(appModel.jobs).wallClockReliable,
276
277
  etlPhases: attributeEtlPhases(appModel.stages),
@@ -0,0 +1,198 @@
1
+ // The machine-readable metrics block of the CLI's JSON output: run-level totals and per-stage rows
2
+ // an automated tuning loop can compare across runs without reading a report. Its schemaVersion
3
+ // moves independently of the evidence report's. A figure the log cannot provide is null, never 0.
4
+ import { computeAllocation, } from './allocation.js';
5
+ import { computeSkewRatio, ENTRY_BY_TYPE, } from './detectors.js';
6
+ import { totalExecutorCpuMs, withEarlierAttempts } from './run-totals.js';
7
+ import { isPythonStage } from './python-stage.js';
8
+ import { stageIdentity } from './run-comparison.js';
9
+ import { hasCompleteApplicationInterval } from './scorecard-estimates.js';
10
+ import { effectiveThresholds } from './threshold-overrides.js';
11
+ import { computeWallClock } from './wall-clock.js';
12
+
13
+
14
+ export const METRICS_SCHEMA_VERSION = 1;
15
+
16
+
17
+
18
+
19
+
20
+
21
+
22
+
23
+
24
+
25
+
26
+
27
+
28
+
29
+
30
+
31
+
32
+
33
+
34
+
35
+
36
+
37
+
38
+
39
+
40
+
41
+
42
+
43
+
44
+
45
+
46
+
47
+
48
+
49
+
50
+
51
+
52
+
53
+
54
+
55
+
56
+
57
+
58
+
59
+
60
+
61
+
62
+
63
+
64
+
65
+
66
+
67
+
68
+
69
+
70
+
71
+
72
+ const finite = (v ) => typeof v === 'number' && Number.isFinite(v);
73
+
74
+ // Σ of a field over the stages that carry a finite value; null when none does.
75
+ function sumOf(stages , pick ) {
76
+ let sum = 0, present = false;
77
+ for (const s of stages) {
78
+ const v = pick(s);
79
+ if (finite(v)) { sum += v; present = true; }
80
+ }
81
+ return present ? sum : null;
82
+ }
83
+
84
+ // 0 means either no execution memory used or not recorded; only a positive peak is reported.
85
+ function peakExecutionMemory(stages ) {
86
+ const peaks = stages.map((s) => s.peakExecutionMemoryMax).filter((v) => finite(v) && v > 0);
87
+ return peaks.length > 0 ? Math.max(...peaks) : null;
88
+ }
89
+
90
+ function maxSkew(stages , minTasksForP95 ) {
91
+ const ratios = stages
92
+ .map((s) => computeSkewRatio(s, minTasksForP95)?.ratio)
93
+ .filter((r) => finite(r));
94
+ return ratios.length > 0 ? Math.max(...ratios) : null;
95
+ }
96
+
97
+ // Failed stage attempts and stages submitted more than once, from the parser's per-stage attempt
98
+ // counts; null when any stage record lacks them.
99
+ function stageAttempts(stages ) {
100
+ let failed = 0, retried = 0;
101
+ for (const s of stages) {
102
+ if (!finite(s.stageAttempts) || !finite(s.failedStageAttempts)) return null;
103
+ failed += s.failedStageAttempts;
104
+ if (s.stageAttempts > 1) retried++;
105
+ }
106
+ return { failed, retried };
107
+ }
108
+
109
+ // Metrics of a set of stages, every attempt included; a row folds the stages sharing one
110
+ // fingerprint. Skew describes the latest attempt's task durations.
111
+ function stageMetrics(stages , minTasksForP95 ) {
112
+ const attempts = withEarlierAttempts(stages);
113
+ // Task-derived figures are null for stages that never finished (no task records).
114
+ // Superseded attempts carry work but no task count of their own, so they stay in once any
115
+ // attempt finished a task.
116
+ const finished = attempts.some((s) => (s.taskCount ?? 0) > 0) ? attempts : [];
117
+ const tasks = sumOf(attempts, (s) => s.taskCount);
118
+ const fromTasks = (pick ) => (finished.length > 0 ? sumOf(finished, pick) : null);
119
+ const durations = attempts.filter((s) => finite(s.submittedAt) && finite(s.completedAt) && (s.completedAt ) >= (s.submittedAt ));
120
+ return {
121
+ durationMs: durations.length > 0 ? sumOf(durations, (s) => (s.completedAt ) - (s.submittedAt )) : null,
122
+ executorCpuTimeMs: totalExecutorCpuMs(finished),
123
+ executorRunTimeMs: fromTasks((s) => s.executorRunTime),
124
+ gcTimeMs: fromTasks((s) => s.jvmGCTime),
125
+ memorySpillBytes: fromTasks((s) => s.memoryBytesSpilled),
126
+ diskSpillBytes: fromTasks((s) => s.diskBytesSpilled),
127
+ shuffleReadBytes: fromTasks((s) => s.shuffleReadBytes),
128
+ shuffleWriteBytes: fromTasks((s) => s.shuffleWriteBytes),
129
+ inputBytes: fromTasks((s) => s.inputBytes),
130
+ outputBytes: fromTasks((s) => s.outputBytes),
131
+ outputRows: sumOf(finished, (s) => s.outputRecords),
132
+ peakExecutionMemoryBytes: peakExecutionMemory(finished),
133
+ taskCount: tasks != null && tasks > 0 ? tasks : null,
134
+ failedTasks: fromTasks((s) => s.failedTasks),
135
+ retriedTasks: fromTasks((s) => (s.wastedAttempts ) ?? 0),
136
+ skew: maxSkew(stages.filter((s) => (s.taskCount ?? 0) > 0), minTasksForP95),
137
+ };
138
+ }
139
+
140
+ /** The run's metrics block. `thresholds` only moves the skew ratio's P95 cutoff, as for the
141
+ * --max-skew budget. */
142
+ export function computeRunMetrics(appModel , thresholds ) {
143
+ const stageList = [...appModel.stages.values()];
144
+ const minTasksForP95 = effectiveThresholds(ENTRY_BY_TYPE.get('skew') , thresholds).minTasksForP95 ;
145
+ const all = stageMetrics(stageList, minTasksForP95);
146
+ const python = stageList.filter((s) => isPythonStage(s, appModel.sql));
147
+ const pythonRunTimeMs = sumOf(withEarlierAttempts(python), (s) => s.executorRunTime);
148
+
149
+ const byFingerprint = new Map ();
150
+ for (const s of stageList) {
151
+ const key = stageIdentity(s, appModel);
152
+ const group = byFingerprint.get(key);
153
+ if (group) group.push(s); else byFingerprint.set(key, [s]);
154
+ }
155
+ const stages = {};
156
+ for (const [key, group] of byFingerprint) {
157
+ const attempts = stageAttempts(group);
158
+ stages[key] = {
159
+ stageIds: group.map((s) => s.id).sort((a, b) => a - b),
160
+ failed: attempts ? attempts.failed > 0 : null,
161
+ retried: attempts ? attempts.retried > 0 : null,
162
+ python: group.some((s) => isPythonStage(s, appModel.sql)),
163
+ ...stageMetrics(group, minTasksForP95),
164
+ };
165
+ }
166
+
167
+ const attempts = stageAttempts(stageList);
168
+ return {
169
+ schemaVersion: METRICS_SCHEMA_VERSION,
170
+ runComplete: appModel.app?.endTime != null,
171
+ time: {
172
+ wallClockMs: hasCompleteApplicationInterval(appModel.app) ? computeWallClock(appModel.app, appModel.stages).total : null,
173
+ executorCpuTimeMs: all.executorCpuTimeMs,
174
+ executorRunTimeMs: all.executorRunTimeMs,
175
+ gcTimeMs: all.gcTimeMs,
176
+ },
177
+ data: {
178
+ memorySpillBytes: all.memorySpillBytes, diskSpillBytes: all.diskSpillBytes,
179
+ shuffleReadBytes: all.shuffleReadBytes, shuffleWriteBytes: all.shuffleWriteBytes,
180
+ inputBytes: all.inputBytes, outputBytes: all.outputBytes, outputRows: all.outputRows,
181
+ peakExecutionMemoryBytes: all.peakExecutionMemoryBytes,
182
+ },
183
+ shape: {
184
+ taskCount: all.taskCount,
185
+ stageCount: stageList.length,
186
+ failedStageAttempts: attempts?.failed ?? null,
187
+ retriedStages: attempts?.retried ?? null,
188
+ failedTasks: all.failedTasks,
189
+ retriedTasks: all.retriedTasks,
190
+ maxSkew: all.skew,
191
+ },
192
+ allocation: computeAllocation(appModel),
193
+ python: {
194
+ shareOfTaskRunTime: all.executorRunTimeMs != null && all.executorRunTimeMs > 0 ? (pythonRunTimeMs ?? 0) / all.executorRunTimeMs : null,
195
+ },
196
+ stages,
197
+ };
198
+ }
@@ -0,0 +1,24 @@
1
+ // The run-level sums every surface reads: the CLI metrics block and the run comparison (so the
2
+ // dashboard's comparison view and the MCP compare_runs tool) share these, so one name never means
3
+ // two figures.
4
+ import { nsToMs } from './format-utils.js';
5
+
6
+
7
+ /** Each stage followed by the work its own figures leave out, shaped like a stage: the attempts a
8
+ * resubmit replaced, and task attempts the stage record dropped (a failed attempt's late tasks,
9
+ * failed retries, losing speculative copies). Summing the result counts every attempt's work,
10
+ * failed ones too. A stage's own record, which the detectors and per-stage views read, is left as
11
+ * it is. */
12
+ export function withEarlierAttempts(stages ) {
13
+ return stages.flatMap((s) => [s, ...[s.earlierAttempts, s.lateAttemptWork]
14
+ .filter((work) => work != null)
15
+ .map(({ durationMs, ...totals }) => ({ id: s.id, ...totals, submittedAt: 0, completedAt: durationMs ?? undefined }))]);
16
+ }
17
+
18
+ /** Summed executor CPU time in ms. Spark records it in nanoseconds and the parser reads an absent
19
+ * metric as 0, so stages that all read 0 never recorded it (older Spark): null, not 0. */
20
+ export function totalExecutorCpuMs(stages ) {
21
+ let ns = 0;
22
+ for (const s of stages) if (typeof s.executorCpuTime === 'number' && Number.isFinite(s.executorCpuTime) && s.executorCpuTime > 0) ns += s.executorCpuTime;
23
+ return ns > 0 ? nsToMs(ns) : null;
24
+ }
@@ -236,13 +236,12 @@ export function verdictSummary(eligible , steps , facts
236
236
  } else {
237
237
  sentences.push(`${plural(eligible.length, 'finding')} in ${plural(steps.length, 'place')}.`);
238
238
  const wallClock = steps[0].lead.impactEstimate?.wallClock;
239
- if (isIdleCapacityStep(steps[0])) {
240
- sentences.push('A smaller cluster or dynamic allocation would free the idle cores for other jobs.');
241
- } else if (wallClock && facts.runMs != null) {
239
+ // An idle-capacity lead needs no sentence here: the title gives the idle share and step 1 the fix.
240
+ if (!isIdleCapacityStep(steps[0]) && wallClock && facts.runMs != null) {
242
241
  sentences.push(`The first fix could save up to ${formatDuration(wallClock.high)} of this ${formatDuration(facts.runMs)} run.`);
243
242
  }
244
243
  if (steps.some((step) => step.related.length > 0)) {
245
- sentences.push('Findings in the same stage usually share one cause, so they are grouped together and their savings overlap rather than add up.');
244
+ sentences.push('Findings on one stage are grouped, and their savings overlap.');
246
245
  }
247
246
  }
248
247
  if (facts.incomplete) {
@@ -45,6 +45,7 @@ export function getScorecardEstimates(appModel )
45
45
  app: appModel.app,
46
46
  stages: appModel.stages,
47
47
  executorsAdded: appModel.executors.added,
48
+ executorsRemoved: appModel.executors.removed,
48
49
  runAggregates: appModel.runAggregates,
49
50
  });
50
51
 
@@ -24,6 +24,8 @@ import { isSupportedEvidenceAvailability } from './evidence-availability.js';
24
24
 
25
25
 
26
26
 
27
+
28
+
27
29
 
28
30
 
29
31
 
@@ -44,6 +46,8 @@ export function captureSnapshot(
44
46
  jobs: new Map(appModel.jobs),
45
47
  runAggregates: appModel.runAggregates,
46
48
  evidenceAvailability: appModel.evidenceAvailability,
49
+ skippedLines: appModel.skippedLines,
50
+ unreadableSqlExecutions: appModel.unreadableSqlExecutions && [...appModel.unreadableSqlExecutions],
47
51
  catalog: [...catalog],
48
52
  taskData: new Map(taskDataCache),
49
53
  };
@@ -71,6 +75,9 @@ export function applySnapshot(
71
75
  appModel.evidenceAvailability = isSupportedEvidenceAvailability(snapshot.evidenceAvailability)
72
76
  ? snapshot.evidenceAvailability
73
77
  : null;
78
+ appModel.skippedLines = snapshot.skippedLines;
79
+ if (snapshot.unreadableSqlExecutions) appModel.unreadableSqlExecutions = [...snapshot.unreadableSqlExecutions];
80
+ else delete appModel.unreadableSqlExecutions;
74
81
 
75
82
  taskDataCache.clear();
76
83
  for (const [k, v] of snapshot.taskData) taskDataCache.set(k, v);
@@ -1,8 +1,8 @@
1
1
  import { z } from 'zod';
2
2
 
3
3
  // Proxy-level error envelope: the JSON body the local server's SHS proxy
4
- // (./shs-request.js) sends back on a non-OK upstream response, e.g.
5
- // `{ code: 'shs-unreachable' }`. This is NOT a SparkListener* event shape, so
4
+ // (./proxy.js) sends back on a non-OK upstream response, e.g.
5
+ // `{ code: 'upstream-unreachable' }`. This is NOT a SparkListener* event shape, so
6
6
  // it lives here rather than in event-schemas.ts. `.passthrough()` since the
7
7
  // proxy may attach extra debugging fields the consumer doesn't care about;
8
8
  // only `code` is read.
@@ -0,0 +1,17 @@
1
+ // Parse a Spark memory-size string to MiB. Spark's JVM-memory configs use bytesConf(ByteUnit.MiB),
2
+ // so a bare number means MiB. A k/m/g/t suffix sets the unit (trailing "b" redundant); a lone "b"
3
+ // ("10b") means bytes.
4
+ export function parseSparkMemoryMB(value ) {
5
+ if (value == null) return null;
6
+ const m = String(value).trim().toLowerCase().match(/^([\d.]+)\s*([kmgt]?)(b?)$/);
7
+ if (!m) return null;
8
+ const n = parseFloat(m[1]);
9
+ if (!Number.isFinite(n)) return null;
10
+ switch (m[2]) {
11
+ case 'k': return Math.round(n / 1024);
12
+ case 'g': return Math.round(n * 1024);
13
+ case 't': return Math.round(n * 1024 * 1024);
14
+ case 'm': return Math.round(n);
15
+ default: return m[3] === 'b' ? Math.round(n / (1024 * 1024)) : Math.round(n);
16
+ }
17
+ }