sparkforensics-mcp 0.3.0 → 0.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (62) hide show
  1. package/bin/sparkforensics-mcp.mjs +41 -10
  2. package/package.json +1 -1
  3. package/vendor-core/analyzer.js +156 -48
  4. package/vendor-core/check-coverage.js +88 -0
  5. package/vendor-core/cli/budgets.js +31 -18
  6. package/vendor-core/cli/collect-run.js +76 -31
  7. package/vendor-core/cli/threshold-config.js +28 -0
  8. package/vendor-core/comparison-verdict.js +177 -0
  9. package/vendor-core/core-source-hash.txt +1 -0
  10. package/vendor-core/core-usage-locality.js +56 -2
  11. package/vendor-core/detector-docs.js +58 -0
  12. package/vendor-core/detectors.js +918 -458
  13. package/vendor-core/docs-config.js +0 -36
  14. package/vendor-core/docs-content/chapters/nav-index.json +31 -0
  15. package/vendor-core/docs-content/detection/cstor.md +9 -0
  16. package/vendor-core/docs-site-config.js +3 -0
  17. package/vendor-core/event-handlers.js +170 -6
  18. package/vendor-core/event-schemas.js +21 -0
  19. package/vendor-core/evidence-report.js +421 -112
  20. package/vendor-core/export-data.js +79 -6
  21. package/vendor-core/finding-action-label.js +9 -88
  22. package/vendor-core/finding-filter-predicate.js +9 -0
  23. package/vendor-core/finding-generic-recommendation.js +6 -104
  24. package/vendor-core/finding-names.js +21 -45
  25. package/vendor-core/finding-presentation.js +333 -0
  26. package/vendor-core/finding-tag-help.js +110 -0
  27. package/vendor-core/finding-types.js +361 -0
  28. package/vendor-core/findings-of-type.js +11 -0
  29. package/vendor-core/format-utils.js +92 -27
  30. package/vendor-core/html-export.js +51 -0
  31. package/vendor-core/impact-band.js +21 -8
  32. package/vendor-core/impact-estimator.js +8 -521
  33. package/vendor-core/impact-format.js +114 -0
  34. package/vendor-core/impact-model.js +175 -0
  35. package/vendor-core/ingest.js +2 -0
  36. package/vendor-core/intervals.js +13 -0
  37. package/vendor-core/list-runs.js +2 -3
  38. package/vendor-core/load-vendored.js +70 -5
  39. package/vendor-core/mcp-server-factory.js +14 -10
  40. package/vendor-core/mcp-tools.js +105 -45
  41. package/vendor-core/model-assembler.js +12 -0
  42. package/vendor-core/occupancy.js +1 -1
  43. package/vendor-core/parser-worker.js +1 -1
  44. package/vendor-core/plan-graph-model.js +3 -2
  45. package/vendor-core/plan-node-detail.js +1 -1
  46. package/vendor-core/recommendation-rollup.js +63 -3
  47. package/vendor-core/redact.js +45 -27
  48. package/vendor-core/run-comparison.js +32 -7
  49. package/vendor-core/run-interpretation.js +290 -0
  50. package/vendor-core/run-outcome.js +74 -0
  51. package/vendor-core/run-payload.js +17 -0
  52. package/vendor-core/run-shape.js +40 -0
  53. package/vendor-core/run-verdict.js +353 -0
  54. package/vendor-core/scaling-sim.js +4 -5
  55. package/vendor-core/scorecard-estimates.js +62 -0
  56. package/vendor-core/sql-stages.js +11 -0
  57. package/vendor-core/stage-quantiles.js +2 -0
  58. package/vendor-core/threshold-overrides.js +160 -0
  59. package/vendor-core/threshold-summary.js +11 -33
  60. package/vendor-core/types.js +6 -42
  61. package/vendor-core/wall-clock.js +1 -12
  62. package/vendor-core/wasted-core-hours.js +2 -2
@@ -2,20 +2,33 @@
2
2
  // then serializes a run summary + findings into a deterministic, byte-stable JSON + Markdown.
3
3
  // Raw task records are never included (privacy baseline); redaction is opt-in via { redact: true }.
4
4
  import { analyze, auditConfig } from './analyzer.js';
5
- import { detectorCatalog } from './detectors.js';
6
- import { typeTag, formatBytes, formatDuration, IMPACT_BAND_ORDER } from './format-utils.js';
7
- import { FINDING_NAMES, titleCase } from './finding-names.js';
8
- import { redactReport } from './redact.js';
9
- import { formatTaskFailureHeadline, } from './task-failure.js';
10
- import { coreFindingActionLabel } from './finding-action-label.js';
11
- import { matchesFindingFilterCriteria } from './finding-filter-predicate.js';
12
- import { buildRecommendationRollup, isEligible, rankFindings, } from './recommendation-rollup.js';
5
+
6
+ import {
7
+ describeTunedThresholds, tunedDetectorCatalog, tunedDetectors, tunedRunNote, tunedThresholdsForType,
8
+ } from './threshold-overrides.js';
13
9
  import { getThresholdSummary } from './threshold-summary.js';
10
+ import {
11
+ typeTag, formatBytes, formatCores, formatDuration, formatRawWaste, formatWallClockRange, IMPACT_BAND_ORDER, readsAsZero,
12
+ } from './format-utils.js';
13
+ import { findingName, titleCase } from './finding-names.js';
14
+ import { redactReport, redactRunModel } from './redact.js';
15
+ import { formatTaskFailureHeadline, } from './task-failure.js';
16
+ import { findingActionLabel } from './finding-action-label.js';
17
+ import { matchesFindingFilterCriteria, singleStageId } from './finding-filter-predicate.js';
18
+ import { buildRecommendationRollup, isEligible, isRealFinding, rankFindings, } from './recommendation-rollup.js';
19
+ import { checkCoverage, isCleanRun } from './check-coverage.js';
20
+ import { buildRunVerdict, stepCopyRecommendation, stepCopyText, } from './run-verdict.js';
21
+ import {
22
+ estimateProvenance, impactEstimateFigure, impactFigure, rawWasteMeaning, savingsMeaning,
23
+ } from './impact-format.js';
24
+ import { computeRunShape, } from './run-shape.js';
25
+ import { detectorInfoByType } from './detector-docs.js';
14
26
 
15
-
27
+
28
+
16
29
 
17
30
 
18
- export const EVIDENCE_SCHEMA_VERSION = 3;
31
+ export const EVIDENCE_SCHEMA_VERSION = 5;
19
32
 
20
33
  // findingRow always sets id/metric/value/recommendation via `?? null` (never omits the key), and
21
34
  // buildJson does the same for evidenceAvailability and summary.app.{id,name,sparkVersion}: these
@@ -23,20 +36,34 @@ export const EVIDENCE_SCHEMA_VERSION = 3;
23
36
  // "unknown version" sentinel). Kept as `?? null`, not `?? undefined`: JSON.stringify drops
24
37
  // undefined keys but keeps null, so undefined would silently strip these from the report.
25
38
  //
26
- // `value` is number|string|null: stageFailed and the configAudit entries put text in Finding.value
27
- // instead of a magnitude, carried through unchanged.
39
+ // `value` is always a magnitude or null. The text-valued findings (stageFailed's failure reason,
40
+ // configAudit's current setting, incompleteRun's 'missing') carry theirs in `valueText` instead,
41
+ // present only on those rows.
28
42
 
29
-
30
-
31
-
43
+
44
+
45
+
32
46
 
33
47
 
34
48
 
35
49
 
36
50
 
37
51
 
52
+
53
+
54
+
55
+
56
+
57
+
58
+
38
59
 
39
60
 
61
+ // One row per finding, discriminated on `type`: `evidence` is that type's public evidence
62
+ // (FindingEvidenceMap, projected through EVIDENCE_KEYS below), never the finding's other fields.
63
+
64
+
65
+
66
+
40
67
  // The `Fix these first` rollup row: one entry per buildRecommendationRollup
41
68
  // group, so the CLI/MCP/download paths get the same impact-ranked aggregation the dashboard shows.
42
69
 
@@ -51,75 +78,185 @@ export const EVIDENCE_SCHEMA_VERSION = 3;
51
78
 
52
79
 
53
80
 
81
+
82
+
83
+
54
84
 
85
+
55
86
 
56
87
 
57
- // One line per detector `type` that fired zero findings, so a flat report can state "these were
58
- // checked and came back clean" like the dashboard's clean-checks table.
88
+ // One line per detector `type` that fired zero findings and could run, so a flat report can state
89
+ // "these were checked and came back clean" like the dashboard's clean-checks table.
59
90
 
60
91
 
61
92
 
62
93
 
94
+
95
+
96
+
97
+
98
+ // A detector `type` with zero findings that the log lacked the data to run (the dashboard's "Not
99
+ // checked on this log" group), so it is not reported as passed. `reason` says why, naming the
100
+ // setting to turn on where the detector gives one.
101
+
102
+
103
+
104
+
105
+
106
+
107
+ // How the run ended, as far as its jobs say (run-outcome.ts, the same summary the dashboard's
108
+ // verdict leads with). Counts only jobs with an end record; `failureReason` is the first line of
109
+ // Spark's own recorded reason, only when a job failed, and `failureReasonStageId` the stage it came
110
+ // from (null when there is no reason or it came from a job's exception).
111
+ // One verdict step: a place to look (a stage, or an app-level problem), led by its best-ranked
112
+ // finding, with the other finding types flagged there. `text` is the step's line in the
113
+ // dashboard's "Copy next steps" checklist.
114
+
115
+
116
+
117
+
118
+
119
+
120
+
121
+
122
+
123
+
124
+
125
+
126
+
127
+
128
+ // The dashboard's run verdict (run-verdict.ts): title, summary sentences, the first steps in the
129
+ // same order, how many more places the full list holds, and the "Copy next steps" text.
130
+
131
+
132
+
133
+
134
+
135
+
136
+
137
+
138
+
139
+
140
+
141
+
142
+
63
143
 
64
144
 
65
145
 
66
146
 
67
147
 
68
148
 
69
-
70
-
149
+
150
+
151
+
152
+
153
+
154
+
155
+
156
+
157
+
158
+
159
+
160
+
161
+
162
+
71
163
 
164
+
72
165
 
73
166
 
74
167
 
75
168
 
76
169
 
170
+
77
171
 
78
172
 
79
- // Fields surfaced as first-class report columns. Everything else on a finding
80
- // becomes its `evidence` payload (sorted for stable key order).
81
- const CORE_KEYS = new Set([
82
- 'id', 'type', 'name', 'impactBand', 'stageId', 'metric', 'value',
83
- 'recommendation', 'detectorVersion', 'confidence', 'validationRequired', 'docAnchor', 'impactEstimate',
84
- 'actionLabel',
85
- ]);
86
-
87
- // Internal-only fields with no meaning to a human reading this report: never surfaced as a core
88
- // column, and also excluded from the generic evidence dump (unlike stageIds, which IS actionable
89
- // to a reader). `planNodeIds` is view-layer plan-graph node ids (Plan Advisor detectors, see
90
- // plan-graph-model.ts): on a real log it can carry a hundred-plus ids, which would otherwise print
91
- // as one unreadable `- planNodeIds: [...]` line and bloat the report for no reader benefit.
92
- const NON_EVIDENCE_KEYS = new Set(['planNodeIds']);
173
+ // Each finding type's public evidence fields: exactly the keys of its FindingEvidenceMap entry
174
+ // (finding-types.ts), checked both ways at compile time. A field a detector adds for another core
175
+ // module (stageShape's totalCores, utilization's unrounded fraction) is left off both, so it never
176
+ // reaches the report; adding, renaming or dropping a key here changes the report contract.
177
+ const EVIDENCE_KEYS = {
178
+ skew: [],
179
+ stageShape: ['rule'],
180
+ shuffle: [],
181
+ partitionSizing: ['rule'],
182
+ spill: ['spillMagnitude'],
183
+ gc: ['direction'],
184
+ slowHost: ['variant', 'host', 'hostTaskShare', 'hostMeanMs', 'dimension', 'executorId', 'execMaxValue'],
185
+ stageSlowness: [],
186
+ stageFailed: ['variant', 'numTasks', 'memoryBytesSpilled', 'failedTaskDetails'],
187
+ failures: ['failedTasks', 'dominantReason', 'dominantError', 'failureGroups', 'otherFailedTasks'],
188
+ straggler: ['unit', 'speculativeTasks', 'stragglerCount'],
189
+ speculationWaste: [],
190
+ retryWaste: ['numTasks', 'memoryBytesSpilled', 'retriedTaskDetails'],
191
+ tinyTask: [],
192
+ incompleteRun: [],
193
+ coldStart: [],
194
+ utilization: ['cpuUtilizationPct'],
195
+ memoryUtilization: ['variant', 'rule', 'executorId', 'heap', 'dataUnavailable'],
196
+ cacheUtilization: [
197
+ 'variant', 'rddId', 'rddName', 'memorySize', 'diskSize', 'numCachedPartitions', 'numPartitions', 'dataUnavailable',
198
+ ],
199
+ coreLocality: ['nonLocalTaskCount'],
200
+ autoscalingChurn: ['shortLivedExecutorCount'],
201
+ cachingOpportunity: ['variant', 'relation', 'format', 'relations', 'operator', 'executionIds', 'totalReadBytes'],
202
+ jobFailureRate: ['failedJobs', 'totalJobs', 'failedTasks', 'totalTasks', 'avgJobDurationMs', 'taskFailureRate'],
203
+ configAudit: ['property'],
204
+ duplicatePlanSubtree: [
205
+ 'executionId', 'stageIds', 'stageShares', 'occurrencesIdentical', 'rootName', 'subtreeSize', 'sampleRelation',
206
+ 'groupIndex',
207
+ ],
208
+ smallFiles: ['executionId', 'stageIds', 'fileCount', 'direction', 'nodeName'],
209
+ underBroadcast: ['executionId', 'stageIds', 'largerSideBytes'],
210
+ overBroadcast: ['executionId', 'stageIds'],
211
+ } ;
93
212
 
94
- function findingRow(f ) {
213
+ // The other direction: an evidence field EVIDENCE_KEYS doesn't list fails here.
214
+
215
+
216
+
217
+
218
+
219
+
220
+ // The finding's evidence, keys sorted for a stable order. An undefined field (spill's
221
+ // spillMagnitude without a magnitude) is absent, as in the JSON.
222
+ function projectEvidence(f ) {
223
+ const fields = f ;
95
224
  const evidence = {};
96
- for (const k of Object.keys(f).sort()) {
97
- // An undefined field (spill's spillMagnitude without a magnitude) is absent, as in the JSON.
98
- const v = (f )[k];
99
- if (!CORE_KEYS.has(k) && !NON_EVIDENCE_KEYS.has(k) && v !== undefined) evidence[k] = v;
225
+ for (const k of [...EVIDENCE_KEYS[f.type]].sort()) {
226
+ if (fields[k] !== undefined) evidence[k] = fields[k];
100
227
  }
101
- const row = {
228
+ return evidence;
229
+ }
230
+
231
+ function findingRow(f ) {
232
+ // Cast: projectEvidence's keys come from EVIDENCE_KEYS[f.type], so `evidence` is that type's
233
+ // FindingEvidenceMap entry, which TypeScript can't correlate with `type` on its own.
234
+ const row = {
102
235
  id: f.id ?? null,
103
236
  type: f.type,
104
- name: titleCase(FINDING_NAMES[f.type] ?? f.type),
237
+ name: titleCase(findingName(f.type)),
105
238
  tag: typeTag(f.type),
106
239
  impactBand: f.impactBand,
107
240
  stageId: f.stageId ?? null,
108
241
  metric: f.metric ?? null,
109
242
  value: f.value ?? null,
243
+ ...(f.valueText != null ? { valueText: f.valueText } : {}),
110
244
  recommendation: f.recommendation ?? null,
111
245
  detectorVersion: f.detectorVersion ?? 1,
112
- evidence,
113
- // Deliberate simplification vs the view layer's REGISTRY fallback: no widget registry here, and
114
- // falling back to the finding's own `type` is fine since coreFindingActionLabel already covers
115
- // every emitted type; only obscure/future sub-variants hit this fallback.
116
- actionLabel: coreFindingActionLabel(f) ?? f.type,
117
- };
246
+ evidence: projectEvidence(f),
247
+ actionLabel: findingActionLabel(f),
248
+ } ;
118
249
  // Threshold/confidence provenance, only when the detector emitted it.
119
250
  if (f.confidence != null) row.confidence = f.confidence;
120
251
  if (f.validationRequired != null) row.validationRequired = f.validationRequired;
121
252
  if (f.docAnchor != null) row.docAnchor = f.docAnchor;
122
253
  if (f.impactEstimate != null) row.impactEstimate = f.impactEstimate;
254
+ const figure = impactEstimateFigure(f.impactEstimate);
255
+ if (figure) {
256
+ row.impact = figure.text;
257
+ row.impactMeaning = figure.meaning;
258
+ }
259
+ if (f.tunedThresholds != null) row.tunedThresholds = f.tunedThresholds;
123
260
  return row;
124
261
  }
125
262
 
@@ -157,7 +294,7 @@ function buildRecommendations(
157
294
  const base = {
158
295
  type: group.type,
159
296
  tag: typeTag(group.type),
160
- actionLabel: coreFindingActionLabel(representative) ?? representative.type,
297
+ actionLabel: findingActionLabel(representative),
161
298
  findingCount: group.findingCount,
162
299
  findingIds,
163
300
  };
@@ -170,15 +307,19 @@ function buildRecommendations(
170
307
  // A point estimate, not a range: matches FixTheseFirst.tsx's trailingStat for time groups,
171
308
  // which prints the same figure twice rather than the finding-level spread computeStageUnionMs collapsed.
172
309
  impact: formatWallClockRange(group.recoverableMsHigh, group.recoverableMsHigh),
310
+ impactMeaning: 'of run time',
173
311
  };
174
312
  }
175
313
  if (group.kind === 'resource') {
314
+ const text = formatRawWaste({ value: group.total, unit: group.unit });
315
+ const shown = readsAsZero(text) ? null : text;
176
316
  return {
177
317
  ...base,
178
318
  kind: 'resource',
179
319
  unit: group.unit,
180
320
  total: group.total,
181
- impact: formatRawWaste({ value: group.total, unit: group.unit }),
321
+ impact: shown,
322
+ impactMeaning: shown ? rawWasteMeaning(group.unit) : null,
182
323
  };
183
324
  }
184
325
  return {
@@ -188,51 +329,138 @@ function buildRecommendations(
188
329
  // No single quantifiable figure for a count group; the impact-band tally
189
330
  // (byImpactBand above) is the payload instead.
190
331
  impact: null,
332
+ impactMeaning: null,
191
333
  };
192
334
  });
193
335
  }
194
336
 
195
- // Detector types that fired zero findings, so a flat report can state "checked and clean". Differs
196
- // from the dashboard's Alerts.tsx "Clean checks", which excludes coreLocality (the one remaining
197
- // always-mounted reference widget, shown elsewhere); a flat report has no such separate surface,
198
- // so this includes it too when it has zero findings.
199
- function buildCleanChecks(findings ) {
200
- const firedTypes = new Set(findings.map((f) => f.type));
201
- const seen = new Set ();
202
- const entries = [];
203
- // detectorCatalog() can list the same type more than once (configAudit has 4 entries); dedupe by
204
- // type, keeping first, so a type with sibling entries contributes exactly one clean-check line.
205
- for (const d of detectorCatalog() ) {
206
- if (firedTypes.has(d.type) || seen.has(d.type)) continue;
207
- seen.add(d.type);
208
- entries.push({ type: d.type, tag: typeTag(d.type), thresholdSummary: getThresholdSummary(d.type) });
337
+ // Finding types with no real finding, split into those that passed and those the log could not
338
+ // run (the rule the dashboard's Clean checks uses, from check-coverage.ts). Differs from
339
+ // Alerts.tsx in one way: the dashboard excludes coreLocality (the one always-mounted reference
340
+ // widget, shown elsewhere); a flat report has no such separate surface, so this includes it too.
341
+ function buildCheckLists(
342
+ findings , stages , thresholds ,
343
+ ) {
344
+ // isRealFinding: a type whose only finding is an evidence caveat (memoryUtilization's
345
+ // dataUnavailable variant) had nothing to check, so it lands in notRunChecks.
346
+ const firedTypes = new Set (findings.filter(isRealFinding).map((f) => f.type));
347
+ const coverage = checkCoverage(stages, findings);
348
+ const cleanChecks = [];
349
+ const notRunChecks = [];
350
+ // One line per emitted finding type (configAudit's four entries give one line;
351
+ // broadcastSizing gives overBroadcast and underBroadcast), the same set the dashboard lists.
352
+ for (const [type, { thresholdSummary }] of Object.entries(detectorInfoByType())) {
353
+ if (firedTypes.has(type)) continue;
354
+ // A tuned check was measured against the tuned criterion, so it says which one.
355
+ const tuned = tunedThresholdsForType(type, thresholds);
356
+ const entry = tuned
357
+ ? { type, tag: typeTag(type), thresholdSummary: getThresholdSummary(type, thresholds), tunedThresholds: tuned }
358
+ : { type, tag: typeTag(type), thresholdSummary };
359
+ const reason = coverage.notRunReason(type);
360
+ if (reason) notRunChecks.push({ ...entry, reason });
361
+ else cleanChecks.push(entry);
209
362
  }
210
- return entries;
363
+ return { cleanChecks, notRunChecks };
364
+ }
365
+
366
+ function countByImpactBand(findings ) {
367
+ const counts = { critical: 0, warning: 0, info: 0 };
368
+ for (const f of findings) if (f.impactBand in counts) counts[f.impactBand ] += 1;
369
+ return counts;
370
+ }
371
+
372
+ function verdictJson(model ) {
373
+ return {
374
+ title: model.title,
375
+ summary: model.summary,
376
+ steps: model.shown.map((step) => {
377
+ const recommendation = stepCopyRecommendation(step, model.outcome);
378
+ return {
379
+ key: step.key,
380
+ stageId: step.stageId,
381
+ type: step.lead.type,
382
+ tag: typeTag(step.lead.type),
383
+ leadFindingId: step.lead.id ?? null,
384
+ actionLabel: findingActionLabel(step.lead),
385
+ recommendation,
386
+ impact: impactFigure(step.lead),
387
+ impactMeaning: savingsMeaning(step.lead),
388
+ relatedTypes: step.related.map((f) => f.type),
389
+ text: stepCopyText(step.lead, recommendation, step.stageId),
390
+ };
391
+ }),
392
+ remainingPlaces: model.remaining,
393
+ copyText: model.copyText,
394
+ };
211
395
  }
212
396
 
213
397
  // Keyed by appModel object identity: mcp-tools.ts caches one fixed appModel per runId (never
214
398
  // mutated), so re-running analyze()/auditConfig() reproduces the same catalog. getFindingEvidence
215
399
  // calls buildEvidenceReport once per drill-down; without this, N lookups meant N detector re-runs.
216
400
  // A WeakMap needs no invalidation: once mcp-tools.ts evicts the appModel, this entry is collectible.
217
- const jsonCache = new WeakMap ();
401
+ // Each cache is split first by the overrides object the report ran under (one fixed, frozen object
402
+ // per CLI invocation or MCP server process; DEFAULT_THRESHOLDS for the specification's).
403
+
404
+ const DEFAULT_THRESHOLDS = {};
405
+ const jsonCache = new WeakMap();
406
+ // The redacted report, keyed by the unredacted appModel it was built from.
407
+ const redactedJsonCache = new WeakMap();
218
408
 
219
- function buildJson(appModel ) {
220
- const cached = jsonCache.get(appModel);
221
- if (cached) return cached;
222
- const { app, stages, executors, sql, jobs, runAggregates, evidenceAvailability } = appModel;
409
+ function cacheFor(cache , thresholds ) {
410
+ const key = thresholds ?? DEFAULT_THRESHOLDS;
411
+ let byModel = cache.get(key);
412
+ if (!byModel) {
413
+ byModel = new WeakMap();
414
+ cache.set(key, byModel);
415
+ }
416
+ return byModel;
417
+ }
418
+
419
+ function runFindings(appModel , thresholds ) {
420
+ const { app, stages, executors, sql, jobs, runAggregates } = appModel;
223
421
  const catalog = analyze(
224
422
  app, stages, executors?.added ?? [], executors?.removed ?? [],
225
423
  jobs ?? new Map(), sql ?? new Map(),
226
- runAggregates ?? null,
424
+ runAggregates ?? null, { thresholds },
227
425
  );
228
- const config = auditConfig(app);
426
+ return { catalog, config: auditConfig(app) };
427
+ }
428
+
429
+ // Redacts the model and findings before the report derives any text from them, the same order
430
+ // the HTML export uses: the verdict truncates Spark's failure reason, and redacting that
431
+ // truncated copy afterwards would miss an identifier the cut left as a fragment.
432
+ function buildRedactedJson(appModel , thresholds ) {
433
+ const cache = cacheFor(redactedJsonCache, thresholds);
434
+ const cached = cache.get(appModel);
435
+ if (cached) return cached;
436
+ const { catalog, config } = runFindings(appModel, thresholds);
437
+ const run = redactRunModel(appModel, catalog, config);
438
+ // redactReport stays as a last pass: idempotent over pseudonyms, and it covers the report's own
439
+ // structured fields (summary.app.id) the same way it always has.
440
+ const result = redactReport(buildJson(run.appModel, thresholds, { catalog: run.catalog, config: run.configFindings }));
441
+ cache.set(appModel, result);
442
+ return result;
443
+ }
444
+
445
+ function buildJson(
446
+ appModel , thresholds , findings ,
447
+ ) {
448
+ const cache = cacheFor(jsonCache, thresholds);
449
+ const cached = cache.get(appModel);
450
+ if (cached) return cached;
451
+ const { app, stages, executors, sql, jobs, evidenceAvailability } = appModel;
452
+ const { catalog, config } = findings ?? runFindings(appModel, thresholds);
229
453
  const allFindings = [...catalog, ...config];
230
454
  const rows = sortFindings(allFindings.map(findingRow));
231
455
  const recommendations = buildRecommendations(allFindings, stages ?? new Map());
232
- const cleanChecks = buildCleanChecks(allFindings);
233
-
234
- const impactBandCounts = { critical: 0, warning: 0, info: 0 };
235
- for (const r of rows) if (r.impactBand in impactBandCounts) impactBandCounts[r.impactBand] += 1;
456
+ const { cleanChecks, notRunChecks } = buildCheckLists(allFindings, stages ?? new Map(), thresholds);
457
+ const tuned = tunedDetectors(thresholds);
458
+ const actionable = allFindings.filter(isEligible);
459
+ const fullModel = {
460
+ ...appModel, stages: stages ?? new Map(), jobs: jobs ?? new Map(), executors: executors ?? { added: [], removed: [] },
461
+ };
462
+ const runVerdict = buildRunVerdict(fullModel, allFindings);
463
+ const runOutcome = runVerdict.outcome;
236
464
 
237
465
  const result = {
238
466
  schemaVersion: EVIDENCE_SCHEMA_VERSION,
@@ -248,17 +476,30 @@ function buildJson(appModel ) {
248
476
  jobCount: jobs?.size ?? 0,
249
477
  sqlExecutionCount: sql?.size ?? 0,
250
478
  findingCount: rows.length,
251
- impactBandCounts,
479
+ impactBandCounts: countByImpactBand(rows),
480
+ actionableFindingCount: actionable.length,
481
+ actionableImpactBandCounts: countByImpactBand(actionable),
482
+ clean: isCleanRun({ jobs: jobs ?? new Map(), stages: stages ?? new Map() }, allFindings),
483
+ outcome: {
484
+ failedJobs: runOutcome.failedJobs,
485
+ totalJobs: runOutcome.totalJobs,
486
+ failureReason: runOutcome.reason,
487
+ failureReasonStageId: runOutcome.reason != null ? runOutcome.reasonStageId : null,
488
+ },
489
+ runShape: computeRunShape(fullModel),
490
+ ...(tuned ? { tunedThresholds: tuned } : {}),
252
491
  },
492
+ verdict: verdictJson(runVerdict),
253
493
  evidenceAvailability: evidenceAvailability ?? null,
254
494
  // Detector metadata so the threshold set that produced each finding travels with the evidence.
255
495
  // Order follows DETECTORS (stable) => byte-stable serialization.
256
- detectors: detectorCatalog(),
496
+ detectors: tunedDetectorCatalog(thresholds),
257
497
  findings: rows,
258
498
  recommendations,
259
499
  cleanChecks,
500
+ notRunChecks,
260
501
  };
261
- jsonCache.set(appModel, result);
502
+ cache.set(appModel, result);
262
503
  return result;
263
504
  }
264
505
 
@@ -285,43 +526,92 @@ function renderFailureGroups(groups ) {
285
526
  return lines;
286
527
  }
287
528
 
288
- function formatWallClockRange(low , high ) {
289
- const fmtMs = (ms ) => (ms === 0 ? '0s' : formatDuration(ms));
290
- return low === high ? `Estimated ${fmtMs(high)}` : `Estimated ${fmtMs(low)}-${fmtMs(high)}`;
529
+ // The finding's "Potential savings" figure as the dashboard shows it (impactEstimateFigure: the
530
+ // range, or the raw waste only without a range, nothing for a zero or informational estimate),
531
+ // followed by what it counts. `basis: 'informational'` findings carry nothing to print.
532
+ function renderImpactEstimate(estimate ) {
533
+ const figure = impactEstimateFigure(estimate);
534
+ if (!figure) return null;
535
+ return `${figure.text}${figure.meaning ? ` ${figure.meaning}` : ''} (estimateMethod: ${estimate.estimateMethod})`;
291
536
  }
292
537
 
293
- function formatRawWaste(rawWaste ) {
294
- const rounded = Math.round(rawWaste.value * 10) / 10;
295
- switch (rawWaste.unit) {
296
- case 'bytes': return formatBytes(rawWaste.value);
297
- case 'ms': return formatDuration(rawWaste.value);
298
- case 'mbSeconds': return `${rounded} MB-s`;
299
- case 'coreHours': return `${rounded.toFixed(1)} core-h`;
300
- case 'coreMs': return `${rounded} core-ms`;
301
- default: return String(rawWaste.value);
538
+ // The run's job results in one line, worded like the dashboard verdict: failures first, with
539
+ // Spark's recorded reason; null when the log records no ended job.
540
+ function renderOutcome(outcome , incomplete ) {
541
+ const { failedJobs, totalJobs, failureReason, failureReasonStageId } = outcome;
542
+ if (totalJobs === 0) return null;
543
+ if (failedJobs > 0) {
544
+ const failed = failedJobs < totalJobs
545
+ ? `${failedJobs} of ${totalJobs} jobs failed.`
546
+ : totalJobs === 1 ? 'The run\'s one job failed.' : `All ${totalJobs} jobs failed.`;
547
+ if (!failureReason) return failed;
548
+ const where = failureReasonStageId != null ? ` (stage ${failureReasonStageId})` : '';
549
+ return `${failed} Spark's recorded reason${where}: ${failureReason}`;
302
550
  }
551
+ const succeeded = totalJobs === 1 ? 'Its one job succeeded.' : `All ${totalJobs} jobs succeeded.`;
552
+ return incomplete ? `${succeeded} The log has no end-of-run record, so jobs still running when it stops are not counted.` : succeeded;
303
553
  }
304
554
 
305
- // `basis: 'informational'` findings carry no wallClock/rawWaste at all, so
306
- // there's nothing quantifiable to print; the caller skips the line entirely.
307
- function renderImpactEstimate(estimate ) {
308
- const rangeText = estimate.wallClock ? formatWallClockRange(estimate.wallClock.low, estimate.wallClock.high) : null;
309
- const wasteText = estimate.rawWaste ? formatRawWaste(estimate.rawWaste) : null;
310
- if (!rangeText && !wasteText) return null;
311
- const parts = [rangeText, wasteText].filter((p) => p != null).join(' · ');
312
- return `${parts} (estimateMethod: ${estimate.estimateMethod})`;
555
+ // The Scorecard, ETL phases and core-usage figures, each worded to say what it measures, since
556
+ // "efficiency" and "unused core time" are different shares. A figure the dashboard cannot show is
557
+ // left out.
558
+ function renderRunShape(shape ) {
559
+ const lines = [];
560
+ if (shape.wallClockMs != null) lines.push(`- Wall-clock: ${formatDuration(shape.wallClockMs)}`);
561
+ if (shape.efficiencyPct != null) lines.push(`- Efficiency: ${shape.efficiencyPct}% (share of the run with a stage running)`);
562
+ if (shape.unusedCoreTimePct != null) {
563
+ lines.push(`- Unused core time: ${shape.unusedCoreTimePct}% (driver idle plus executor slack, as a share of available core time)`);
564
+ }
565
+ if (shape.peakBusyCores != null) lines.push(`- Peak busy cores: ${formatCores(shape.peakBusyCores)}`);
566
+ if (shape.etlPhasesMs) {
567
+ const { extract, transform, load } = shape.etlPhasesMs;
568
+ lines.push(`- ETL phases (summed stage time): extract ${formatDuration(extract)}, transform ${formatDuration(transform)}, load ${formatDuration(load)}`);
569
+ }
570
+ return lines;
571
+ }
572
+
573
+ // The dashboard verdict card as text: title, summary, then the numbered steps worded as its
574
+ // "Copy next steps" checklist, each followed by the other finding types flagged at that place.
575
+ function renderVerdict(verdict ) {
576
+ const lines = ['## Verdict', '', verdict.title];
577
+ if (verdict.summary.length > 0) lines.push('', verdict.summary.join(' '));
578
+ if (verdict.steps.length > 0) {
579
+ lines.push('');
580
+ verdict.steps.forEach((step, i) => {
581
+ lines.push(`${i + 1}. [${step.tag}] ${step.text}`);
582
+ if (step.relatedTypes.length > 0) {
583
+ const related = step.relatedTypes.map(findingName).join(', ');
584
+ lines.push(` - Also flagged here: ${related}. These often share this cause, so the same fix may clear them too.`);
585
+ }
586
+ });
587
+ if (verdict.remainingPlaces > 0) {
588
+ const places = `${verdict.remainingPlaces} more place${verdict.remainingPlaces === 1 ? '' : 's'}`;
589
+ lines.push('', `${places} to look at in the Findings section below.`);
590
+ }
591
+ }
592
+ lines.push('');
593
+ return lines;
313
594
  }
314
595
 
315
- function renderMarkdown(json ) {
316
- const { summary, findings, evidenceAvailability, detectors, recommendations, cleanChecks } = json;
596
+ // `incomplete` comes from the unfiltered findings, since a findingsFilter can drop the incompleteRun row.
597
+ function renderMarkdown(json , incomplete ) {
598
+ const { summary, verdict, findings, evidenceAvailability, detectors, recommendations, cleanChecks, notRunChecks } = json;
317
599
  const lines = [];
318
600
  lines.push('# Spark run evidence report');
319
601
  lines.push('');
320
602
  lines.push(`- Application: ${summary.app.name ?? '(unknown)'} (${summary.app.id ?? 'n/a'})`);
321
603
  lines.push(`- Spark version: ${summary.app.sparkVersion ?? 'n/a'}`);
604
+ if (summary.tunedThresholds) lines.push(`- Tuned thresholds: ${tunedRunNote(summary.tunedThresholds)}`);
322
605
  lines.push(`- Stages: ${summary.stageCount} · Jobs: ${summary.jobCount} · SQL executions: ${summary.sqlExecutionCount}`);
323
606
  lines.push(`- Findings: ${summary.findingCount} (critical ${summary.impactBandCounts.critical}, warning ${summary.impactBandCounts.warning}, info ${summary.impactBandCounts.info})`);
607
+ const actionableCounts = summary.actionableImpactBandCounts;
608
+ lines.push(`- Findings to act on: ${summary.actionableFindingCount} (critical ${actionableCounts.critical}, warning ${actionableCounts.warning}, info ${actionableCounts.info})`);
609
+ const outcomeLine = renderOutcome(summary.outcome, incomplete);
610
+ if (outcomeLine) lines.push(`- Outcome: ${outcomeLine}`);
611
+ if (summary.clean) lines.push('- Clean run: no findings, no failed jobs, and every check could run.');
612
+ lines.push(...renderRunShape(summary.runShape));
324
613
  lines.push('');
614
+ lines.push(...renderVerdict(verdict));
325
615
  if (recommendations.length > 0) {
326
616
  lines.push(`## Fix these first (${recommendations.length})`);
327
617
  lines.push('');
@@ -329,8 +619,8 @@ function renderMarkdown(json ) {
329
619
  lines.push(`${i + 1}. [${r.tag}] ${r.actionLabel}`);
330
620
  const detail = r.kind === 'count'
331
621
  ? Object.entries(r.byImpactBand ?? {}).map(([impactBand, count]) => `${count} ${impactBand}`).join(', ')
332
- : r.impact;
333
- lines.push(` - ${detail} · ×${r.findingCount} finding(s)`);
622
+ : r.impact && `${r.impact}${r.impactMeaning ? ` ${r.impactMeaning}` : ''}`;
623
+ lines.push(` - ${detail ? `${detail} · ` : ''}×${r.findingCount} finding(s)`);
334
624
  });
335
625
  lines.push('');
336
626
  }
@@ -340,12 +630,15 @@ function renderMarkdown(json ) {
340
630
  const where = r.stageId != null ? ` (stage ${r.stageId})` : '';
341
631
  lines.push(`### ${r.name} · ${r.impactBand}${where}`);
342
632
  lines.push(`- action: ${r.actionLabel}`);
343
- if (r.metric != null) lines.push(`- ${r.metric}: ${r.value}`);
633
+ if (r.metric != null) lines.push(`- ${r.metric}: ${r.valueText ?? r.value}`);
344
634
  if (r.recommendation) lines.push(`- ${r.recommendation}`);
345
635
  if (r.confidence) lines.push(`- confidence: ${r.confidence}`);
346
636
  if (r.validationRequired) lines.push(`- validation: ${r.validationRequired}`);
637
+ if (r.tunedThresholds) lines.push(`- tuned thresholds: ${describeTunedThresholds(r.tunedThresholds)}`);
347
638
  const impactText = r.impactEstimate ? renderImpactEstimate(r.impactEstimate) : null;
348
639
  if (impactText) lines.push(`- impact: ${impactText}`);
640
+ const provenance = r.impactEstimate ? estimateProvenance(r) : null;
641
+ if (provenance) lines.push(`- estimate: ${provenance}`);
349
642
  lines.push(`- detector version: ${r.detectorVersion}`);
350
643
  // Evidence payload (sorted for stable order) so two rows differing only by evidence (two
351
644
  // smallFiles by direction, two partitionSizing by rule) render distinctly.
@@ -372,7 +665,18 @@ function renderMarkdown(json ) {
372
665
  lines.push('## Detectors');
373
666
  lines.push('');
374
667
  for (const d of detectors) {
375
- lines.push(`- ${d.type} (v${d.version}, ${d.scope}), thresholds: ${JSON.stringify(d.thresholds)}`);
668
+ const tuned = d.tunedThresholds ? ` (tuned: ${describeTunedThresholds(d.tunedThresholds)})` : '';
669
+ lines.push(`- ${d.type} (v${d.version}, ${d.scope}), thresholds: ${JSON.stringify(d.thresholds)}${tuned}`);
670
+ }
671
+ lines.push('');
672
+ }
673
+ if (notRunChecks.length > 0) {
674
+ lines.push(`## Not checked on this log (${notRunChecks.length})`);
675
+ lines.push('');
676
+ lines.push('The log lacked the data these checks need, so they neither passed nor failed.');
677
+ lines.push('');
678
+ for (const c of notRunChecks) {
679
+ lines.push(`- [${c.tag}] ${c.type}: ${c.reason}`);
376
680
  }
377
681
  lines.push('');
378
682
  }
@@ -380,7 +684,8 @@ function renderMarkdown(json ) {
380
684
  lines.push(`## Clean checks (${cleanChecks.length})`);
381
685
  lines.push('');
382
686
  for (const c of cleanChecks) {
383
- lines.push(`- [${c.tag}] ${c.type}: ${c.thresholdSummary}`);
687
+ const tuned = c.tunedThresholds ? ` (tuned: ${describeTunedThresholds(c.tunedThresholds)})` : '';
688
+ lines.push(`- [${c.tag}] ${c.type}: ${c.thresholdSummary}${tuned}`);
384
689
  }
385
690
  lines.push('');
386
691
  }
@@ -396,7 +701,10 @@ function renderMarkdown(json ) {
396
701
  // CLI/MCP-facing filter over FindingRow, delegating to the shared core predicate that also backs
397
702
  // the dashboard's finding-filter.
398
703
  function matchesFindingsFilter(row , filter ) {
399
- return matchesFindingFilterCriteria(row, filter);
704
+ // A sql-scope finding carries its stages in evidence.stageIds, not a stageId column: it matches
705
+ // the one stage it touches, as the dashboard's Stage details lists it.
706
+ const stageIds = 'stageIds' in row.evidence ? row.evidence.stageIds : null;
707
+ return matchesFindingFilterCriteria({ ...row, stageId: singleStageId({ stageId: row.stageId, stageIds }) }, filter);
400
708
  }
401
709
 
402
710
  /** Build a FindingsFilter from the three optional CLI/MCP filter dimensions, or undefined when
@@ -411,20 +719,21 @@ export function toFindingsFilter(
411
719
  * Build a portable evidence report from an appModel.
412
720
  * @param opts redact=true pseudonymizes app ids / hosts; markdown=false skips the Markdown string;
413
721
  * findingsFilter narrows json.findings (and the Markdown Findings section) only, summary,
414
- * recommendations, and cleanChecks stay computed from the full set, so a narrow filter never
415
- * hides that other checks passed or other fixes exist.
722
+ * recommendations, cleanChecks and notRunChecks stay computed from the full set, so a narrow filter never
723
+ * hides that other checks passed or other fixes exist. thresholds runs the detectors with a user's
724
+ * validated overrides (CLI/MCP only) and labels whatever they changed.
416
725
  */
417
726
  export function buildEvidenceReport(
418
727
  appModel ,
419
- { redact = false, markdown: computeMarkdown = true, findingsFilter }
420
-
728
+ { redact = false, markdown: computeMarkdown = true, findingsFilter, thresholds }
729
+
421
730
  = {},
422
731
  ) {
423
- let json = buildJson(appModel);
424
- if (redact) json = redactReport(json);
732
+ let json = redact ? buildRedactedJson(appModel, thresholds) : buildJson(appModel, thresholds);
733
+ const incomplete = json.findings.some((row) => row.type === 'incompleteRun');
425
734
  // Filter after redact, not before: redaction only replaces string values on surviving rows,
426
735
  // never adds/removes rows, so the two orderings produce identical final content.
427
736
  if (findingsFilter) json = { ...json, findings: json.findings.filter((row) => matchesFindingsFilter(row, findingsFilter)) };
428
- const markdown = computeMarkdown ? renderMarkdown(json) : '';
737
+ const markdown = computeMarkdown ? renderMarkdown(json, incomplete) : '';
429
738
  return { markdown, json };
430
739
  }