sparkforensics-mcp 0.2.4 → 0.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (71) hide show
  1. package/LICENSE +21 -0
  2. package/README.md +7 -1
  3. package/bin/sparkforensics-mcp.mjs +41 -10
  4. package/package.json +1 -1
  5. package/vendor-core/analyzer.js +156 -48
  6. package/vendor-core/check-coverage.js +88 -0
  7. package/vendor-core/cli/budgets.js +31 -18
  8. package/vendor-core/cli/collect-run.js +76 -31
  9. package/vendor-core/cli/native-zstd.js +2 -2
  10. package/vendor-core/cli/threshold-config.js +28 -0
  11. package/vendor-core/comparison-verdict.js +177 -0
  12. package/vendor-core/core-source-hash.txt +1 -0
  13. package/vendor-core/core-usage-locality.js +56 -2
  14. package/vendor-core/detector-docs.js +58 -0
  15. package/vendor-core/detectors.js +933 -459
  16. package/vendor-core/docs-config.js +0 -36
  17. package/vendor-core/docs-content/chapters/nav-index.json +31 -0
  18. package/vendor-core/docs-content/detection/cstor.md +9 -0
  19. package/vendor-core/docs-content/detection/fail.md +6 -2
  20. package/vendor-core/docs-site-config.js +3 -0
  21. package/vendor-core/event-handlers.js +191 -6
  22. package/vendor-core/event-schemas.js +29 -0
  23. package/vendor-core/evidence-report.js +440 -112
  24. package/vendor-core/export-data.js +79 -6
  25. package/vendor-core/finding-action-label.js +9 -88
  26. package/vendor-core/finding-filter-predicate.js +9 -0
  27. package/vendor-core/finding-generic-recommendation.js +6 -104
  28. package/vendor-core/finding-names.js +21 -45
  29. package/vendor-core/finding-presentation.js +333 -0
  30. package/vendor-core/finding-tag-help.js +110 -0
  31. package/vendor-core/finding-types.js +361 -0
  32. package/vendor-core/findings-of-type.js +11 -0
  33. package/vendor-core/format-utils.js +92 -27
  34. package/vendor-core/html-export.js +51 -0
  35. package/vendor-core/impact-band.js +21 -8
  36. package/vendor-core/impact-estimator.js +8 -521
  37. package/vendor-core/impact-format.js +114 -0
  38. package/vendor-core/impact-model.js +175 -0
  39. package/vendor-core/ingest.js +2 -0
  40. package/vendor-core/intervals.js +13 -0
  41. package/vendor-core/list-runs.js +2 -3
  42. package/vendor-core/load-vendored.js +70 -5
  43. package/vendor-core/mcp-server-factory.js +14 -10
  44. package/vendor-core/mcp-tools.js +105 -45
  45. package/vendor-core/model-assembler.js +12 -0
  46. package/vendor-core/occupancy.js +1 -1
  47. package/vendor-core/parser-worker.js +22 -5
  48. package/vendor-core/plan-graph-model.js +3 -2
  49. package/vendor-core/plan-node-detail.js +1 -1
  50. package/vendor-core/recommendation-rollup.js +63 -3
  51. package/vendor-core/redact.js +68 -28
  52. package/vendor-core/run-comparison.js +40 -7
  53. package/vendor-core/run-interpretation.js +290 -0
  54. package/vendor-core/run-outcome.js +74 -0
  55. package/vendor-core/run-payload.js +17 -0
  56. package/vendor-core/run-shape.js +40 -0
  57. package/vendor-core/run-verdict.js +353 -0
  58. package/vendor-core/scaling-sim.js +4 -5
  59. package/vendor-core/scorecard-estimates.js +62 -0
  60. package/vendor-core/shs-fetch.js +175 -65
  61. package/vendor-core/shs-load.js +1 -1
  62. package/vendor-core/sql-stages.js +11 -0
  63. package/vendor-core/stage-quantiles.js +14 -0
  64. package/vendor-core/task-failure.js +151 -0
  65. package/vendor-core/threshold-overrides.js +160 -0
  66. package/vendor-core/threshold-summary.js +11 -33
  67. package/vendor-core/types.js +6 -42
  68. package/vendor-core/vendor/fflate.js +1 -1
  69. package/vendor-core/wall-clock.js +1 -12
  70. package/vendor-core/wasted-core-hours.js +2 -2
  71. package/vendor-core/zip-archive.js +167 -0
@@ -2,19 +2,33 @@
2
2
  // then serializes a run summary + findings into a deterministic, byte-stable JSON + Markdown.
3
3
  // Raw task records are never included (privacy baseline); redaction is opt-in via { redact: true }.
4
4
  import { analyze, auditConfig } from './analyzer.js';
5
- import { detectorCatalog } from './detectors.js';
6
- import { typeTag, formatBytes, formatDuration, IMPACT_BAND_ORDER } from './format-utils.js';
7
- import { FINDING_NAMES, titleCase } from './finding-names.js';
8
- import { redactReport } from './redact.js';
9
- import { coreFindingActionLabel } from './finding-action-label.js';
10
- import { matchesFindingFilterCriteria } from './finding-filter-predicate.js';
11
- import { buildRecommendationRollup, isEligible, rankFindings, } from './recommendation-rollup.js';
5
+
6
+ import {
7
+ describeTunedThresholds, tunedDetectorCatalog, tunedDetectors, tunedRunNote, tunedThresholdsForType,
8
+ } from './threshold-overrides.js';
12
9
  import { getThresholdSummary } from './threshold-summary.js';
10
+ import {
11
+ typeTag, formatBytes, formatCores, formatDuration, formatRawWaste, formatWallClockRange, IMPACT_BAND_ORDER, readsAsZero,
12
+ } from './format-utils.js';
13
+ import { findingName, titleCase } from './finding-names.js';
14
+ import { redactReport, redactRunModel } from './redact.js';
15
+ import { formatTaskFailureHeadline, } from './task-failure.js';
16
+ import { findingActionLabel } from './finding-action-label.js';
17
+ import { matchesFindingFilterCriteria, singleStageId } from './finding-filter-predicate.js';
18
+ import { buildRecommendationRollup, isEligible, isRealFinding, rankFindings, } from './recommendation-rollup.js';
19
+ import { checkCoverage, isCleanRun } from './check-coverage.js';
20
+ import { buildRunVerdict, stepCopyRecommendation, stepCopyText, } from './run-verdict.js';
21
+ import {
22
+ estimateProvenance, impactEstimateFigure, impactFigure, rawWasteMeaning, savingsMeaning,
23
+ } from './impact-format.js';
24
+ import { computeRunShape, } from './run-shape.js';
25
+ import { detectorInfoByType } from './detector-docs.js';
13
26
 
14
-
27
+
28
+
15
29
 
16
30
 
17
- export const EVIDENCE_SCHEMA_VERSION = 3;
31
+ export const EVIDENCE_SCHEMA_VERSION = 5;
18
32
 
19
33
  // findingRow always sets id/metric/value/recommendation via `?? null` (never omits the key), and
20
34
  // buildJson does the same for evidenceAvailability and summary.app.{id,name,sparkVersion}: these
@@ -22,20 +36,34 @@ export const EVIDENCE_SCHEMA_VERSION = 3;
22
36
  // "unknown version" sentinel). Kept as `?? null`, not `?? undefined`: JSON.stringify drops
23
37
  // undefined keys but keeps null, so undefined would silently strip these from the report.
24
38
  //
25
- // `value` is number|string|null: stageFailed and the configAudit entries put text in Finding.value
26
- // instead of a magnitude, carried through unchanged.
39
+ // `value` is always a magnitude or null. The text-valued findings (stageFailed's failure reason,
40
+ // configAudit's current setting, incompleteRun's 'missing') carry theirs in `valueText` instead,
41
+ // present only on those rows.
27
42
 
28
-
29
-
30
-
43
+
44
+
45
+
31
46
 
32
47
 
33
48
 
34
49
 
35
50
 
36
51
 
52
+
53
+
54
+
55
+
56
+
57
+
58
+
37
59
 
38
60
 
61
+ // One row per finding, discriminated on `type`: `evidence` is that type's public evidence
62
+ // (FindingEvidenceMap, projected through EVIDENCE_KEYS below), never the finding's other fields.
63
+
64
+
65
+
66
+
39
67
  // The `Fix these first` rollup row: one entry per buildRecommendationRollup
40
68
  // group, so the CLI/MCP/download paths get the same impact-ranked aggregation the dashboard shows.
41
69
 
@@ -50,75 +78,185 @@ export const EVIDENCE_SCHEMA_VERSION = 3;
50
78
 
51
79
 
52
80
 
81
+
82
+
83
+
53
84
 
85
+
54
86
 
55
87
 
56
- // One line per detector `type` that fired zero findings, so a flat report can state "these were
57
- // checked and came back clean" like the dashboard's clean-checks table.
88
+ // One line per detector `type` that fired zero findings and could run, so a flat report can state
89
+ // "these were checked and came back clean" like the dashboard's clean-checks table.
58
90
 
59
91
 
60
92
 
61
93
 
94
+
95
+
96
+
97
+
98
+ // A detector `type` with zero findings that the log lacked the data to run (the dashboard's "Not
99
+ // checked on this log" group), so it is not reported as passed. `reason` says why, naming the
100
+ // setting to turn on where the detector gives one.
101
+
102
+
103
+
104
+
105
+
106
+
107
+ // How the run ended, as far as its jobs say (run-outcome.ts, the same summary the dashboard's
108
+ // verdict leads with). Counts only jobs with an end record; `failureReason` is the first line of
109
+ // Spark's own recorded reason, only when a job failed, and `failureReasonStageId` the stage it came
110
+ // from (null when there is no reason or it came from a job's exception).
111
+ // One verdict step: a place to look (a stage, or an app-level problem), led by its best-ranked
112
+ // finding, with the other finding types flagged there. `text` is the step's line in the
113
+ // dashboard's "Copy next steps" checklist.
114
+
115
+
116
+
117
+
118
+
119
+
120
+
121
+
122
+
123
+
124
+
125
+
126
+
127
+
128
+ // The dashboard's run verdict (run-verdict.ts): title, summary sentences, the first steps in the
129
+ // same order, how many more places the full list holds, and the "Copy next steps" text.
130
+
131
+
132
+
133
+
134
+
135
+
136
+
137
+
138
+
139
+
140
+
141
+
142
+
62
143
 
63
144
 
64
145
 
65
146
 
66
147
 
67
148
 
68
-
69
-
149
+
150
+
151
+
152
+
153
+
154
+
155
+
156
+
157
+
158
+
159
+
160
+
161
+
162
+
70
163
 
164
+
71
165
 
72
166
 
73
167
 
74
168
 
75
169
 
170
+
76
171
 
77
172
 
78
- // Fields surfaced as first-class report columns. Everything else on a finding
79
- // becomes its `evidence` payload (sorted for stable key order).
80
- const CORE_KEYS = new Set([
81
- 'id', 'type', 'name', 'impactBand', 'stageId', 'metric', 'value',
82
- 'recommendation', 'detectorVersion', 'confidence', 'validationRequired', 'docAnchor', 'impactEstimate',
83
- 'actionLabel',
84
- ]);
85
-
86
- // Internal-only fields with no meaning to a human reading this report: never surfaced as a core
87
- // column, and also excluded from the generic evidence dump (unlike stageIds, which IS actionable
88
- // to a reader). `planNodeIds` is view-layer plan-graph node ids (Plan Advisor detectors, see
89
- // plan-graph-model.ts): on a real log it can carry a hundred-plus ids, which would otherwise print
90
- // as one unreadable `- planNodeIds: [...]` line and bloat the report for no reader benefit.
91
- const NON_EVIDENCE_KEYS = new Set(['planNodeIds']);
173
+ // Each finding type's public evidence fields: exactly the keys of its FindingEvidenceMap entry
174
+ // (finding-types.ts), checked both ways at compile time. A field a detector adds for another core
175
+ // module (stageShape's totalCores, utilization's unrounded fraction) is left off both, so it never
176
+ // reaches the report; adding, renaming or dropping a key here changes the report contract.
177
+ const EVIDENCE_KEYS = {
178
+ skew: [],
179
+ stageShape: ['rule'],
180
+ shuffle: [],
181
+ partitionSizing: ['rule'],
182
+ spill: ['spillMagnitude'],
183
+ gc: ['direction'],
184
+ slowHost: ['variant', 'host', 'hostTaskShare', 'hostMeanMs', 'dimension', 'executorId', 'execMaxValue'],
185
+ stageSlowness: [],
186
+ stageFailed: ['variant', 'numTasks', 'memoryBytesSpilled', 'failedTaskDetails'],
187
+ failures: ['failedTasks', 'dominantReason', 'dominantError', 'failureGroups', 'otherFailedTasks'],
188
+ straggler: ['unit', 'speculativeTasks', 'stragglerCount'],
189
+ speculationWaste: [],
190
+ retryWaste: ['numTasks', 'memoryBytesSpilled', 'retriedTaskDetails'],
191
+ tinyTask: [],
192
+ incompleteRun: [],
193
+ coldStart: [],
194
+ utilization: ['cpuUtilizationPct'],
195
+ memoryUtilization: ['variant', 'rule', 'executorId', 'heap', 'dataUnavailable'],
196
+ cacheUtilization: [
197
+ 'variant', 'rddId', 'rddName', 'memorySize', 'diskSize', 'numCachedPartitions', 'numPartitions', 'dataUnavailable',
198
+ ],
199
+ coreLocality: ['nonLocalTaskCount'],
200
+ autoscalingChurn: ['shortLivedExecutorCount'],
201
+ cachingOpportunity: ['variant', 'relation', 'format', 'relations', 'operator', 'executionIds', 'totalReadBytes'],
202
+ jobFailureRate: ['failedJobs', 'totalJobs', 'failedTasks', 'totalTasks', 'avgJobDurationMs', 'taskFailureRate'],
203
+ configAudit: ['property'],
204
+ duplicatePlanSubtree: [
205
+ 'executionId', 'stageIds', 'stageShares', 'occurrencesIdentical', 'rootName', 'subtreeSize', 'sampleRelation',
206
+ 'groupIndex',
207
+ ],
208
+ smallFiles: ['executionId', 'stageIds', 'fileCount', 'direction', 'nodeName'],
209
+ underBroadcast: ['executionId', 'stageIds', 'largerSideBytes'],
210
+ overBroadcast: ['executionId', 'stageIds'],
211
+ } ;
92
212
 
93
- function findingRow(f ) {
213
+ // The other direction: an evidence field EVIDENCE_KEYS doesn't list fails here.
214
+
215
+
216
+
217
+
218
+
219
+
220
+ // The finding's evidence, keys sorted for a stable order. An undefined field (spill's
221
+ // spillMagnitude without a magnitude) is absent, as in the JSON.
222
+ function projectEvidence(f ) {
223
+ const fields = f ;
94
224
  const evidence = {};
95
- for (const k of Object.keys(f).sort()) {
96
- // An undefined field (spill's spillMagnitude without a magnitude) is absent, as in the JSON.
97
- const v = (f )[k];
98
- if (!CORE_KEYS.has(k) && !NON_EVIDENCE_KEYS.has(k) && v !== undefined) evidence[k] = v;
225
+ for (const k of [...EVIDENCE_KEYS[f.type]].sort()) {
226
+ if (fields[k] !== undefined) evidence[k] = fields[k];
99
227
  }
100
- const row = {
228
+ return evidence;
229
+ }
230
+
231
+ function findingRow(f ) {
232
+ // Cast: projectEvidence's keys come from EVIDENCE_KEYS[f.type], so `evidence` is that type's
233
+ // FindingEvidenceMap entry, which TypeScript can't correlate with `type` on its own.
234
+ const row = {
101
235
  id: f.id ?? null,
102
236
  type: f.type,
103
- name: titleCase(FINDING_NAMES[f.type] ?? f.type),
237
+ name: titleCase(findingName(f.type)),
104
238
  tag: typeTag(f.type),
105
239
  impactBand: f.impactBand,
106
240
  stageId: f.stageId ?? null,
107
241
  metric: f.metric ?? null,
108
242
  value: f.value ?? null,
243
+ ...(f.valueText != null ? { valueText: f.valueText } : {}),
109
244
  recommendation: f.recommendation ?? null,
110
245
  detectorVersion: f.detectorVersion ?? 1,
111
- evidence,
112
- // Deliberate simplification vs the view layer's REGISTRY fallback: no widget registry here, and
113
- // falling back to the finding's own `type` is fine since coreFindingActionLabel already covers
114
- // every emitted type; only obscure/future sub-variants hit this fallback.
115
- actionLabel: coreFindingActionLabel(f) ?? f.type,
116
- };
246
+ evidence: projectEvidence(f),
247
+ actionLabel: findingActionLabel(f),
248
+ } ;
117
249
  // Threshold/confidence provenance, only when the detector emitted it.
118
250
  if (f.confidence != null) row.confidence = f.confidence;
119
251
  if (f.validationRequired != null) row.validationRequired = f.validationRequired;
120
252
  if (f.docAnchor != null) row.docAnchor = f.docAnchor;
121
253
  if (f.impactEstimate != null) row.impactEstimate = f.impactEstimate;
254
+ const figure = impactEstimateFigure(f.impactEstimate);
255
+ if (figure) {
256
+ row.impact = figure.text;
257
+ row.impactMeaning = figure.meaning;
258
+ }
259
+ if (f.tunedThresholds != null) row.tunedThresholds = f.tunedThresholds;
122
260
  return row;
123
261
  }
124
262
 
@@ -156,7 +294,7 @@ function buildRecommendations(
156
294
  const base = {
157
295
  type: group.type,
158
296
  tag: typeTag(group.type),
159
- actionLabel: coreFindingActionLabel(representative) ?? representative.type,
297
+ actionLabel: findingActionLabel(representative),
160
298
  findingCount: group.findingCount,
161
299
  findingIds,
162
300
  };
@@ -169,15 +307,19 @@ function buildRecommendations(
169
307
  // A point estimate, not a range: matches FixTheseFirst.tsx's trailingStat for time groups,
170
308
  // which prints the same figure twice rather than the finding-level spread computeStageUnionMs collapsed.
171
309
  impact: formatWallClockRange(group.recoverableMsHigh, group.recoverableMsHigh),
310
+ impactMeaning: 'of run time',
172
311
  };
173
312
  }
174
313
  if (group.kind === 'resource') {
314
+ const text = formatRawWaste({ value: group.total, unit: group.unit });
315
+ const shown = readsAsZero(text) ? null : text;
175
316
  return {
176
317
  ...base,
177
318
  kind: 'resource',
178
319
  unit: group.unit,
179
320
  total: group.total,
180
- impact: formatRawWaste({ value: group.total, unit: group.unit }),
321
+ impact: shown,
322
+ impactMeaning: shown ? rawWasteMeaning(group.unit) : null,
181
323
  };
182
324
  }
183
325
  return {
@@ -187,51 +329,138 @@ function buildRecommendations(
187
329
  // No single quantifiable figure for a count group; the impact-band tally
188
330
  // (byImpactBand above) is the payload instead.
189
331
  impact: null,
332
+ impactMeaning: null,
190
333
  };
191
334
  });
192
335
  }
193
336
 
194
- // Detector types that fired zero findings, so a flat report can state "checked and clean". Differs
195
- // from the dashboard's Alerts.tsx "Clean checks", which excludes coreLocality (the one remaining
196
- // always-mounted reference widget, shown elsewhere); a flat report has no such separate surface,
197
- // so this includes it too when it has zero findings.
198
- function buildCleanChecks(findings ) {
199
- const firedTypes = new Set(findings.map((f) => f.type));
200
- const seen = new Set ();
201
- const entries = [];
202
- // detectorCatalog() can list the same type more than once (configAudit has 4 entries); dedupe by
203
- // type, keeping first, so a type with sibling entries contributes exactly one clean-check line.
204
- for (const d of detectorCatalog() ) {
205
- if (firedTypes.has(d.type) || seen.has(d.type)) continue;
206
- seen.add(d.type);
207
- entries.push({ type: d.type, tag: typeTag(d.type), thresholdSummary: getThresholdSummary(d.type) });
337
+ // Finding types with no real finding, split into those that passed and those the log could not
338
+ // run (the rule the dashboard's Clean checks uses, from check-coverage.ts). Differs from
339
+ // Alerts.tsx in one way: the dashboard excludes coreLocality (the one always-mounted reference
340
+ // widget, shown elsewhere); a flat report has no such separate surface, so this includes it too.
341
+ function buildCheckLists(
342
+ findings , stages , thresholds ,
343
+ ) {
344
+ // isRealFinding: a type whose only finding is an evidence caveat (memoryUtilization's
345
+ // dataUnavailable variant) had nothing to check, so it lands in notRunChecks.
346
+ const firedTypes = new Set (findings.filter(isRealFinding).map((f) => f.type));
347
+ const coverage = checkCoverage(stages, findings);
348
+ const cleanChecks = [];
349
+ const notRunChecks = [];
350
+ // One line per emitted finding type (configAudit's four entries give one line;
351
+ // broadcastSizing gives overBroadcast and underBroadcast), the same set the dashboard lists.
352
+ for (const [type, { thresholdSummary }] of Object.entries(detectorInfoByType())) {
353
+ if (firedTypes.has(type)) continue;
354
+ // A tuned check was measured against the tuned criterion, so it says which one.
355
+ const tuned = tunedThresholdsForType(type, thresholds);
356
+ const entry = tuned
357
+ ? { type, tag: typeTag(type), thresholdSummary: getThresholdSummary(type, thresholds), tunedThresholds: tuned }
358
+ : { type, tag: typeTag(type), thresholdSummary };
359
+ const reason = coverage.notRunReason(type);
360
+ if (reason) notRunChecks.push({ ...entry, reason });
361
+ else cleanChecks.push(entry);
208
362
  }
209
- return entries;
363
+ return { cleanChecks, notRunChecks };
364
+ }
365
+
366
+ function countByImpactBand(findings ) {
367
+ const counts = { critical: 0, warning: 0, info: 0 };
368
+ for (const f of findings) if (f.impactBand in counts) counts[f.impactBand ] += 1;
369
+ return counts;
370
+ }
371
+
372
+ function verdictJson(model ) {
373
+ return {
374
+ title: model.title,
375
+ summary: model.summary,
376
+ steps: model.shown.map((step) => {
377
+ const recommendation = stepCopyRecommendation(step, model.outcome);
378
+ return {
379
+ key: step.key,
380
+ stageId: step.stageId,
381
+ type: step.lead.type,
382
+ tag: typeTag(step.lead.type),
383
+ leadFindingId: step.lead.id ?? null,
384
+ actionLabel: findingActionLabel(step.lead),
385
+ recommendation,
386
+ impact: impactFigure(step.lead),
387
+ impactMeaning: savingsMeaning(step.lead),
388
+ relatedTypes: step.related.map((f) => f.type),
389
+ text: stepCopyText(step.lead, recommendation, step.stageId),
390
+ };
391
+ }),
392
+ remainingPlaces: model.remaining,
393
+ copyText: model.copyText,
394
+ };
210
395
  }
211
396
 
212
397
  // Keyed by appModel object identity: mcp-tools.ts caches one fixed appModel per runId (never
213
398
  // mutated), so re-running analyze()/auditConfig() reproduces the same catalog. getFindingEvidence
214
399
  // calls buildEvidenceReport once per drill-down; without this, N lookups meant N detector re-runs.
215
400
  // A WeakMap needs no invalidation: once mcp-tools.ts evicts the appModel, this entry is collectible.
216
- const jsonCache = new WeakMap ();
401
+ // Each cache is split first by the overrides object the report ran under (one fixed, frozen object
402
+ // per CLI invocation or MCP server process; DEFAULT_THRESHOLDS for the specification's).
403
+
404
+ const DEFAULT_THRESHOLDS = {};
405
+ const jsonCache = new WeakMap();
406
+ // The redacted report, keyed by the unredacted appModel it was built from.
407
+ const redactedJsonCache = new WeakMap();
217
408
 
218
- function buildJson(appModel ) {
219
- const cached = jsonCache.get(appModel);
220
- if (cached) return cached;
221
- const { app, stages, executors, sql, jobs, runAggregates, evidenceAvailability } = appModel;
409
+ function cacheFor(cache , thresholds ) {
410
+ const key = thresholds ?? DEFAULT_THRESHOLDS;
411
+ let byModel = cache.get(key);
412
+ if (!byModel) {
413
+ byModel = new WeakMap();
414
+ cache.set(key, byModel);
415
+ }
416
+ return byModel;
417
+ }
418
+
419
+ function runFindings(appModel , thresholds ) {
420
+ const { app, stages, executors, sql, jobs, runAggregates } = appModel;
222
421
  const catalog = analyze(
223
422
  app, stages, executors?.added ?? [], executors?.removed ?? [],
224
423
  jobs ?? new Map(), sql ?? new Map(),
225
- runAggregates ?? null,
424
+ runAggregates ?? null, { thresholds },
226
425
  );
227
- const config = auditConfig(app);
426
+ return { catalog, config: auditConfig(app) };
427
+ }
428
+
429
+ // Redacts the model and findings before the report derives any text from them, the same order
430
+ // the HTML export uses: the verdict truncates Spark's failure reason, and redacting that
431
+ // truncated copy afterwards would miss an identifier the cut left as a fragment.
432
+ function buildRedactedJson(appModel , thresholds ) {
433
+ const cache = cacheFor(redactedJsonCache, thresholds);
434
+ const cached = cache.get(appModel);
435
+ if (cached) return cached;
436
+ const { catalog, config } = runFindings(appModel, thresholds);
437
+ const run = redactRunModel(appModel, catalog, config);
438
+ // redactReport stays as a last pass: idempotent over pseudonyms, and it covers the report's own
439
+ // structured fields (summary.app.id) the same way it always has.
440
+ const result = redactReport(buildJson(run.appModel, thresholds, { catalog: run.catalog, config: run.configFindings }));
441
+ cache.set(appModel, result);
442
+ return result;
443
+ }
444
+
445
+ function buildJson(
446
+ appModel , thresholds , findings ,
447
+ ) {
448
+ const cache = cacheFor(jsonCache, thresholds);
449
+ const cached = cache.get(appModel);
450
+ if (cached) return cached;
451
+ const { app, stages, executors, sql, jobs, evidenceAvailability } = appModel;
452
+ const { catalog, config } = findings ?? runFindings(appModel, thresholds);
228
453
  const allFindings = [...catalog, ...config];
229
454
  const rows = sortFindings(allFindings.map(findingRow));
230
455
  const recommendations = buildRecommendations(allFindings, stages ?? new Map());
231
- const cleanChecks = buildCleanChecks(allFindings);
232
-
233
- const impactBandCounts = { critical: 0, warning: 0, info: 0 };
234
- for (const r of rows) if (r.impactBand in impactBandCounts) impactBandCounts[r.impactBand] += 1;
456
+ const { cleanChecks, notRunChecks } = buildCheckLists(allFindings, stages ?? new Map(), thresholds);
457
+ const tuned = tunedDetectors(thresholds);
458
+ const actionable = allFindings.filter(isEligible);
459
+ const fullModel = {
460
+ ...appModel, stages: stages ?? new Map(), jobs: jobs ?? new Map(), executors: executors ?? { added: [], removed: [] },
461
+ };
462
+ const runVerdict = buildRunVerdict(fullModel, allFindings);
463
+ const runOutcome = runVerdict.outcome;
235
464
 
236
465
  const result = {
237
466
  schemaVersion: EVIDENCE_SCHEMA_VERSION,
@@ -247,17 +476,30 @@ function buildJson(appModel ) {
247
476
  jobCount: jobs?.size ?? 0,
248
477
  sqlExecutionCount: sql?.size ?? 0,
249
478
  findingCount: rows.length,
250
- impactBandCounts,
479
+ impactBandCounts: countByImpactBand(rows),
480
+ actionableFindingCount: actionable.length,
481
+ actionableImpactBandCounts: countByImpactBand(actionable),
482
+ clean: isCleanRun({ jobs: jobs ?? new Map(), stages: stages ?? new Map() }, allFindings),
483
+ outcome: {
484
+ failedJobs: runOutcome.failedJobs,
485
+ totalJobs: runOutcome.totalJobs,
486
+ failureReason: runOutcome.reason,
487
+ failureReasonStageId: runOutcome.reason != null ? runOutcome.reasonStageId : null,
488
+ },
489
+ runShape: computeRunShape(fullModel),
490
+ ...(tuned ? { tunedThresholds: tuned } : {}),
251
491
  },
492
+ verdict: verdictJson(runVerdict),
252
493
  evidenceAvailability: evidenceAvailability ?? null,
253
494
  // Detector metadata so the threshold set that produced each finding travels with the evidence.
254
495
  // Order follows DETECTORS (stable) => byte-stable serialization.
255
- detectors: detectorCatalog(),
496
+ detectors: tunedDetectorCatalog(thresholds),
256
497
  findings: rows,
257
498
  recommendations,
258
499
  cleanChecks,
500
+ notRunChecks,
259
501
  };
260
- jsonCache.set(appModel, result);
502
+ cache.set(appModel, result);
261
503
  return result;
262
504
  }
263
505
 
@@ -269,43 +511,107 @@ function renderEvidenceValue(key , value ) {
269
511
  return String(value);
270
512
  }
271
513
 
272
- function formatWallClockRange(low , high ) {
273
- const fmtMs = (ms ) => (ms === 0 ? '0s' : formatDuration(ms));
274
- return low === high ? `Estimated ${fmtMs(high)}` : `Estimated ${fmtMs(low)}-${fmtMs(high)}`;
514
+ // The `failures` finding's distinct errors: a headline per group, its stack excerpt as an indented
515
+ // code block (indented, not fenced, so no excerpt content can close it early).
516
+ function renderFailureGroups(groups ) {
517
+ const lines = [` - failureGroups: ${groups.length}`];
518
+ for (const g of groups) {
519
+ lines.push(` - ${g.count} task(s): ${formatTaskFailureHeadline(g)}`);
520
+ if (g.stackExcerpt) {
521
+ lines.push('');
522
+ for (const l of g.stackExcerpt.split('\n')) lines.push(` ${l}`);
523
+ lines.push('');
524
+ }
525
+ }
526
+ return lines;
527
+ }
528
+
529
+ // The finding's "Potential savings" figure as the dashboard shows it (impactEstimateFigure: the
530
+ // range, or the raw waste only without a range, nothing for a zero or informational estimate),
531
+ // followed by what it counts. `basis: 'informational'` findings carry nothing to print.
532
+ function renderImpactEstimate(estimate ) {
533
+ const figure = impactEstimateFigure(estimate);
534
+ if (!figure) return null;
535
+ return `${figure.text}${figure.meaning ? ` ${figure.meaning}` : ''} (estimateMethod: ${estimate.estimateMethod})`;
275
536
  }
276
537
 
277
- function formatRawWaste(rawWaste ) {
278
- const rounded = Math.round(rawWaste.value * 10) / 10;
279
- switch (rawWaste.unit) {
280
- case 'bytes': return formatBytes(rawWaste.value);
281
- case 'ms': return formatDuration(rawWaste.value);
282
- case 'mbSeconds': return `${rounded} MB-s`;
283
- case 'coreHours': return `${rounded.toFixed(1)} core-h`;
284
- case 'coreMs': return `${rounded} core-ms`;
285
- default: return String(rawWaste.value);
538
+ // The run's job results in one line, worded like the dashboard verdict: failures first, with
539
+ // Spark's recorded reason; null when the log records no ended job.
540
+ function renderOutcome(outcome , incomplete ) {
541
+ const { failedJobs, totalJobs, failureReason, failureReasonStageId } = outcome;
542
+ if (totalJobs === 0) return null;
543
+ if (failedJobs > 0) {
544
+ const failed = failedJobs < totalJobs
545
+ ? `${failedJobs} of ${totalJobs} jobs failed.`
546
+ : totalJobs === 1 ? 'The run\'s one job failed.' : `All ${totalJobs} jobs failed.`;
547
+ if (!failureReason) return failed;
548
+ const where = failureReasonStageId != null ? ` (stage ${failureReasonStageId})` : '';
549
+ return `${failed} Spark's recorded reason${where}: ${failureReason}`;
286
550
  }
551
+ const succeeded = totalJobs === 1 ? 'Its one job succeeded.' : `All ${totalJobs} jobs succeeded.`;
552
+ return incomplete ? `${succeeded} The log has no end-of-run record, so jobs still running when it stops are not counted.` : succeeded;
287
553
  }
288
554
 
289
- // `basis: 'informational'` findings carry no wallClock/rawWaste at all, so
290
- // there's nothing quantifiable to print; the caller skips the line entirely.
291
- function renderImpactEstimate(estimate ) {
292
- const rangeText = estimate.wallClock ? formatWallClockRange(estimate.wallClock.low, estimate.wallClock.high) : null;
293
- const wasteText = estimate.rawWaste ? formatRawWaste(estimate.rawWaste) : null;
294
- if (!rangeText && !wasteText) return null;
295
- const parts = [rangeText, wasteText].filter((p) => p != null).join(' · ');
296
- return `${parts} (estimateMethod: ${estimate.estimateMethod})`;
555
+ // The Scorecard, ETL phases and core-usage figures, each worded to say what it measures, since
556
+ // "efficiency" and "unused core time" are different shares. A figure the dashboard cannot show is
557
+ // left out.
558
+ function renderRunShape(shape ) {
559
+ const lines = [];
560
+ if (shape.wallClockMs != null) lines.push(`- Wall-clock: ${formatDuration(shape.wallClockMs)}`);
561
+ if (shape.efficiencyPct != null) lines.push(`- Efficiency: ${shape.efficiencyPct}% (share of the run with a stage running)`);
562
+ if (shape.unusedCoreTimePct != null) {
563
+ lines.push(`- Unused core time: ${shape.unusedCoreTimePct}% (driver idle plus executor slack, as a share of available core time)`);
564
+ }
565
+ if (shape.peakBusyCores != null) lines.push(`- Peak busy cores: ${formatCores(shape.peakBusyCores)}`);
566
+ if (shape.etlPhasesMs) {
567
+ const { extract, transform, load } = shape.etlPhasesMs;
568
+ lines.push(`- ETL phases (summed stage time): extract ${formatDuration(extract)}, transform ${formatDuration(transform)}, load ${formatDuration(load)}`);
569
+ }
570
+ return lines;
297
571
  }
298
572
 
299
- function renderMarkdown(json ) {
300
- const { summary, findings, evidenceAvailability, detectors, recommendations, cleanChecks } = json;
573
+ // The dashboard verdict card as text: title, summary, then the numbered steps worded as its
574
+ // "Copy next steps" checklist, each followed by the other finding types flagged at that place.
575
+ function renderVerdict(verdict ) {
576
+ const lines = ['## Verdict', '', verdict.title];
577
+ if (verdict.summary.length > 0) lines.push('', verdict.summary.join(' '));
578
+ if (verdict.steps.length > 0) {
579
+ lines.push('');
580
+ verdict.steps.forEach((step, i) => {
581
+ lines.push(`${i + 1}. [${step.tag}] ${step.text}`);
582
+ if (step.relatedTypes.length > 0) {
583
+ const related = step.relatedTypes.map(findingName).join(', ');
584
+ lines.push(` - Also flagged here: ${related}. These often share this cause, so the same fix may clear them too.`);
585
+ }
586
+ });
587
+ if (verdict.remainingPlaces > 0) {
588
+ const places = `${verdict.remainingPlaces} more place${verdict.remainingPlaces === 1 ? '' : 's'}`;
589
+ lines.push('', `${places} to look at in the Findings section below.`);
590
+ }
591
+ }
592
+ lines.push('');
593
+ return lines;
594
+ }
595
+
596
+ // `incomplete` comes from the unfiltered findings, since a findingsFilter can drop the incompleteRun row.
597
+ function renderMarkdown(json , incomplete ) {
598
+ const { summary, verdict, findings, evidenceAvailability, detectors, recommendations, cleanChecks, notRunChecks } = json;
301
599
  const lines = [];
302
600
  lines.push('# Spark run evidence report');
303
601
  lines.push('');
304
602
  lines.push(`- Application: ${summary.app.name ?? '(unknown)'} (${summary.app.id ?? 'n/a'})`);
305
603
  lines.push(`- Spark version: ${summary.app.sparkVersion ?? 'n/a'}`);
604
+ if (summary.tunedThresholds) lines.push(`- Tuned thresholds: ${tunedRunNote(summary.tunedThresholds)}`);
306
605
  lines.push(`- Stages: ${summary.stageCount} · Jobs: ${summary.jobCount} · SQL executions: ${summary.sqlExecutionCount}`);
307
606
  lines.push(`- Findings: ${summary.findingCount} (critical ${summary.impactBandCounts.critical}, warning ${summary.impactBandCounts.warning}, info ${summary.impactBandCounts.info})`);
607
+ const actionableCounts = summary.actionableImpactBandCounts;
608
+ lines.push(`- Findings to act on: ${summary.actionableFindingCount} (critical ${actionableCounts.critical}, warning ${actionableCounts.warning}, info ${actionableCounts.info})`);
609
+ const outcomeLine = renderOutcome(summary.outcome, incomplete);
610
+ if (outcomeLine) lines.push(`- Outcome: ${outcomeLine}`);
611
+ if (summary.clean) lines.push('- Clean run: no findings, no failed jobs, and every check could run.');
612
+ lines.push(...renderRunShape(summary.runShape));
308
613
  lines.push('');
614
+ lines.push(...renderVerdict(verdict));
309
615
  if (recommendations.length > 0) {
310
616
  lines.push(`## Fix these first (${recommendations.length})`);
311
617
  lines.push('');
@@ -313,8 +619,8 @@ function renderMarkdown(json ) {
313
619
  lines.push(`${i + 1}. [${r.tag}] ${r.actionLabel}`);
314
620
  const detail = r.kind === 'count'
315
621
  ? Object.entries(r.byImpactBand ?? {}).map(([impactBand, count]) => `${count} ${impactBand}`).join(', ')
316
- : r.impact;
317
- lines.push(` - ${detail} · ×${r.findingCount} finding(s)`);
622
+ : r.impact && `${r.impact}${r.impactMeaning ? ` ${r.impactMeaning}` : ''}`;
623
+ lines.push(` - ${detail ? `${detail} · ` : ''}×${r.findingCount} finding(s)`);
318
624
  });
319
625
  lines.push('');
320
626
  }
@@ -324,19 +630,25 @@ function renderMarkdown(json ) {
324
630
  const where = r.stageId != null ? ` (stage ${r.stageId})` : '';
325
631
  lines.push(`### ${r.name} · ${r.impactBand}${where}`);
326
632
  lines.push(`- action: ${r.actionLabel}`);
327
- if (r.metric != null) lines.push(`- ${r.metric}: ${r.value}`);
633
+ if (r.metric != null) lines.push(`- ${r.metric}: ${r.valueText ?? r.value}`);
328
634
  if (r.recommendation) lines.push(`- ${r.recommendation}`);
329
635
  if (r.confidence) lines.push(`- confidence: ${r.confidence}`);
330
636
  if (r.validationRequired) lines.push(`- validation: ${r.validationRequired}`);
637
+ if (r.tunedThresholds) lines.push(`- tuned thresholds: ${describeTunedThresholds(r.tunedThresholds)}`);
331
638
  const impactText = r.impactEstimate ? renderImpactEstimate(r.impactEstimate) : null;
332
639
  if (impactText) lines.push(`- impact: ${impactText}`);
640
+ const provenance = r.impactEstimate ? estimateProvenance(r) : null;
641
+ if (provenance) lines.push(`- estimate: ${provenance}`);
333
642
  lines.push(`- detector version: ${r.detectorVersion}`);
334
643
  // Evidence payload (sorted for stable order) so two rows differing only by evidence (two
335
644
  // smallFiles by direction, two partitionSizing by rule) render distinctly.
336
645
  const evidence = Object.entries(r.evidence ?? {}).sort(([a], [b]) => (a < b ? -1 : a > b ? 1 : 0));
337
646
  if (evidence.length) {
338
647
  lines.push('- evidence:');
339
- for (const [k, v] of evidence) lines.push(` - ${k}: ${renderEvidenceValue(k, v)}`);
648
+ for (const [k, v] of evidence) {
649
+ if (k === 'failureGroups' && Array.isArray(v)) lines.push(...renderFailureGroups(v ));
650
+ else lines.push(` - ${k}: ${renderEvidenceValue(k, v)}`);
651
+ }
340
652
  }
341
653
  lines.push('');
342
654
  }
@@ -353,7 +665,18 @@ function renderMarkdown(json ) {
353
665
  lines.push('## Detectors');
354
666
  lines.push('');
355
667
  for (const d of detectors) {
356
- lines.push(`- ${d.type} (v${d.version}, ${d.scope}), thresholds: ${JSON.stringify(d.thresholds)}`);
668
+ const tuned = d.tunedThresholds ? ` (tuned: ${describeTunedThresholds(d.tunedThresholds)})` : '';
669
+ lines.push(`- ${d.type} (v${d.version}, ${d.scope}), thresholds: ${JSON.stringify(d.thresholds)}${tuned}`);
670
+ }
671
+ lines.push('');
672
+ }
673
+ if (notRunChecks.length > 0) {
674
+ lines.push(`## Not checked on this log (${notRunChecks.length})`);
675
+ lines.push('');
676
+ lines.push('The log lacked the data these checks need, so they neither passed nor failed.');
677
+ lines.push('');
678
+ for (const c of notRunChecks) {
679
+ lines.push(`- [${c.tag}] ${c.type}: ${c.reason}`);
357
680
  }
358
681
  lines.push('');
359
682
  }
@@ -361,7 +684,8 @@ function renderMarkdown(json ) {
361
684
  lines.push(`## Clean checks (${cleanChecks.length})`);
362
685
  lines.push('');
363
686
  for (const c of cleanChecks) {
364
- lines.push(`- [${c.tag}] ${c.type}: ${c.thresholdSummary}`);
687
+ const tuned = c.tunedThresholds ? ` (tuned: ${describeTunedThresholds(c.tunedThresholds)})` : '';
688
+ lines.push(`- [${c.tag}] ${c.type}: ${c.thresholdSummary}${tuned}`);
365
689
  }
366
690
  lines.push('');
367
691
  }
@@ -377,7 +701,10 @@ function renderMarkdown(json ) {
377
701
  // CLI/MCP-facing filter over FindingRow, delegating to the shared core predicate that also backs
378
702
  // the dashboard's finding-filter.
379
703
  function matchesFindingsFilter(row , filter ) {
380
- return matchesFindingFilterCriteria(row, filter);
704
+ // A sql-scope finding carries its stages in evidence.stageIds, not a stageId column: it matches
705
+ // the one stage it touches, as the dashboard's Stage details lists it.
706
+ const stageIds = 'stageIds' in row.evidence ? row.evidence.stageIds : null;
707
+ return matchesFindingFilterCriteria({ ...row, stageId: singleStageId({ stageId: row.stageId, stageIds }) }, filter);
381
708
  }
382
709
 
383
710
  /** Build a FindingsFilter from the three optional CLI/MCP filter dimensions, or undefined when
@@ -392,20 +719,21 @@ export function toFindingsFilter(
392
719
  * Build a portable evidence report from an appModel.
393
720
  * @param opts redact=true pseudonymizes app ids / hosts; markdown=false skips the Markdown string;
394
721
  * findingsFilter narrows json.findings (and the Markdown Findings section) only, summary,
395
- * recommendations, and cleanChecks stay computed from the full set, so a narrow filter never
396
- * hides that other checks passed or other fixes exist.
722
+ * recommendations, cleanChecks and notRunChecks stay computed from the full set, so a narrow filter never
723
+ * hides that other checks passed or other fixes exist. thresholds runs the detectors with a user's
724
+ * validated overrides (CLI/MCP only) and labels whatever they changed.
397
725
  */
398
726
  export function buildEvidenceReport(
399
727
  appModel ,
400
- { redact = false, markdown: computeMarkdown = true, findingsFilter }
401
-
728
+ { redact = false, markdown: computeMarkdown = true, findingsFilter, thresholds }
729
+
402
730
  = {},
403
731
  ) {
404
- let json = buildJson(appModel);
405
- if (redact) json = redactReport(json);
732
+ let json = redact ? buildRedactedJson(appModel, thresholds) : buildJson(appModel, thresholds);
733
+ const incomplete = json.findings.some((row) => row.type === 'incompleteRun');
406
734
  // Filter after redact, not before: redaction only replaces string values on surviving rows,
407
735
  // never adds/removes rows, so the two orderings produce identical final content.
408
736
  if (findingsFilter) json = { ...json, findings: json.findings.filter((row) => matchesFindingsFilter(row, findingsFilter)) };
409
- const markdown = computeMarkdown ? renderMarkdown(json) : '';
737
+ const markdown = computeMarkdown ? renderMarkdown(json, incomplete) : '';
410
738
  return { markdown, json };
411
739
  }