sparkforensics-mcp 0.3.0 → 0.5.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (92) hide show
  1. package/bin/sparkforensics-mcp.mjs +41 -10
  2. package/package.json +1 -1
  3. package/vendor-core/allocation.js +106 -0
  4. package/vendor-core/analyzer.js +168 -60
  5. package/vendor-core/check-coverage.js +88 -0
  6. package/vendor-core/cli/budgets.js +54 -27
  7. package/vendor-core/cli/collect-run.js +84 -32
  8. package/vendor-core/cli/regression-budgets.js +83 -0
  9. package/vendor-core/cli/threshold-config.js +28 -0
  10. package/vendor-core/comparison-verdict.js +177 -0
  11. package/vendor-core/core-source-hash.txt +1 -0
  12. package/vendor-core/core-usage-locality.js +56 -2
  13. package/vendor-core/detector-docs.js +58 -0
  14. package/vendor-core/detectors.js +1094 -500
  15. package/vendor-core/docs-config.js +0 -36
  16. package/vendor-core/docs-content/chapters/nav-index.json +31 -0
  17. package/vendor-core/docs-content/detection/cache.md +3 -2
  18. package/vendor-core/docs-content/detection/cfg.md +9 -8
  19. package/vendor-core/docs-content/detection/chrn.md +1 -2
  20. package/vendor-core/docs-content/detection/cold.md +4 -2
  21. package/vendor-core/docs-content/detection/cstor.md +9 -0
  22. package/vendor-core/docs-content/detection/fail.md +3 -2
  23. package/vendor-core/docs-content/detection/gc.md +3 -2
  24. package/vendor-core/docs-content/detection/host.md +2 -1
  25. package/vendor-core/docs-content/detection/local.md +1 -1
  26. package/vendor-core/docs-content/detection/mem.md +5 -2
  27. package/vendor-core/docs-content/detection/plan.md +2 -1
  28. package/vendor-core/docs-content/detection/sfail.md +2 -1
  29. package/vendor-core/docs-content/detection/shape.md +5 -4
  30. package/vendor-core/docs-content/detection/skew.md +3 -1
  31. package/vendor-core/docs-content/detection/slow.md +2 -2
  32. package/vendor-core/docs-content/detection/spec.md +2 -3
  33. package/vendor-core/docs-content/detection/spill.md +1 -1
  34. package/vendor-core/docs-site-config.js +3 -0
  35. package/vendor-core/effective-conf.js +107 -0
  36. package/vendor-core/efficiency-model.js +8 -6
  37. package/vendor-core/event-handlers.js +321 -44
  38. package/vendor-core/event-schemas.js +23 -0
  39. package/vendor-core/evidence-report.js +432 -115
  40. package/vendor-core/export-data.js +79 -6
  41. package/vendor-core/finding-action-label.js +9 -88
  42. package/vendor-core/finding-filter-predicate.js +9 -0
  43. package/vendor-core/finding-generic-recommendation.js +26 -105
  44. package/vendor-core/finding-names.js +28 -45
  45. package/vendor-core/finding-presentation.js +368 -0
  46. package/vendor-core/finding-tag-help.js +110 -0
  47. package/vendor-core/finding-types.js +373 -0
  48. package/vendor-core/findings-of-type.js +11 -0
  49. package/vendor-core/format-utils.js +96 -30
  50. package/vendor-core/html-export.js +51 -0
  51. package/vendor-core/impact-band.js +21 -8
  52. package/vendor-core/impact-estimator.js +25 -520
  53. package/vendor-core/impact-format.js +115 -0
  54. package/vendor-core/impact-model.js +197 -0
  55. package/vendor-core/ingest.js +6 -2
  56. package/vendor-core/intervals.js +13 -0
  57. package/vendor-core/list-runs.js +7 -5
  58. package/vendor-core/load-vendored.js +70 -5
  59. package/vendor-core/mcp-server-factory.js +14 -10
  60. package/vendor-core/mcp-tools.js +105 -45
  61. package/vendor-core/model-assembler.js +35 -1
  62. package/vendor-core/occupancy.js +1 -1
  63. package/vendor-core/parser-worker.js +2 -2
  64. package/vendor-core/plan-graph-model.js +3 -2
  65. package/vendor-core/plan-node-detail.js +1 -1
  66. package/vendor-core/proxy.js +3 -1
  67. package/vendor-core/python-stage.js +25 -0
  68. package/vendor-core/recommendation-rollup.js +70 -3
  69. package/vendor-core/redact.js +96 -37
  70. package/vendor-core/remediation.js +20 -0
  71. package/vendor-core/run-comparison.js +73 -29
  72. package/vendor-core/run-interpretation.js +291 -0
  73. package/vendor-core/run-metrics.js +198 -0
  74. package/vendor-core/run-outcome.js +74 -0
  75. package/vendor-core/run-payload.js +17 -0
  76. package/vendor-core/run-shape.js +40 -0
  77. package/vendor-core/run-totals.js +24 -0
  78. package/vendor-core/run-verdict.js +352 -0
  79. package/vendor-core/scaling-sim.js +4 -5
  80. package/vendor-core/scorecard-estimates.js +63 -0
  81. package/vendor-core/session-snapshot.js +7 -0
  82. package/vendor-core/shs-schemas.js +2 -2
  83. package/vendor-core/spark-memory.js +17 -0
  84. package/vendor-core/sql-stages.js +11 -0
  85. package/vendor-core/stage-plan-nodes.js +18 -0
  86. package/vendor-core/stage-quantiles.js +6 -0
  87. package/vendor-core/threshold-overrides.js +160 -0
  88. package/vendor-core/threshold-summary.js +11 -33
  89. package/vendor-core/types.js +54 -42
  90. package/vendor-core/wall-clock.js +1 -12
  91. package/vendor-core/wasted-core-hours.js +12 -9
  92. package/vendor-core/write-targets.js +312 -0
@@ -2,20 +2,34 @@
2
2
  // then serializes a run summary + findings into a deterministic, byte-stable JSON + Markdown.
3
3
  // Raw task records are never included (privacy baseline); redaction is opt-in via { redact: true }.
4
4
  import { analyze, auditConfig } from './analyzer.js';
5
- import { detectorCatalog } from './detectors.js';
6
- import { typeTag, formatBytes, formatDuration, IMPACT_BAND_ORDER } from './format-utils.js';
7
- import { FINDING_NAMES, titleCase } from './finding-names.js';
8
- import { redactReport } from './redact.js';
9
- import { formatTaskFailureHeadline, } from './task-failure.js';
10
- import { coreFindingActionLabel } from './finding-action-label.js';
11
- import { matchesFindingFilterCriteria } from './finding-filter-predicate.js';
12
- import { buildRecommendationRollup, isEligible, rankFindings, } from './recommendation-rollup.js';
5
+
6
+ import {
7
+ describeTunedThresholds, tunedDetectorCatalog, tunedDetectors, tunedRunNote, tunedThresholdsForType,
8
+ } from './threshold-overrides.js';
13
9
  import { getThresholdSummary } from './threshold-summary.js';
10
+ import {
11
+ typeTag, formatBytes, formatCores, formatDuration, formatWallClockRange, IMPACT_BAND_ORDER,
12
+ } from './format-utils.js';
13
+ import { findingName, titleCase } from './finding-names.js';
14
+ import { redactReport, redactRunModel } from './redact.js';
15
+ import { formatTaskFailureHeadline, } from './task-failure.js';
16
+ import { findingActionLabel } from './finding-action-label.js';
17
+ import { matchesFindingFilterCriteria, singleStageId } from './finding-filter-predicate.js';
18
+ import { buildRecommendationRollup, isEligible, isRealFinding, rankFindings, resourceGroupTotal, } from './recommendation-rollup.js';
19
+ import { checkCoverage, isCleanRun } from './check-coverage.js';
20
+ import { buildRunVerdict, stepCopyRecommendation, stepCopyText, } from './run-verdict.js';
21
+ import {
22
+ estimateProvenance, impactEstimateFigure, impactFigure, rawWasteMeaning, savingsMeaning,
23
+ } from './impact-format.js';
24
+ import { computeRunShape, } from './run-shape.js';
25
+ import { extractWriteTargets, } from './write-targets.js';
26
+ import { detectorInfoByType } from './detector-docs.js';
14
27
 
15
-
28
+
29
+
16
30
 
17
31
 
18
- export const EVIDENCE_SCHEMA_VERSION = 3;
32
+ export const EVIDENCE_SCHEMA_VERSION = 5;
19
33
 
20
34
  // findingRow always sets id/metric/value/recommendation via `?? null` (never omits the key), and
21
35
  // buildJson does the same for evidenceAvailability and summary.app.{id,name,sparkVersion}: these
@@ -23,19 +37,35 @@ export const EVIDENCE_SCHEMA_VERSION = 3;
23
37
  // "unknown version" sentinel). Kept as `?? null`, not `?? undefined`: JSON.stringify drops
24
38
  // undefined keys but keeps null, so undefined would silently strip these from the report.
25
39
  //
26
- // `value` is number|string|null: stageFailed and the configAudit entries put text in Finding.value
27
- // instead of a magnitude, carried through unchanged.
40
+ // `value` is always a magnitude or null. The text-valued findings (stageFailed's failure reason,
41
+ // configAudit's current setting, incompleteRun's 'missing') carry theirs in `valueText` instead,
42
+ // present only on those rows.
28
43
 
29
-
30
-
31
-
32
-
33
-
34
-
44
+
45
+
46
+
47
+
48
+
49
+
50
+
51
+
35
52
 
36
53
 
37
54
 
55
+
56
+
57
+
58
+
59
+
60
+
61
+
38
62
 
63
+
64
+ // One row per finding, discriminated on `type`: `evidence` is that type's public evidence
65
+ // (FindingEvidenceMap, projected through EVIDENCE_KEYS below), never the finding's other fields.
66
+
67
+
68
+
39
69
 
40
70
  // The `Fix these first` rollup row: one entry per buildRecommendationRollup
41
71
  // group, so the CLI/MCP/download paths get the same impact-ranked aggregation the dashboard shows.
@@ -51,75 +81,188 @@ export const EVIDENCE_SCHEMA_VERSION = 3;
51
81
 
52
82
 
53
83
 
84
+
85
+
86
+
54
87
 
88
+
55
89
 
56
90
 
57
- // One line per detector `type` that fired zero findings, so a flat report can state "these were
58
- // checked and came back clean" like the dashboard's clean-checks table.
91
+ // One line per detector `type` that fired zero findings and could run, so a flat report can state
92
+ // "these were checked and came back clean" like the dashboard's clean-checks table.
59
93
 
60
94
 
61
95
 
62
96
 
97
+
98
+
99
+
100
+
101
+ // A detector `type` with zero findings that the log lacked the data to run (the dashboard's "Not
102
+ // checked on this log" group), so it is not reported as passed. `reason` says why, naming the
103
+ // setting to turn on where the detector gives one.
104
+
105
+
106
+
107
+
108
+
109
+
110
+ // How the run ended, as far as its jobs say (run-outcome.ts, the same summary the dashboard's
111
+ // verdict leads with). Counts only jobs with an end record; `failureReason` is the first line of
112
+ // Spark's own recorded reason, only when a job failed, and `failureReasonStageId` the stage it came
113
+ // from (null when there is no reason or it came from a job's exception).
114
+ // One verdict step: a place to look (a stage, or an app-level problem), led by its best-ranked
115
+ // finding, with the other finding types flagged there. `text` is the step's line in the
116
+ // dashboard's "Copy next steps" checklist.
117
+
118
+
119
+
120
+
121
+
122
+
123
+
124
+
125
+
126
+
127
+
128
+
129
+
130
+
131
+ // The dashboard's run verdict (run-verdict.ts): title, summary sentences, the first steps in the
132
+ // same order, how many more places the full list holds, and the "Copy next steps" text.
133
+
134
+
135
+
136
+
137
+
138
+
139
+
140
+
141
+
142
+
143
+
144
+
145
+
63
146
 
64
147
 
65
148
 
66
149
 
67
150
 
68
151
 
69
-
70
-
152
+
153
+
154
+
155
+
156
+
157
+
158
+
159
+
160
+
161
+
162
+
163
+
164
+
165
+
71
166
 
167
+
72
168
 
73
169
 
74
170
 
171
+
172
+
75
173
 
76
174
 
175
+
77
176
 
78
177
 
79
- // Fields surfaced as first-class report columns. Everything else on a finding
80
- // becomes its `evidence` payload (sorted for stable key order).
81
- const CORE_KEYS = new Set([
82
- 'id', 'type', 'name', 'impactBand', 'stageId', 'metric', 'value',
83
- 'recommendation', 'detectorVersion', 'confidence', 'validationRequired', 'docAnchor', 'impactEstimate',
84
- 'actionLabel',
85
- ]);
86
-
87
- // Internal-only fields with no meaning to a human reading this report: never surfaced as a core
88
- // column, and also excluded from the generic evidence dump (unlike stageIds, which IS actionable
89
- // to a reader). `planNodeIds` is view-layer plan-graph node ids (Plan Advisor detectors, see
90
- // plan-graph-model.ts): on a real log it can carry a hundred-plus ids, which would otherwise print
91
- // as one unreadable `- planNodeIds: [...]` line and bloat the report for no reader benefit.
92
- const NON_EVIDENCE_KEYS = new Set(['planNodeIds']);
178
+ // Each finding type's public evidence fields: exactly the keys of its FindingEvidenceMap entry
179
+ // (finding-types.ts), checked both ways at compile time. A field a detector adds for another core
180
+ // module (stageShape's totalCores, utilization's unrounded fraction) is left off both, so it never
181
+ // reaches the report; adding, renaming or dropping a key here changes the report contract.
182
+ const EVIDENCE_KEYS = {
183
+ skew: [],
184
+ stageShape: ['rule'],
185
+ shuffle: [],
186
+ partitionSizing: ['rule'],
187
+ spill: ['spillMagnitude'],
188
+ gc: ['direction'],
189
+ slowHost: ['variant', 'host', 'hostTaskShare', 'hostMeanMs', 'dimension', 'executorId', 'execMaxValue'],
190
+ stageSlowness: [],
191
+ stageFailed: ['variant', 'numTasks', 'memoryBytesSpilled', 'failedTaskDetails'],
192
+ failures: ['failedTasks', 'dominantReason', 'dominantError', 'failureGroups', 'otherFailedTasks'],
193
+ straggler: ['unit', 'speculativeTasks', 'stragglerCount'],
194
+ speculationWaste: [],
195
+ retryWaste: ['numTasks', 'memoryBytesSpilled', 'retriedTaskDetails'],
196
+ tinyTask: [],
197
+ incompleteRun: [],
198
+ coldStart: [],
199
+ utilization: ['cpuUtilizationPct'],
200
+ memoryUtilization: ['variant', 'rule', 'executorId', 'heap', 'dataUnavailable'],
201
+ cacheUtilization: [
202
+ 'variant', 'rddId', 'rddName', 'memorySize', 'diskSize', 'numCachedPartitions', 'numPartitions', 'dataUnavailable',
203
+ ],
204
+ coreLocality: ['nonLocalTaskCount'],
205
+ autoscalingChurn: ['shortLivedExecutorCount'],
206
+ cachingOpportunity: ['variant', 'relation', 'format', 'relations', 'operator', 'executionIds', 'totalReadBytes'],
207
+ jobFailureRate: ['failedJobs', 'totalJobs', 'failedTasks', 'totalTasks', 'avgJobDurationMs', 'taskFailureRate'],
208
+ configAudit: ['property'],
209
+ duplicatePlanSubtree: [
210
+ 'executionId', 'stageIds', 'stageShares', 'occurrencesIdentical', 'rootName', 'subtreeSize', 'sampleRelation',
211
+ 'groupIndex',
212
+ ],
213
+ smallFiles: ['executionId', 'stageIds', 'fileCount', 'direction', 'nodeName'],
214
+ underBroadcast: ['executionId', 'stageIds', 'largerSideBytes'],
215
+ overBroadcast: ['executionId', 'stageIds'],
216
+ } ;
93
217
 
94
- function findingRow(f ) {
218
+ // The other direction: an evidence field EVIDENCE_KEYS doesn't list fails here.
219
+
220
+
221
+
222
+
223
+
224
+
225
+ // The finding's evidence, keys sorted for a stable order. An undefined field (spill's
226
+ // spillMagnitude without a magnitude) is absent, as in the JSON.
227
+ function projectEvidence(f ) {
228
+ const fields = f ;
95
229
  const evidence = {};
96
- for (const k of Object.keys(f).sort()) {
97
- // An undefined field (spill's spillMagnitude without a magnitude) is absent, as in the JSON.
98
- const v = (f )[k];
99
- if (!CORE_KEYS.has(k) && !NON_EVIDENCE_KEYS.has(k) && v !== undefined) evidence[k] = v;
230
+ for (const k of [...EVIDENCE_KEYS[f.type]].sort()) {
231
+ if (fields[k] !== undefined) evidence[k] = fields[k];
100
232
  }
101
- const row = {
233
+ return evidence;
234
+ }
235
+
236
+ function findingRow(f ) {
237
+ // Cast: projectEvidence's keys come from EVIDENCE_KEYS[f.type], so `evidence` is that type's
238
+ // FindingEvidenceMap entry, which TypeScript can't correlate with `type` on its own.
239
+ const row = {
102
240
  id: f.id ?? null,
103
241
  type: f.type,
104
- name: titleCase(FINDING_NAMES[f.type] ?? f.type),
242
+ name: titleCase(findingName(f.type)),
105
243
  tag: typeTag(f.type),
106
244
  impactBand: f.impactBand,
107
245
  stageId: f.stageId ?? null,
108
246
  metric: f.metric ?? null,
109
247
  value: f.value ?? null,
248
+ ...(f.valueText != null ? { valueText: f.valueText } : {}),
110
249
  recommendation: f.recommendation ?? null,
111
250
  detectorVersion: f.detectorVersion ?? 1,
112
- evidence,
113
- // Deliberate simplification vs the view layer's REGISTRY fallback: no widget registry here, and
114
- // falling back to the finding's own `type` is fine since coreFindingActionLabel already covers
115
- // every emitted type; only obscure/future sub-variants hit this fallback.
116
- actionLabel: coreFindingActionLabel(f) ?? f.type,
117
- };
251
+ evidence: projectEvidence(f),
252
+ remediation: f.remediation ?? [],
253
+ actionLabel: findingActionLabel(f),
254
+ } ;
118
255
  // Threshold/confidence provenance, only when the detector emitted it.
119
256
  if (f.confidence != null) row.confidence = f.confidence;
120
257
  if (f.validationRequired != null) row.validationRequired = f.validationRequired;
121
258
  if (f.docAnchor != null) row.docAnchor = f.docAnchor;
122
259
  if (f.impactEstimate != null) row.impactEstimate = f.impactEstimate;
260
+ const figure = impactEstimateFigure(f.impactEstimate);
261
+ if (figure) {
262
+ row.impact = figure.text;
263
+ row.impactMeaning = figure.meaning;
264
+ }
265
+ if (f.tunedThresholds != null) row.tunedThresholds = f.tunedThresholds;
123
266
  return row;
124
267
  }
125
268
 
@@ -157,7 +300,7 @@ function buildRecommendations(
157
300
  const base = {
158
301
  type: group.type,
159
302
  tag: typeTag(group.type),
160
- actionLabel: coreFindingActionLabel(representative) ?? representative.type,
303
+ actionLabel: findingActionLabel(representative),
161
304
  findingCount: group.findingCount,
162
305
  findingIds,
163
306
  };
@@ -170,15 +313,18 @@ function buildRecommendations(
170
313
  // A point estimate, not a range: matches FixTheseFirst.tsx's trailingStat for time groups,
171
314
  // which prints the same figure twice rather than the finding-level spread computeStageUnionMs collapsed.
172
315
  impact: formatWallClockRange(group.recoverableMsHigh, group.recoverableMsHigh),
316
+ impactMeaning: 'of run time',
173
317
  };
174
318
  }
175
319
  if (group.kind === 'resource') {
320
+ const shown = resourceGroupTotal(group);
176
321
  return {
177
322
  ...base,
178
323
  kind: 'resource',
179
324
  unit: group.unit,
180
325
  total: group.total,
181
- impact: formatRawWaste({ value: group.total, unit: group.unit }),
326
+ impact: shown,
327
+ impactMeaning: shown ? rawWasteMeaning(representative.impactEstimate?.rawWaste) : null,
182
328
  };
183
329
  }
184
330
  return {
@@ -188,51 +334,138 @@ function buildRecommendations(
188
334
  // No single quantifiable figure for a count group; the impact-band tally
189
335
  // (byImpactBand above) is the payload instead.
190
336
  impact: null,
337
+ impactMeaning: null,
191
338
  };
192
339
  });
193
340
  }
194
341
 
195
- // Detector types that fired zero findings, so a flat report can state "checked and clean". Differs
196
- // from the dashboard's Alerts.tsx "Clean checks", which excludes coreLocality (the one remaining
197
- // always-mounted reference widget, shown elsewhere); a flat report has no such separate surface,
198
- // so this includes it too when it has zero findings.
199
- function buildCleanChecks(findings ) {
200
- const firedTypes = new Set(findings.map((f) => f.type));
201
- const seen = new Set ();
202
- const entries = [];
203
- // detectorCatalog() can list the same type more than once (configAudit has 4 entries); dedupe by
204
- // type, keeping first, so a type with sibling entries contributes exactly one clean-check line.
205
- for (const d of detectorCatalog() ) {
206
- if (firedTypes.has(d.type) || seen.has(d.type)) continue;
207
- seen.add(d.type);
208
- entries.push({ type: d.type, tag: typeTag(d.type), thresholdSummary: getThresholdSummary(d.type) });
342
+ // Finding types with no real finding, split into those that passed and those the log could not
343
+ // run (the rule the dashboard's Clean checks uses, from check-coverage.ts). Differs from
344
+ // Alerts.tsx in one way: the dashboard excludes coreLocality (the one always-mounted reference
345
+ // widget, shown elsewhere); a flat report has no such separate surface, so this includes it too.
346
+ function buildCheckLists(
347
+ findings , stages , thresholds ,
348
+ ) {
349
+ // isRealFinding: a type whose only finding is an evidence caveat (memoryUtilization's
350
+ // dataUnavailable variant) had nothing to check, so it lands in notRunChecks.
351
+ const firedTypes = new Set (findings.filter(isRealFinding).map((f) => f.type));
352
+ const coverage = checkCoverage(stages, findings);
353
+ const cleanChecks = [];
354
+ const notRunChecks = [];
355
+ // One line per emitted finding type (configAudit's four entries give one line;
356
+ // broadcastSizing gives overBroadcast and underBroadcast), the same set the dashboard lists.
357
+ for (const [type, { thresholdSummary }] of Object.entries(detectorInfoByType())) {
358
+ if (firedTypes.has(type)) continue;
359
+ // A tuned check was measured against the tuned criterion, so it says which one.
360
+ const tuned = tunedThresholdsForType(type, thresholds);
361
+ const entry = tuned
362
+ ? { type, tag: typeTag(type), thresholdSummary: getThresholdSummary(type, thresholds), tunedThresholds: tuned }
363
+ : { type, tag: typeTag(type), thresholdSummary };
364
+ const reason = coverage.notRunReason(type);
365
+ if (reason) notRunChecks.push({ ...entry, reason });
366
+ else cleanChecks.push(entry);
209
367
  }
210
- return entries;
368
+ return { cleanChecks, notRunChecks };
369
+ }
370
+
371
+ function countByImpactBand(findings ) {
372
+ const counts = { critical: 0, warning: 0, info: 0 };
373
+ for (const f of findings) if (f.impactBand in counts) counts[f.impactBand ] += 1;
374
+ return counts;
375
+ }
376
+
377
+ function verdictJson(model ) {
378
+ return {
379
+ title: model.title,
380
+ summary: model.summary,
381
+ steps: model.shown.map((step) => {
382
+ const recommendation = stepCopyRecommendation(step, model.outcome);
383
+ return {
384
+ key: step.key,
385
+ stageId: step.stageId,
386
+ type: step.lead.type,
387
+ tag: typeTag(step.lead.type),
388
+ leadFindingId: step.lead.id ?? null,
389
+ actionLabel: findingActionLabel(step.lead),
390
+ recommendation,
391
+ impact: impactFigure(step.lead),
392
+ impactMeaning: savingsMeaning(step.lead),
393
+ relatedTypes: step.related.map((f) => f.type),
394
+ text: stepCopyText(step.lead, recommendation, step.stageId),
395
+ };
396
+ }),
397
+ remainingPlaces: model.remaining,
398
+ copyText: model.copyText,
399
+ };
211
400
  }
212
401
 
213
402
  // Keyed by appModel object identity: mcp-tools.ts caches one fixed appModel per runId (never
214
403
  // mutated), so re-running analyze()/auditConfig() reproduces the same catalog. getFindingEvidence
215
404
  // calls buildEvidenceReport once per drill-down; without this, N lookups meant N detector re-runs.
216
405
  // A WeakMap needs no invalidation: once mcp-tools.ts evicts the appModel, this entry is collectible.
217
- const jsonCache = new WeakMap ();
406
+ // Each cache is split first by the overrides object the report ran under (one fixed, frozen object
407
+ // per CLI invocation or MCP server process; DEFAULT_THRESHOLDS for the specification's).
408
+
409
+ const DEFAULT_THRESHOLDS = {};
410
+ const jsonCache = new WeakMap();
411
+ // The redacted report, keyed by the unredacted appModel it was built from.
412
+ const redactedJsonCache = new WeakMap();
218
413
 
219
- function buildJson(appModel ) {
220
- const cached = jsonCache.get(appModel);
221
- if (cached) return cached;
222
- const { app, stages, executors, sql, jobs, runAggregates, evidenceAvailability } = appModel;
414
+ function cacheFor(cache , thresholds ) {
415
+ const key = thresholds ?? DEFAULT_THRESHOLDS;
416
+ let byModel = cache.get(key);
417
+ if (!byModel) {
418
+ byModel = new WeakMap();
419
+ cache.set(key, byModel);
420
+ }
421
+ return byModel;
422
+ }
423
+
424
+ function runFindings(appModel , thresholds ) {
425
+ const { app, stages, executors, sql, jobs, runAggregates } = appModel;
223
426
  const catalog = analyze(
224
427
  app, stages, executors?.added ?? [], executors?.removed ?? [],
225
428
  jobs ?? new Map(), sql ?? new Map(),
226
- runAggregates ?? null,
429
+ runAggregates ?? null, { thresholds },
227
430
  );
228
- const config = auditConfig(app);
431
+ return { catalog, config: auditConfig(app) };
432
+ }
433
+
434
+ // Redacts the model and findings before the report derives any text from them, the same order
435
+ // the HTML export uses: the verdict truncates Spark's failure reason, and redacting that
436
+ // truncated copy afterwards would miss an identifier the cut left as a fragment.
437
+ function buildRedactedJson(appModel , thresholds ) {
438
+ const cache = cacheFor(redactedJsonCache, thresholds);
439
+ const cached = cache.get(appModel);
440
+ if (cached) return cached;
441
+ const { catalog, config } = runFindings(appModel, thresholds);
442
+ const run = redactRunModel(appModel, catalog, config);
443
+ // redactReport stays as a last pass: idempotent over pseudonyms, and it covers the report's own
444
+ // structured fields (summary.app.id) the same way it always has.
445
+ const result = redactReport(buildJson(run.appModel, thresholds, { catalog: run.catalog, config: run.configFindings }));
446
+ cache.set(appModel, result);
447
+ return result;
448
+ }
449
+
450
+ function buildJson(
451
+ appModel , thresholds , findings ,
452
+ ) {
453
+ const cache = cacheFor(jsonCache, thresholds);
454
+ const cached = cache.get(appModel);
455
+ if (cached) return cached;
456
+ const { app, stages, executors, sql, jobs, evidenceAvailability } = appModel;
457
+ const { catalog, config } = findings ?? runFindings(appModel, thresholds);
229
458
  const allFindings = [...catalog, ...config];
230
459
  const rows = sortFindings(allFindings.map(findingRow));
231
460
  const recommendations = buildRecommendations(allFindings, stages ?? new Map());
232
- const cleanChecks = buildCleanChecks(allFindings);
233
-
234
- const impactBandCounts = { critical: 0, warning: 0, info: 0 };
235
- for (const r of rows) if (r.impactBand in impactBandCounts) impactBandCounts[r.impactBand] += 1;
461
+ const { cleanChecks, notRunChecks } = buildCheckLists(allFindings, stages ?? new Map(), thresholds);
462
+ const tuned = tunedDetectors(thresholds);
463
+ const actionable = allFindings.filter(isEligible);
464
+ const fullModel = {
465
+ ...appModel, stages: stages ?? new Map(), jobs: jobs ?? new Map(), executors: executors ?? { added: [], removed: [] },
466
+ };
467
+ const runVerdict = buildRunVerdict(fullModel, allFindings);
468
+ const runOutcome = runVerdict.outcome;
236
469
 
237
470
  const result = {
238
471
  schemaVersion: EVIDENCE_SCHEMA_VERSION,
@@ -248,17 +481,33 @@ function buildJson(appModel ) {
248
481
  jobCount: jobs?.size ?? 0,
249
482
  sqlExecutionCount: sql?.size ?? 0,
250
483
  findingCount: rows.length,
251
- impactBandCounts,
484
+ impactBandCounts: countByImpactBand(rows),
485
+ actionableFindingCount: actionable.length,
486
+ actionableImpactBandCounts: countByImpactBand(actionable),
487
+ clean: isCleanRun({ jobs: jobs ?? new Map(), stages: stages ?? new Map() }, allFindings),
488
+ outcome: {
489
+ failedJobs: runOutcome.failedJobs,
490
+ totalJobs: runOutcome.totalJobs,
491
+ failureReason: runOutcome.reason,
492
+ failureReasonStageId: runOutcome.reason != null ? runOutcome.reasonStageId : null,
493
+ },
494
+ runShape: computeRunShape(fullModel),
495
+ ...(tuned ? { tunedThresholds: tuned } : {}),
252
496
  },
497
+ verdict: verdictJson(runVerdict),
253
498
  evidenceAvailability: evidenceAvailability ?? null,
254
499
  // Detector metadata so the threshold set that produced each finding travels with the evidence.
255
500
  // Order follows DETECTORS (stable) => byte-stable serialization.
256
- detectors: detectorCatalog(),
501
+ detectors: tunedDetectorCatalog(thresholds),
257
502
  findings: rows,
503
+ writeTargets: extractWriteTargets(sql ?? new Map(), {
504
+ skippedLines: appModel.skippedLines, unreadableSqlExecutions: appModel.unreadableSqlExecutions,
505
+ }),
258
506
  recommendations,
259
507
  cleanChecks,
508
+ notRunChecks,
260
509
  };
261
- jsonCache.set(appModel, result);
510
+ cache.set(appModel, result);
262
511
  return result;
263
512
  }
264
513
 
@@ -285,43 +534,92 @@ function renderFailureGroups(groups ) {
285
534
  return lines;
286
535
  }
287
536
 
288
- function formatWallClockRange(low , high ) {
289
- const fmtMs = (ms ) => (ms === 0 ? '0s' : formatDuration(ms));
290
- return low === high ? `Estimated ${fmtMs(high)}` : `Estimated ${fmtMs(low)}-${fmtMs(high)}`;
537
+ // The finding's "Potential savings" figure as the dashboard shows it (impactEstimateFigure: the
538
+ // range, or the raw waste only without a range, nothing for a zero or informational estimate),
539
+ // followed by what it counts. `basis: 'informational'` findings carry nothing to print.
540
+ function renderImpactEstimate(estimate ) {
541
+ const figure = impactEstimateFigure(estimate);
542
+ if (!figure) return null;
543
+ return `${figure.text}${figure.meaning ? ` ${figure.meaning}` : ''} (estimateMethod: ${estimate.estimateMethod})`;
291
544
  }
292
545
 
293
- function formatRawWaste(rawWaste ) {
294
- const rounded = Math.round(rawWaste.value * 10) / 10;
295
- switch (rawWaste.unit) {
296
- case 'bytes': return formatBytes(rawWaste.value);
297
- case 'ms': return formatDuration(rawWaste.value);
298
- case 'mbSeconds': return `${rounded} MB-s`;
299
- case 'coreHours': return `${rounded.toFixed(1)} core-h`;
300
- case 'coreMs': return `${rounded} core-ms`;
301
- default: return String(rawWaste.value);
546
+ // The run's job results in one line, worded like the dashboard verdict: failures first, with
547
+ // Spark's recorded reason; null when the log records no ended job.
548
+ function renderOutcome(outcome , incomplete ) {
549
+ const { failedJobs, totalJobs, failureReason, failureReasonStageId } = outcome;
550
+ if (totalJobs === 0) return null;
551
+ if (failedJobs > 0) {
552
+ const failed = failedJobs < totalJobs
553
+ ? `${failedJobs} of ${totalJobs} jobs failed.`
554
+ : totalJobs === 1 ? 'The run\'s one job failed.' : `All ${totalJobs} jobs failed.`;
555
+ if (!failureReason) return failed;
556
+ const where = failureReasonStageId != null ? ` (stage ${failureReasonStageId})` : '';
557
+ return `${failed} Spark's recorded reason${where}: ${failureReason}`;
302
558
  }
559
+ const succeeded = totalJobs === 1 ? 'Its one job succeeded.' : `All ${totalJobs} jobs succeeded.`;
560
+ return incomplete ? `${succeeded} The log has no end-of-run record, so jobs still running when it stops are not counted.` : succeeded;
303
561
  }
304
562
 
305
- // `basis: 'informational'` findings carry no wallClock/rawWaste at all, so
306
- // there's nothing quantifiable to print; the caller skips the line entirely.
307
- function renderImpactEstimate(estimate ) {
308
- const rangeText = estimate.wallClock ? formatWallClockRange(estimate.wallClock.low, estimate.wallClock.high) : null;
309
- const wasteText = estimate.rawWaste ? formatRawWaste(estimate.rawWaste) : null;
310
- if (!rangeText && !wasteText) return null;
311
- const parts = [rangeText, wasteText].filter((p) => p != null).join(' · ');
312
- return `${parts} (estimateMethod: ${estimate.estimateMethod})`;
563
+ // The Scorecard, ETL phases and core-usage figures, each worded to say what it measures, since
564
+ // "efficiency" and "unused core time" are different shares. A figure the dashboard cannot show is
565
+ // left out.
566
+ function renderRunShape(shape ) {
567
+ const lines = [];
568
+ if (shape.wallClockMs != null) lines.push(`- Wall-clock: ${formatDuration(shape.wallClockMs)}`);
569
+ if (shape.efficiencyPct != null) lines.push(`- Efficiency: ${shape.efficiencyPct}% (share of the run with a stage running)`);
570
+ if (shape.unusedCoreTimePct != null) {
571
+ lines.push(`- Unused core time: ${shape.unusedCoreTimePct}% (driver idle plus executor slack, as a share of available core time)`);
572
+ }
573
+ if (shape.peakBusyCores != null) lines.push(`- Peak busy cores: ${formatCores(shape.peakBusyCores)}`);
574
+ if (shape.etlPhasesMs) {
575
+ const { extract, transform, load } = shape.etlPhasesMs;
576
+ lines.push(`- ETL phases (summed stage time): extract ${formatDuration(extract)}, transform ${formatDuration(transform)}, load ${formatDuration(load)}`);
577
+ }
578
+ return lines;
313
579
  }
314
580
 
315
- function renderMarkdown(json ) {
316
- const { summary, findings, evidenceAvailability, detectors, recommendations, cleanChecks } = json;
581
+ // The dashboard verdict card as text: title, summary, then the numbered steps worded as its
582
+ // "Copy next steps" checklist, each followed by the other finding types flagged at that place.
583
+ function renderVerdict(verdict ) {
584
+ const lines = ['## Verdict', '', verdict.title];
585
+ if (verdict.summary.length > 0) lines.push('', verdict.summary.join(' '));
586
+ if (verdict.steps.length > 0) {
587
+ lines.push('');
588
+ verdict.steps.forEach((step, i) => {
589
+ lines.push(`${i + 1}. [${step.tag}] ${step.text}`);
590
+ if (step.relatedTypes.length > 0) {
591
+ const related = step.relatedTypes.map(findingName).join(', ');
592
+ lines.push(` - Also flagged here, likely the same cause: ${related}.`);
593
+ }
594
+ });
595
+ if (verdict.remainingPlaces > 0) {
596
+ const places = `${verdict.remainingPlaces} more place${verdict.remainingPlaces === 1 ? '' : 's'}`;
597
+ lines.push('', `${places} to look at in the Findings section below.`);
598
+ }
599
+ }
600
+ lines.push('');
601
+ return lines;
602
+ }
603
+
604
+ // `incomplete` comes from the unfiltered findings, since a findingsFilter can drop the incompleteRun row.
605
+ function renderMarkdown(json , incomplete ) {
606
+ const { summary, verdict, findings, evidenceAvailability, detectors, recommendations, cleanChecks, notRunChecks } = json;
317
607
  const lines = [];
318
608
  lines.push('# Spark run evidence report');
319
609
  lines.push('');
320
610
  lines.push(`- Application: ${summary.app.name ?? '(unknown)'} (${summary.app.id ?? 'n/a'})`);
321
611
  lines.push(`- Spark version: ${summary.app.sparkVersion ?? 'n/a'}`);
612
+ if (summary.tunedThresholds) lines.push(`- Tuned thresholds: ${tunedRunNote(summary.tunedThresholds)}`);
322
613
  lines.push(`- Stages: ${summary.stageCount} · Jobs: ${summary.jobCount} · SQL executions: ${summary.sqlExecutionCount}`);
323
614
  lines.push(`- Findings: ${summary.findingCount} (critical ${summary.impactBandCounts.critical}, warning ${summary.impactBandCounts.warning}, info ${summary.impactBandCounts.info})`);
615
+ const actionableCounts = summary.actionableImpactBandCounts;
616
+ lines.push(`- Findings to act on: ${summary.actionableFindingCount} (critical ${actionableCounts.critical}, warning ${actionableCounts.warning}, info ${actionableCounts.info})`);
617
+ const outcomeLine = renderOutcome(summary.outcome, incomplete);
618
+ if (outcomeLine) lines.push(`- Outcome: ${outcomeLine}`);
619
+ if (summary.clean) lines.push('- Clean run: no findings, no failed jobs, and every check could run.');
620
+ lines.push(...renderRunShape(summary.runShape));
324
621
  lines.push('');
622
+ lines.push(...renderVerdict(verdict));
325
623
  if (recommendations.length > 0) {
326
624
  lines.push(`## Fix these first (${recommendations.length})`);
327
625
  lines.push('');
@@ -329,8 +627,8 @@ function renderMarkdown(json ) {
329
627
  lines.push(`${i + 1}. [${r.tag}] ${r.actionLabel}`);
330
628
  const detail = r.kind === 'count'
331
629
  ? Object.entries(r.byImpactBand ?? {}).map(([impactBand, count]) => `${count} ${impactBand}`).join(', ')
332
- : r.impact;
333
- lines.push(` - ${detail} · ×${r.findingCount} finding(s)`);
630
+ : r.impact && `${r.impact}${r.impactMeaning ? ` ${r.impactMeaning}` : ''}`;
631
+ lines.push(` - ${detail ? `${detail} · ` : ''}×${r.findingCount} finding(s)`);
334
632
  });
335
633
  lines.push('');
336
634
  }
@@ -340,12 +638,15 @@ function renderMarkdown(json ) {
340
638
  const where = r.stageId != null ? ` (stage ${r.stageId})` : '';
341
639
  lines.push(`### ${r.name} · ${r.impactBand}${where}`);
342
640
  lines.push(`- action: ${r.actionLabel}`);
343
- if (r.metric != null) lines.push(`- ${r.metric}: ${r.value}`);
641
+ if (r.metric != null) lines.push(`- ${r.metric}: ${r.valueText ?? r.value}`);
344
642
  if (r.recommendation) lines.push(`- ${r.recommendation}`);
345
643
  if (r.confidence) lines.push(`- confidence: ${r.confidence}`);
346
644
  if (r.validationRequired) lines.push(`- validation: ${r.validationRequired}`);
645
+ if (r.tunedThresholds) lines.push(`- tuned thresholds: ${describeTunedThresholds(r.tunedThresholds)}`);
347
646
  const impactText = r.impactEstimate ? renderImpactEstimate(r.impactEstimate) : null;
348
647
  if (impactText) lines.push(`- impact: ${impactText}`);
648
+ const provenance = r.impactEstimate ? estimateProvenance(r) : null;
649
+ if (provenance) lines.push(`- estimate: ${provenance}`);
349
650
  lines.push(`- detector version: ${r.detectorVersion}`);
350
651
  // Evidence payload (sorted for stable order) so two rows differing only by evidence (two
351
652
  // smallFiles by direction, two partitionSizing by rule) render distinctly.
@@ -372,7 +673,18 @@ function renderMarkdown(json ) {
372
673
  lines.push('## Detectors');
373
674
  lines.push('');
374
675
  for (const d of detectors) {
375
- lines.push(`- ${d.type} (v${d.version}, ${d.scope}), thresholds: ${JSON.stringify(d.thresholds)}`);
676
+ const tuned = d.tunedThresholds ? ` (tuned: ${describeTunedThresholds(d.tunedThresholds)})` : '';
677
+ lines.push(`- ${d.type} (v${d.version}, ${d.scope}), thresholds: ${JSON.stringify(d.thresholds)}${tuned}`);
678
+ }
679
+ lines.push('');
680
+ }
681
+ if (notRunChecks.length > 0) {
682
+ lines.push(`## Not checked on this log (${notRunChecks.length})`);
683
+ lines.push('');
684
+ lines.push('The log lacked the data these checks need, so they neither passed nor failed.');
685
+ lines.push('');
686
+ for (const c of notRunChecks) {
687
+ lines.push(`- [${c.tag}] ${c.type}: ${c.reason}`);
376
688
  }
377
689
  lines.push('');
378
690
  }
@@ -380,7 +692,8 @@ function renderMarkdown(json ) {
380
692
  lines.push(`## Clean checks (${cleanChecks.length})`);
381
693
  lines.push('');
382
694
  for (const c of cleanChecks) {
383
- lines.push(`- [${c.tag}] ${c.type}: ${c.thresholdSummary}`);
695
+ const tuned = c.tunedThresholds ? ` (tuned: ${describeTunedThresholds(c.tunedThresholds)})` : '';
696
+ lines.push(`- [${c.tag}] ${c.type}: ${c.thresholdSummary}${tuned}`);
384
697
  }
385
698
  lines.push('');
386
699
  }
@@ -396,7 +709,10 @@ function renderMarkdown(json ) {
396
709
  // CLI/MCP-facing filter over FindingRow, delegating to the shared core predicate that also backs
397
710
  // the dashboard's finding-filter.
398
711
  function matchesFindingsFilter(row , filter ) {
399
- return matchesFindingFilterCriteria(row, filter);
712
+ // A sql-scope finding carries its stages in evidence.stageIds, not a stageId column: it matches
713
+ // the one stage it touches, as the dashboard's Stage details lists it.
714
+ const stageIds = 'stageIds' in row.evidence ? row.evidence.stageIds : null;
715
+ return matchesFindingFilterCriteria({ ...row, stageId: singleStageId({ stageId: row.stageId, stageIds }) }, filter);
400
716
  }
401
717
 
402
718
  /** Build a FindingsFilter from the three optional CLI/MCP filter dimensions, or undefined when
@@ -411,20 +727,21 @@ export function toFindingsFilter(
411
727
  * Build a portable evidence report from an appModel.
412
728
  * @param opts redact=true pseudonymizes app ids / hosts; markdown=false skips the Markdown string;
413
729
  * findingsFilter narrows json.findings (and the Markdown Findings section) only, summary,
414
- * recommendations, and cleanChecks stay computed from the full set, so a narrow filter never
415
- * hides that other checks passed or other fixes exist.
730
+ * recommendations, cleanChecks and notRunChecks stay computed from the full set, so a narrow filter never
731
+ * hides that other checks passed or other fixes exist. thresholds runs the detectors with a user's
732
+ * validated overrides (CLI/MCP only) and labels whatever they changed.
416
733
  */
417
734
  export function buildEvidenceReport(
418
735
  appModel ,
419
- { redact = false, markdown: computeMarkdown = true, findingsFilter }
420
-
736
+ { redact = false, markdown: computeMarkdown = true, findingsFilter, thresholds }
737
+
421
738
  = {},
422
739
  ) {
423
- let json = buildJson(appModel);
424
- if (redact) json = redactReport(json);
740
+ let json = redact ? buildRedactedJson(appModel, thresholds) : buildJson(appModel, thresholds);
741
+ const incomplete = json.findings.some((row) => row.type === 'incompleteRun');
425
742
  // Filter after redact, not before: redaction only replaces string values on surviving rows,
426
743
  // never adds/removes rows, so the two orderings produce identical final content.
427
744
  if (findingsFilter) json = { ...json, findings: json.findings.filter((row) => matchesFindingsFilter(row, findingsFilter)) };
428
- const markdown = computeMarkdown ? renderMarkdown(json) : '';
745
+ const markdown = computeMarkdown ? renderMarkdown(json, incomplete) : '';
429
746
  return { markdown, json };
430
747
  }