sparkforensics-mcp 0.3.0 → 0.5.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (92) hide show
  1. package/bin/sparkforensics-mcp.mjs +41 -10
  2. package/package.json +1 -1
  3. package/vendor-core/allocation.js +106 -0
  4. package/vendor-core/analyzer.js +168 -60
  5. package/vendor-core/check-coverage.js +88 -0
  6. package/vendor-core/cli/budgets.js +54 -27
  7. package/vendor-core/cli/collect-run.js +84 -32
  8. package/vendor-core/cli/regression-budgets.js +83 -0
  9. package/vendor-core/cli/threshold-config.js +28 -0
  10. package/vendor-core/comparison-verdict.js +177 -0
  11. package/vendor-core/core-source-hash.txt +1 -0
  12. package/vendor-core/core-usage-locality.js +56 -2
  13. package/vendor-core/detector-docs.js +58 -0
  14. package/vendor-core/detectors.js +1094 -500
  15. package/vendor-core/docs-config.js +0 -36
  16. package/vendor-core/docs-content/chapters/nav-index.json +31 -0
  17. package/vendor-core/docs-content/detection/cache.md +3 -2
  18. package/vendor-core/docs-content/detection/cfg.md +9 -8
  19. package/vendor-core/docs-content/detection/chrn.md +1 -2
  20. package/vendor-core/docs-content/detection/cold.md +4 -2
  21. package/vendor-core/docs-content/detection/cstor.md +9 -0
  22. package/vendor-core/docs-content/detection/fail.md +3 -2
  23. package/vendor-core/docs-content/detection/gc.md +3 -2
  24. package/vendor-core/docs-content/detection/host.md +2 -1
  25. package/vendor-core/docs-content/detection/local.md +1 -1
  26. package/vendor-core/docs-content/detection/mem.md +5 -2
  27. package/vendor-core/docs-content/detection/plan.md +2 -1
  28. package/vendor-core/docs-content/detection/sfail.md +2 -1
  29. package/vendor-core/docs-content/detection/shape.md +5 -4
  30. package/vendor-core/docs-content/detection/skew.md +3 -1
  31. package/vendor-core/docs-content/detection/slow.md +2 -2
  32. package/vendor-core/docs-content/detection/spec.md +2 -3
  33. package/vendor-core/docs-content/detection/spill.md +1 -1
  34. package/vendor-core/docs-site-config.js +3 -0
  35. package/vendor-core/effective-conf.js +107 -0
  36. package/vendor-core/efficiency-model.js +8 -6
  37. package/vendor-core/event-handlers.js +321 -44
  38. package/vendor-core/event-schemas.js +23 -0
  39. package/vendor-core/evidence-report.js +432 -115
  40. package/vendor-core/export-data.js +79 -6
  41. package/vendor-core/finding-action-label.js +9 -88
  42. package/vendor-core/finding-filter-predicate.js +9 -0
  43. package/vendor-core/finding-generic-recommendation.js +26 -105
  44. package/vendor-core/finding-names.js +28 -45
  45. package/vendor-core/finding-presentation.js +368 -0
  46. package/vendor-core/finding-tag-help.js +110 -0
  47. package/vendor-core/finding-types.js +373 -0
  48. package/vendor-core/findings-of-type.js +11 -0
  49. package/vendor-core/format-utils.js +96 -30
  50. package/vendor-core/html-export.js +51 -0
  51. package/vendor-core/impact-band.js +21 -8
  52. package/vendor-core/impact-estimator.js +25 -520
  53. package/vendor-core/impact-format.js +115 -0
  54. package/vendor-core/impact-model.js +197 -0
  55. package/vendor-core/ingest.js +6 -2
  56. package/vendor-core/intervals.js +13 -0
  57. package/vendor-core/list-runs.js +7 -5
  58. package/vendor-core/load-vendored.js +70 -5
  59. package/vendor-core/mcp-server-factory.js +14 -10
  60. package/vendor-core/mcp-tools.js +105 -45
  61. package/vendor-core/model-assembler.js +35 -1
  62. package/vendor-core/occupancy.js +1 -1
  63. package/vendor-core/parser-worker.js +2 -2
  64. package/vendor-core/plan-graph-model.js +3 -2
  65. package/vendor-core/plan-node-detail.js +1 -1
  66. package/vendor-core/proxy.js +3 -1
  67. package/vendor-core/python-stage.js +25 -0
  68. package/vendor-core/recommendation-rollup.js +70 -3
  69. package/vendor-core/redact.js +96 -37
  70. package/vendor-core/remediation.js +20 -0
  71. package/vendor-core/run-comparison.js +73 -29
  72. package/vendor-core/run-interpretation.js +291 -0
  73. package/vendor-core/run-metrics.js +198 -0
  74. package/vendor-core/run-outcome.js +74 -0
  75. package/vendor-core/run-payload.js +17 -0
  76. package/vendor-core/run-shape.js +40 -0
  77. package/vendor-core/run-totals.js +24 -0
  78. package/vendor-core/run-verdict.js +352 -0
  79. package/vendor-core/scaling-sim.js +4 -5
  80. package/vendor-core/scorecard-estimates.js +63 -0
  81. package/vendor-core/session-snapshot.js +7 -0
  82. package/vendor-core/shs-schemas.js +2 -2
  83. package/vendor-core/spark-memory.js +17 -0
  84. package/vendor-core/sql-stages.js +11 -0
  85. package/vendor-core/stage-plan-nodes.js +18 -0
  86. package/vendor-core/stage-quantiles.js +6 -0
  87. package/vendor-core/threshold-overrides.js +160 -0
  88. package/vendor-core/threshold-summary.js +11 -33
  89. package/vendor-core/types.js +54 -42
  90. package/vendor-core/wall-clock.js +1 -12
  91. package/vendor-core/wasted-core-hours.js +12 -9
  92. package/vendor-core/write-targets.js +312 -0
@@ -7,14 +7,14 @@
7
7
  // Deterministic (sorted assignment), idempotent (pseudonyms map to themselves),
8
8
  // and non-mutating (returns a fresh, deep-copied tree).
9
9
 
10
-
11
- import { redactTaskFailureGroup, } from './task-failure.js';
10
+ import { decodeCollections, encodeCollections, } from './export-data.js';
11
+
12
+ import { REDACTED_TEXT, redactTaskFailureGroup, } from './task-failure.js';
12
13
 
13
14
  // Host / IP identifier patterns. Used to enumerate host names that surface only
14
- // inside free text: recommendation strings, `stageFailed`'s failure-reason
15
- // value, SQL relation/node names, never as a structured `host` field, so
16
- // redaction reaches those residuals too. Pseudonyms (`host-1`) match neither
17
- // pattern, keeping the scan idempotent.
15
+ // inside free text: recommendation strings, SQL relation/node names, never as a
16
+ // structured `host` field, so redaction reaches those residuals too. Pseudonyms
17
+ // (`host-1`) match neither pattern, keeping the scan idempotent.
18
18
  const HOST_PATTERNS = [
19
19
  // EC2-style ip-10-1-2-3 with an optional dotted domain (ip-10-1-2-3.ec2.internal).
20
20
  // Each domain label must start with an alphanumeric, so a trailing sentence
@@ -35,7 +35,7 @@ const APP_ID_PATTERNS = [/\bapplication_\d{10,}_\d+\b/g];
35
35
  // Walk every string in the tree once, collecting matches for each `{ patterns,
36
36
  // out }` sink. One shared traversal for every token kind (instead of one
37
37
  // traversal per kind) keeps redactComparison's dual host+app-id scan the same
38
- // cost as the single-kind scan redactReport/redactAppIdentity already do.
38
+ // cost as the single-kind scan redactReport already does.
39
39
  function scanTokens(node , sinks ) {
40
40
  if (typeof node === 'string') {
41
41
  for (const { patterns, out } of sinks) {
@@ -55,11 +55,6 @@ function scanTokens(node , sinks
55
55
  }
56
56
  }
57
57
 
58
- // Walk every string in the tree, collecting host/IP tokens into `hosts`.
59
- function scanHostTokens(node , hosts ) {
60
- scanTokens(node, [{ patterns: HOST_PATTERNS, out: hosts }]);
61
- }
62
-
63
58
  // Recursively collects every string value found under a key literally named
64
59
  // `host`, anywhere in the tree. Host names surface at several depths, a
65
60
  // finding's own `host`, `evidence.host`, and now
@@ -99,12 +94,37 @@ function redactFailureGroups (node ) {
99
94
  return node;
100
95
  }
101
96
 
97
+ // A stage's failure reason is Spark's free-text message and can carry file paths and data values
98
+ // too, so it is replaced outright, like a failure group's message: a `stageFailed` finding's
99
+ // `valueText`, the `stageFailureReason` of a stage record, a job's `exception` (the run outcome's
100
+ // fallback reason), and the report's `failureReason` copy. Walking by key name, like
101
+ // collectHostFields, needs no path list. Returns a fresh tree.
102
+ const REASON_KEYS = new Set(['stageFailureReason', 'exception', 'failureReason']);
103
+ function redactStageFailureReasons (node ) {
104
+ if (Array.isArray(node)) return node.map((n) => redactStageFailureReasons(n)) ;
105
+ if (node && typeof node === 'object') {
106
+ const isStageFailed = (node ).type === 'stageFailed';
107
+ const out = {};
108
+ for (const [k, v] of Object.entries(node)) {
109
+ const isReason = REASON_KEYS.has(k) || (isStageFailed && k === 'valueText');
110
+ out[k] = isReason && typeof v === 'string' ? REDACTED_TEXT : redactStageFailureReasons(v);
111
+ }
112
+ return out ;
113
+ }
114
+ return node;
115
+ }
116
+
117
+ // Both passes for free text no host/app-id pattern recognizes.
118
+ function redactFailureText (node ) {
119
+ return redactStageFailureReasons(redactFailureGroups(node));
120
+ }
121
+
102
122
  // Spark config keys ending in `host`/`hostname` (e.g. spark.driver.host,
103
123
  // spark.yarn.am.hostname) carry plain FQDN host names that neither
104
124
  // HOST_PATTERNS matches (no IP/EC2 shape) nor collectHostFields's by-key-name
105
125
  // walk catches (the literal key is the dotted Spark property name, never
106
126
  // `host` itself). app.config is a flat Record<string, string> unique to
107
- // redactExportData: no other redact* export ships a raw Spark config dict.
127
+ // redactRunModel: no other redact* export ships a raw Spark config dict.
108
128
  // Known gap: a hostname value under a differently-named key isn't caught by
109
129
  // this suffix check. Confirmed against a real cluster config: spark.master,
110
130
  // spark.yarn.historyServer.address, and the plural YARN proxy/HA keys
@@ -126,7 +146,7 @@ function collectConfigHostValues(config , hos
126
146
 
127
147
 
128
148
 
129
-
149
+
130
150
 
131
151
 
132
152
 
@@ -192,28 +212,19 @@ function applyReplacements (node , ids
192
212
  return deepReplace(node, merged) ;
193
213
  }
194
214
 
195
- export function redactReport (input ) {
196
- const report = redactFailureGroups(input);
197
- const { appIds, hosts } = collectIds(report);
198
- return applyReplacements(report, { appIds, hosts });
215
+ // The app name identifies a job as much as its id, so a redacted name becomes the id's
216
+ // pseudonym, as list_runs does. Only the structured field is replaced, never substrings: a
217
+ // short name ("t", "etl") would otherwise corrupt every string it happens to occur in.
218
+ function withRedactedName (app ) {
219
+ return app.name == null ? app : { ...app, name: app.id ?? null };
199
220
  }
200
221
 
201
- // Narrow counterpart to redactReport(), for getRunSummary()'s standalone app
202
- // object (no findings tree to walk). There's exactly one app id here, so no
203
- // Set/Map/sort is needed for it; name/sparkVersion still go through the
204
- // shared host/IP scan-and-replace since either can carry a host token as
205
- // free text.
206
- export function redactAppIdentity(
207
- app ,
208
- ) {
209
- const hosts = new Set ();
210
- scanHostTokens(app.name, hosts);
211
- scanHostTokens(app.sparkVersion, hosts);
212
- return {
213
- id: typeof app.id === 'string' && app.id.length > 0 ? 'app-1' : app.id,
214
- name: applyReplacements(app.name, { hosts }),
215
- sparkVersion: applyReplacements(app.sparkVersion, { hosts }),
216
- };
222
+ export function redactReport (input ) {
223
+ const report = redactFailureText(input);
224
+ const { appIds, hosts } = collectIds(report);
225
+ const out = applyReplacements(report, { appIds, hosts });
226
+ const app = out.summary?.app;
227
+ return app ? { ...out, summary: { ...out.summary, app: withRedactedName(app) } } : out;
217
228
  }
218
229
 
219
230
  // Run-comparison counterpart: no single app-id *field* to pseudonymize
@@ -237,8 +248,10 @@ export function redactComparison (comparison ) {
237
248
  // this also walks executors.added/removed for their literal `host` field
238
249
  // (ExecutorAddedEvent.host), since raw executor records: not just findings
239
250
  //: reach data.js.
240
- export function redactExportData(input ) {
241
- const data = redactFailureGroups(input);
251
+
252
+
253
+ function redactRunTree (input ) {
254
+ const data = redactFailureText(input);
242
255
  const appIds = new Set ();
243
256
  const hosts = new Set ();
244
257
  const appId = data.app?.id;
@@ -249,5 +262,51 @@ export function redactExportData(input ) {
249
262
  collectHostFields(data.configFindings, hosts);
250
263
  collectConfigHostValues(data.app?.config, hosts);
251
264
  scanTokens(data, [{ patterns: HOST_PATTERNS, out: hosts }, { patterns: APP_ID_PATTERNS, out: appIds }]);
252
- return applyReplacements(data, { appIds, hosts });
265
+ const out = applyReplacements(data, { appIds, hosts });
266
+ if (!out.app) return out;
267
+ const app = withRedactedName(out.app);
268
+ // spark.app.name repeats the name in the config the HTML export ships.
269
+ if (app.config?.['spark.app.name'] == null) return { ...out, app };
270
+ return { ...out, app: { ...app, config: { ...app.config, 'spark.app.name': app.id ?? '' } } };
271
+ }
272
+
273
+ /** A run's model and findings with every identifier pseudonymized. Redact this before anything derives
274
+ * text from the run: the verdict truncates Spark's failure reason, and an
275
+ * identifier cut by that truncation is a fragment no later pass can match. */
276
+ export function redactRunModel(
277
+ appModel ,
278
+ catalog ,
279
+ configFindings ,
280
+ ) {
281
+ // Maps and Sets would lose their entries in the deep copy, so they cross as
282
+ // tagged plain objects, the way the export payload carries them.
283
+ const tree = encodeCollections({
284
+ app: appModel.app,
285
+ stages: [...appModel.stages.values()],
286
+ jobs: [...appModel.jobs.values()],
287
+ sql: [...appModel.sql.values()],
288
+ executors: appModel.executors,
289
+ runAggregates: appModel.runAggregates,
290
+ evidenceAvailability: appModel.evidenceAvailability,
291
+ catalog,
292
+ configFindings,
293
+ }) ;
294
+ const redacted = decodeCollections(redactRunTree(tree)) ;
295
+ const byId = (items ) => new Map((items ).map((item) => [item.id, item]));
296
+ return {
297
+ appModel: {
298
+ app: redacted.app,
299
+ stages: byId(redacted.stages),
300
+ jobs: byId(redacted.jobs),
301
+ sql: byId(redacted.sql),
302
+ executors: redacted.executors,
303
+ runAggregates: redacted.runAggregates ,
304
+ evidenceAvailability: redacted.evidenceAvailability ,
305
+ // Counts and execution ids, nothing to pseudonymize.
306
+ skippedLines: appModel.skippedLines,
307
+ unreadableSqlExecutions: appModel.unreadableSqlExecutions,
308
+ } ,
309
+ catalog: redacted.catalog,
310
+ configFindings: redacted.configFindings,
311
+ };
253
312
  }
@@ -0,0 +1,20 @@
1
+ // Constructors for the structured `remediation` a detector attaches next to its prose
2
+ // `recommendation`: only where the detector already names the Spark property, never an invented one.
3
+
4
+
5
+
6
+
7
+ /** The property should be raised; `suggested` is the value when the detector computed one. */
8
+ export function increaseConf(key , suggested = null) {
9
+ return { kind: 'conf', key, direction: 'increase', suggested };
10
+ }
11
+
12
+ /** The property should be lowered; `suggested` is the value when the detector computed one. */
13
+ export function decreaseConf(key , suggested = null) {
14
+ return { kind: 'conf', key, direction: 'decrease', suggested };
15
+ }
16
+
17
+ /** The property should take a specific value (a switch or a class name), not move up or down. */
18
+ export function setConf(key , suggested = null) {
19
+ return { kind: 'conf', key, direction: 'set', suggested };
20
+ }
@@ -1,9 +1,16 @@
1
1
  import { computeWallClock } from './wall-clock.js';
2
2
  import { normalizeDetail } from './detectors.js';
3
3
  import { cyrb53 } from './string-hash.js';
4
+ import { computeAllocation } from './allocation.js';
5
+ import { totalExecutorCpuMs, withEarlierAttempts } from './run-totals.js';
6
+ import { planNodesOfStage } from './stage-plan-nodes.js';
4
7
  import { captureSnapshot } from './session-snapshot.js';
5
-
8
+ import { tunedRunNote } from './threshold-overrides.js';
9
+
6
10
 
11
+
12
+ import { isIncompleteRun } from './check-coverage.js';
13
+ import { summarizeRunOutcome } from './run-outcome.js';
7
14
 
8
15
 
9
16
 
@@ -12,7 +19,7 @@ import { captureSnapshot } from './session-snapshot.js';
12
19
 
13
20
 
14
21
 
15
-
22
+
16
23
 
17
24
 
18
25
 
@@ -23,7 +30,16 @@ import { captureSnapshot } from './session-snapshot.js';
23
30
 
24
31
 
25
32
 
33
+
34
+
35
+
36
+
26
37
 
38
+
39
+ function jobOutcome(snapshot ) {
40
+ const { failedJobs, totalJobs } = summarizeRunOutcome(snapshot.jobs, snapshot.catalog);
41
+ return { failedJobs, totalJobs, incomplete: isIncompleteRun(snapshot.catalog) };
42
+ }
27
43
 
28
44
  // Replace run-varying tokens (digit runs, long hex ids) with a stable marker so
29
45
  // the same logical stage across two runs normalizes to one identity.
@@ -63,23 +79,18 @@ function planTreeIdentity(root ) {
63
79
  // Falls back to the coarser whole-tree identity when the stage has no
64
80
  // attributed nodes (hand-built snapshots without `stageIds`, or unmatched
65
81
  // accumulables).
66
- function sqlNodeIdentity(stage , snapshot ) {
82
+ function sqlNodeIdentity(stage , snapshot ) {
67
83
  const execId = stage.sqlExecutionId;
68
84
  if (execId == null) return '';
69
85
  const root = snapshot.sql.get(execId)?.planTree ?? null;
70
86
  if (!root) return '';
71
- const fingerprints = [];
72
- (function collect(node ) {
73
- if (node.stageIds?.includes(stage.id)) {
74
- fingerprints.push(JSON.stringify([normalizeStageName(node.name ?? ''), normalizeDetail(node.detail ?? '')]));
75
- }
76
- for (const child of node.children ?? []) collect(child);
77
- })(root);
87
+ const fingerprints = planNodesOfStage(stage, snapshot.sql)
88
+ .map((node) => JSON.stringify([normalizeStageName(node.name ?? ''), normalizeDetail(node.detail ?? '')]));
78
89
  if (fingerprints.length === 0) return planTreeIdentity(root) ?? '';
79
90
  return cyrb53(JSON.stringify(fingerprints.sort()));
80
91
  }
81
92
 
82
- export function stageIdentity(stage , snapshot ) {
93
+ export function stageIdentity(stage , snapshot ) {
83
94
  return normalizeStageName(stage.name ?? '') + '§' + sqlNodeIdentity(stage, snapshot);
84
95
  }
85
96
 
@@ -186,6 +197,7 @@ function skewRatios(stages ) {
186
197
  export const COMPARISON_METRIC_KEYS = [
187
198
  'wallClock', 'shuffleSpill', 'taskSkew', 'failedTaskRate', 'diskSpill', 'gcTime',
188
199
  'inputBytes', 'outputBytes', 'executorRunTime', 'taskCount', 'executorsAdded',
200
+ 'executorCpuTime', 'allocatedCoreHours',
189
201
  ];
190
202
 
191
203
  // Volume/count metrics, not cost metrics: more or less input/output data, or
@@ -215,20 +227,24 @@ function metric(
215
227
 
216
228
  export function metricDeltas(baseSnap , candSnap ) {
217
229
  const out = [];
230
+ // Run totals count every task attempt, failed and speculative ones too (run-totals.ts), the
231
+ // same sums the CLI metrics block reports; skew stays on each stage's latest attempt.
232
+ const attemptsOf = (snap ) => withEarlierAttempts([...snap.stages.values()]);
218
233
 
219
234
  // Wall-clock: always computable (computeWallClock tolerates a null app).
220
235
  out.push(metric('wallClock', 'Wall-clock duration',
221
236
  computeWallClock(baseSnap.app, baseSnap.stages).total,
222
237
  computeWallClock(candSnap.app, candSnap.stages).total));
223
238
 
224
- // Shuffle spill over the whole run: a sum needs no stage matching, and
225
- // matching is unreliable on real logs (see plan rationale), so scope it to
226
- // all stages exactly like task-skew and failed-rate below.
227
- const bSpill = sumField([...baseSnap.stages.values()], 'memoryBytesSpilled');
228
- const cSpill = sumField([...candSnap.stages.values()], 'memoryBytesSpilled');
229
- out.push(metric('shuffleSpill', 'Shuffle spill',
239
+ // Memory spill (Spark's memoryBytesSpilled) over the whole run: a sum needs no
240
+ // stage matching, and matching is unreliable on real logs, so scope it to all
241
+ // stages exactly like task-skew and failed-rate below. The key stays
242
+ // `shuffleSpill` so existing --regression-metric callers keep working.
243
+ const bSpill = sumField(attemptsOf(baseSnap), 'memoryBytesSpilled');
244
+ const cSpill = sumField(attemptsOf(candSnap), 'memoryBytesSpilled');
245
+ out.push(metric('shuffleSpill', 'Memory spill',
230
246
  bSpill.present ? bSpill.sum : null, cSpill.present ? cSpill.sum : null,
231
- { unavailableReason: bSpill.present && cSpill.present ? undefined : 'No shuffle-spill data recorded for a run' }));
247
+ { unavailableReason: bSpill.present && cSpill.present ? undefined : 'No memory-spill data recorded for a run' }));
232
248
 
233
249
  // Task skew: p95 of per-stage ratio across the whole run.
234
250
  const bSkew = p95(skewRatios([...baseSnap.stages.values()]));
@@ -237,10 +253,10 @@ export function metricDeltas(baseSnap , candSnap
237
253
  { unavailableReason: bSkew != null && cSkew != null ? undefined : 'No stage had measurable duration for a run' }));
238
254
 
239
255
  // Failed-task rate: Σ failedTasks / Σ taskCount.
240
- const bTasks = sumField([...baseSnap.stages.values()], 'taskCount');
241
- const cTasks = sumField([...candSnap.stages.values()], 'taskCount');
242
- const bFailed = sumField([...baseSnap.stages.values()], 'failedTasks');
243
- const cFailed = sumField([...candSnap.stages.values()], 'failedTasks');
256
+ const bTasks = sumField(attemptsOf(baseSnap), 'taskCount');
257
+ const cTasks = sumField(attemptsOf(candSnap), 'taskCount');
258
+ const bFailed = sumField(attemptsOf(baseSnap), 'failedTasks');
259
+ const cFailed = sumField(attemptsOf(candSnap), 'failedTasks');
244
260
  // Guard on BOTH inputs: a missing `failedTasks` field must render Unavailable,
245
261
  // not a false 0% rate (dividing an absent-and-therefore-0 numerator).
246
262
  const bRate = bTasks.present && bTasks.sum > 0 && bFailed.present ? bFailed.sum / bTasks.sum : null;
@@ -252,8 +268,8 @@ export function metricDeltas(baseSnap , candSnap
252
268
  // carries (set in finalizeStage). Correct at any match coverage, like the
253
269
  // sums above; no parser or detector change.
254
270
  const sumMetric = (key , label , field , reason ) => {
255
- const b = sumField([...baseSnap.stages.values()], field);
256
- const c = sumField([...candSnap.stages.values()], field);
271
+ const b = sumField(attemptsOf(baseSnap), field);
272
+ const c = sumField(attemptsOf(candSnap), field);
257
273
  out.push(metric(key, label, b.present ? b.sum : null, c.present ? c.sum : null,
258
274
  { unavailableReason: b.present && c.present ? undefined : reason }));
259
275
  };
@@ -272,6 +288,18 @@ export function metricDeltas(baseSnap , candSnap
272
288
  out.push(metric('executorsAdded', 'Executors added', bExec, cExec,
273
289
  { unavailableReason: bExec != null && cExec != null ? undefined : 'No executor events recorded for a run' }));
274
290
 
291
+ // Executor CPU time (ms) and allocated core-hours cost resources, so less is better. CPU time is
292
+ // null, not 0, on a run whose log never recorded it (older Spark).
293
+ const cpuMs = (snap ) => totalExecutorCpuMs(attemptsOf(snap));
294
+ const bCpu = cpuMs(baseSnap), cCpu = cpuMs(candSnap);
295
+ out.push(metric('executorCpuTime', 'Executor CPU time', bCpu, cCpu,
296
+ { unavailableReason: bCpu != null && cCpu != null ? undefined : 'No executor CPU time recorded for a run' }));
297
+ const coreHours = (snap ) =>
298
+ (snap.executors ? computeAllocation(snap).coreHours : null);
299
+ const bCore = coreHours(baseSnap), cCore = coreHours(candSnap);
300
+ out.push(metric('allocatedCoreHours', 'Allocated core-hours', bCore, cCore,
301
+ { unavailableReason: bCore != null && cCore != null ? undefined : 'No executor lifecycle or core count recorded for a run' }));
302
+
275
303
  return out;
276
304
  }
277
305
 
@@ -282,14 +310,14 @@ export function metricDeltas(baseSnap , candSnap
282
310
  export function findingsDelta(baseSnap , candSnap )
283
311
 
284
312
  {
285
-
313
+
286
314
  const tally = (snap ) => {
287
315
  const m = new Map (); // `${rule}§${impactBand}` -> { rule, impactBand, count, stages:Set }
288
316
  for (const f of snap.catalog) {
289
- const rule = typeof f.rule === 'string' ? f.rule : f.type;
317
+ const rule = 'rule' in f && typeof f.rule === 'string' ? f.rule : f.type;
290
318
  const impactBand = f.impactBand ?? 'unknown';
291
319
  const key = `${rule}§${impactBand}`;
292
- const e = m.get(key) ?? { rule, impactBand, count: 0, stages: new Set () };
320
+ const e = m.get(key) ?? { rule, type: f.type, impactBand, count: 0, stages: new Set () };
293
321
  e.count++;
294
322
  // Resolve stageId → name on this snapshot only: a single-side lookup, so
295
323
  // it needs no cross-run identity. App-level findings (stageId null) add none.
@@ -308,7 +336,7 @@ export function findingsDelta(baseSnap , candSnap
308
336
  if (candCount === baseCount) continue;
309
337
  const meta = ce ?? be ;
310
338
  const more = candCount > baseCount ? ce : be ; // the side with more supplies the labels
311
- const row = { rule: meta.rule, impactBand: meta.impactBand, baseCount, candCount,
339
+ const row = { rule: meta.rule, type: meta.type, impactBand: meta.impactBand, baseCount, candCount,
312
340
  delta: candCount - baseCount, stages: [...more.stages].sort() };
313
341
  (candCount > baseCount ? introduced : resolved).push(row);
314
342
  }
@@ -428,6 +456,7 @@ export function compareRuns(
428
456
  stageSkew: stageSkewDeltas(baseSnap, candSnap, match),
429
457
  baseStages: stageList(baseSnap),
430
458
  candStages: stageList(candSnap),
459
+ jobOutcomes: { baseline: jobOutcome(baseSnap), candidate: jobOutcome(candSnap) },
431
460
  };
432
461
  }
433
462
 
@@ -445,8 +474,23 @@ function renderFindingsSection(title , findings )
445
474
  // evidence-report.ts's renderMarkdown house style (## section heading, ###
446
475
  // subheadings, `- key: value` bullets). Shared by the CLI's --baseline
447
476
  // markdown output and the MCP server's compare_runs `format: 'md'`.
448
- export function renderComparisonMarkdown(comparison ) {
477
+ /** `tuned`: tunedDetectors() for the overrides both runs were analyzed with, named in the output
478
+ * as the evidence report names them. */
479
+ export function renderComparisonMarkdown(
480
+ comparison , verdict , tuned ,
481
+ ) {
449
482
  const lines = ['', '## Comparison to baseline', ''];
483
+ if (tuned) lines.push(`- Tuned thresholds (both runs): ${tunedRunNote(tuned)}`, '');
484
+ if (verdict) {
485
+ // Names each run by its label (MCP's run IDs), unless the labels are just the role names the
486
+ // CLI passes, where the verdict below already says baseline and candidate.
487
+ if (comparison.baselineLabel !== 'baseline' || comparison.candidateLabel !== 'candidate') {
488
+ lines.push(`Baseline: ${comparison.baselineLabel} · Candidate: ${comparison.candidateLabel}`, '');
489
+ }
490
+ lines.push(verdict.title);
491
+ if (verdict.sentences.length > 0) lines.push('', verdict.sentences.join(' '));
492
+ lines.push('');
493
+ }
450
494
  if (comparison.confidence === 'low') {
451
495
  lines.push(`- confidence: low, ${comparison.reason}`);
452
496
  lines.push('');