sparkforensics-mcp 0.3.0 → 0.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (62) hide show
  1. package/bin/sparkforensics-mcp.mjs +41 -10
  2. package/package.json +1 -1
  3. package/vendor-core/analyzer.js +156 -48
  4. package/vendor-core/check-coverage.js +88 -0
  5. package/vendor-core/cli/budgets.js +31 -18
  6. package/vendor-core/cli/collect-run.js +76 -31
  7. package/vendor-core/cli/threshold-config.js +28 -0
  8. package/vendor-core/comparison-verdict.js +177 -0
  9. package/vendor-core/core-source-hash.txt +1 -0
  10. package/vendor-core/core-usage-locality.js +56 -2
  11. package/vendor-core/detector-docs.js +58 -0
  12. package/vendor-core/detectors.js +918 -458
  13. package/vendor-core/docs-config.js +0 -36
  14. package/vendor-core/docs-content/chapters/nav-index.json +31 -0
  15. package/vendor-core/docs-content/detection/cstor.md +9 -0
  16. package/vendor-core/docs-site-config.js +3 -0
  17. package/vendor-core/event-handlers.js +170 -6
  18. package/vendor-core/event-schemas.js +21 -0
  19. package/vendor-core/evidence-report.js +421 -112
  20. package/vendor-core/export-data.js +79 -6
  21. package/vendor-core/finding-action-label.js +9 -88
  22. package/vendor-core/finding-filter-predicate.js +9 -0
  23. package/vendor-core/finding-generic-recommendation.js +6 -104
  24. package/vendor-core/finding-names.js +21 -45
  25. package/vendor-core/finding-presentation.js +333 -0
  26. package/vendor-core/finding-tag-help.js +110 -0
  27. package/vendor-core/finding-types.js +361 -0
  28. package/vendor-core/findings-of-type.js +11 -0
  29. package/vendor-core/format-utils.js +92 -27
  30. package/vendor-core/html-export.js +51 -0
  31. package/vendor-core/impact-band.js +21 -8
  32. package/vendor-core/impact-estimator.js +8 -521
  33. package/vendor-core/impact-format.js +114 -0
  34. package/vendor-core/impact-model.js +175 -0
  35. package/vendor-core/ingest.js +2 -0
  36. package/vendor-core/intervals.js +13 -0
  37. package/vendor-core/list-runs.js +2 -3
  38. package/vendor-core/load-vendored.js +70 -5
  39. package/vendor-core/mcp-server-factory.js +14 -10
  40. package/vendor-core/mcp-tools.js +105 -45
  41. package/vendor-core/model-assembler.js +12 -0
  42. package/vendor-core/occupancy.js +1 -1
  43. package/vendor-core/parser-worker.js +1 -1
  44. package/vendor-core/plan-graph-model.js +3 -2
  45. package/vendor-core/plan-node-detail.js +1 -1
  46. package/vendor-core/recommendation-rollup.js +63 -3
  47. package/vendor-core/redact.js +45 -27
  48. package/vendor-core/run-comparison.js +32 -7
  49. package/vendor-core/run-interpretation.js +290 -0
  50. package/vendor-core/run-outcome.js +74 -0
  51. package/vendor-core/run-payload.js +17 -0
  52. package/vendor-core/run-shape.js +40 -0
  53. package/vendor-core/run-verdict.js +353 -0
  54. package/vendor-core/scaling-sim.js +4 -5
  55. package/vendor-core/scorecard-estimates.js +62 -0
  56. package/vendor-core/sql-stages.js +11 -0
  57. package/vendor-core/stage-quantiles.js +2 -0
  58. package/vendor-core/threshold-overrides.js +160 -0
  59. package/vendor-core/threshold-summary.js +11 -33
  60. package/vendor-core/types.js +6 -42
  61. package/vendor-core/wall-clock.js +1 -12
  62. package/vendor-core/wasted-core-hours.js +2 -2
@@ -0,0 +1,175 @@
1
+ // The waste models every detector entry's estimate() builds its ImpactEstimate from: the assumed
2
+ // throughputs, the per-stage measurements behind them and the occupancy clip wrappers. Each
3
+ // finding type's own composition of these lives on its DETECTORS entry, next to its detect().
4
+
5
+ import { nsToMs } from './format-utils.js';
6
+ import {
7
+ estimateSingleStage, estimateMultiStage,
8
+
9
+ } from './occupancy.js';
10
+
11
+ /** What every estimate() is handed: built once per analyze(), and the same object detect() gates
12
+ * its runtime floors against (DetectorCtx.impact), so a floor and the savings displayed for it
13
+ * read one occupancy sweep. */
14
+
15
+
16
+
17
+
18
+
19
+
20
+
21
+ // Assumed shuffle-network throughput per executor link, ~1 Gbps. Starting assumption, unvalidated.
22
+ export const SHUFFLE_THROUGHPUT_BPS = 125_000_000;
23
+ // Assumed disk I/O throughput per executor for spilled data, ~200 MB/s (conservative HDD/SSD blend).
24
+ export const SPILL_IO_THROUGHPUT_BPS = 200_000_000;
25
+
26
+ // A stage's own per-task overhead: task wall time (launch to finish, summed per executor by
27
+ // finalizeStage) minus executorRunTime, i.e. deserialization, result serialization and the
28
+ // launch round trip that coalescing tasks removes. `concurrency` is the stage's achieved task
29
+ // concurrency (task time / stage duration). Null when the stage has no such data or no overhead.
30
+ export function measuredTaskOverhead(stage ) {
31
+ const executorStats = Array.isArray(stage.executorStats)
32
+ ? (stage.executorStats ) : [];
33
+ const taskTimeMs = executorStats.reduce((sum, e) => sum + (e.totalDuration ?? 0), 0);
34
+ const taskCount = stage.taskCount ?? 0;
35
+ const durationMs = (stage.completedAt ?? 0) - (stage.submittedAt ?? 0);
36
+ const overheadMs = taskTimeMs - (stage.executorRunTime ?? 0);
37
+ if (taskTimeMs <= 0 || taskCount <= 0 || durationMs <= 0 || overheadMs <= 0) return null;
38
+ return { perTaskMs: overheadMs / taskCount, concurrency: taskTimeMs / durationMs };
39
+ }
40
+
41
+ // Both constants above are single-device figures (one NIC, one local disk). A stage's shuffle
42
+ // reads and spills are spread over every executor that ran its tasks, each moving its own share
43
+ // in parallel, so the stage's aggregate bandwidth scales with that executor count. Dividing a
44
+ // stage-wide byte total by one device's bandwidth modeled the whole cluster as one link: on 14
45
+ // real logs that claimed up to 1,870s of shuffle time on stages whose tasks measured ~0s of
46
+ // shuffle fetch wait (222 of 284 shuffle findings). No executor data falls back to one device.
47
+ export function stageIoParallelism(stage ) {
48
+ const executors = Array.isArray(stage.executorStats) ? stage.executorStats.length : 0;
49
+ return Math.max(1, executors);
50
+ }
51
+
52
+ // Wall-clock the stage's tasks spent blocked fetching shuffle blocks: fetchWaitTime is a
53
+ // cross-task sum like executorRunTime, so dividing by the stage's average concurrency converts it
54
+ // (the gc estimate's conversion). Null when the stage has no run time or duration to convert with.
55
+ // On 14 real logs the link model claimed 2780s over 284 shuffle findings, 256s once capped at
56
+ // this, with about zero fetch wait on 5 of the 10 non-info ones: reads that overlapped compute
57
+ // stalled nothing.
58
+ export function fetchWaitWallClockMs(stage ) {
59
+ const fetchWaitMs = stage.fetchWaitTime;
60
+ const runTimeMs = stage.executorRunTime ?? 0;
61
+ const durationMs = (stage.completedAt ?? 0) - (stage.submittedAt ?? 0);
62
+ if (typeof fetchWaitMs !== 'number' || runTimeMs <= 0 || durationMs <= 0) return null;
63
+ return fetchWaitMs / (runTimeMs / durationMs);
64
+ }
65
+
66
+ // Wall-clock a stage's wasted (retried) attempts cost it. Each one delayed only its own task, and
67
+ // attempts of different tasks ran side by side: one lost executor fails every task it was running
68
+ // at once (4 wasted attempts of 36.6s each, all first attempts, on a real stage that ran 41 tasks
69
+ // at once, claimed as 146.6s). So the stage lost at most its longest retry chain, the most attempts
70
+ // one task wasted (its highest attempt number + 1) times the mean wasted attempt, or their summed
71
+ // time spread over its slots, whichever is larger. Without a sample of every wasted attempt
72
+ // (retryTaskSamples is capped), the chain isn't known: the summed time.
73
+ export function retryWallClockMs(stage ) {
74
+ const totalMs = (stage.retryWasteMs ) ?? 0;
75
+ const attempts = (stage.wastedAttempts ) ?? 0;
76
+ const samples = Array.isArray(stage.retryTaskSamples)
77
+ ? (stage.retryTaskSamples ) : [];
78
+ if (totalMs <= 0 || attempts <= 0 || samples.length < attempts) return totalMs;
79
+ const chain = samples.reduce((longest, s) => Math.max(longest, (s.attemptNumber ?? 0) + 1), 1);
80
+ const slots = Math.max(1, stage.peakConcurrentTasks ?? 1);
81
+ return Math.min(totalMs, Math.max((chain * totalMs) / attempts, totalMs / slots));
82
+ }
83
+ // Spark's classic recommended shuffle partition size.
84
+ export const IDEAL_BYTES_PER_PARTITION_TASK = 128 * 1024 * 1024;
85
+ // Assumed per-task scheduling/launch overhead: the fallback when a stage lacks the per-executor
86
+ // task-time sums measuredTaskOverhead needs.
87
+ export const TASK_SCHEDULING_OVERHEAD_MS = 50;
88
+ // Assumed per-file open latency (small-file overhead).
89
+ export const FILE_OPEN_OVERHEAD_MS = 10;
90
+ // Assumed broadcast-transfer bandwidth, shared with overBroadcast/underBroadcast.
91
+ export const BROADCAST_BANDWIDTH_BPS = 125_000_000;
92
+ // Assumed per-non-local-task network-fetch penalty, reported as extra core-time.
93
+ export const NETWORK_FETCH_PENALTY_MS = 20;
94
+ // Assumed executor JVM+container startup overhead.
95
+ export const EXECUTOR_STARTUP_OVERHEAD_MS = 15000;
96
+ // Assumed re-read throughput, shared by cachingOpportunity and cacheUtilization.
97
+ export const RE_READ_THROUGHPUT_BPS = 125_000_000;
98
+
99
+ // Below this share of executorRunTime spent on CPU, a stage's tasks were idle, waiting on something
100
+ // outside Spark: on 14 real logs (2026-09-23) every non-Python stage under 1% was a JDBC read, a
101
+ // file listing or a Delta log read, while file writes, which more partitions do parallelize,
102
+ // start at 2%.
103
+ const IDLE_CPU_SHARE_MAX = 0.01;
104
+
105
+ // True when the stage's tasks spent under IDLE_CPU_SHARE_MAX of their run time on CPU. False when
106
+ // the share can't be trusted: no CPU time recorded (older Spark logs omit the metric), or Python
107
+ // code run through PythonRDD, whose worker-process CPU executorCpuTime (the JVM task thread's)
108
+ // never counts (such stages read 0.1% on the same logs while computing).
109
+ export function tasksMostlyIdle(stage ) {
110
+ const runMs = stage.executorRunTime ?? 0;
111
+ const cpuMs = nsToMs(stage.executorCpuTime ?? 0);
112
+ if (runMs <= 0 || cpuMs <= 0) return false;
113
+ if (/PythonRDD/.test(stage.name ?? '') || /org\.apache\.spark\.api\.python\./.test(stage.details ?? '')) return false;
114
+ return cpuMs / runMs < IDLE_CPU_SHARE_MAX;
115
+ }
116
+
117
+ // skew, straggler and stageSlowness claim time off the stage's longest task itself, so the
118
+ // occupancy clip must not floor them at that same task (see estimateSingleStage).
119
+ export const TAIL_CLAIM = { shortensLongestTask: true };
120
+
121
+ // No quantifiable magnitude -> 'informational'; a rawWaste figure with no stage window ->
122
+ // 'resourceOnly'. Never a fake {low:0, high:0}: wallClock is null in both cases.
123
+ export function costOnly(estimateMethod , rawWaste ) {
124
+ return rawWaste
125
+ ? { basis: 'resourceOnly', wallClock: null, estimateMethod, rawWaste }
126
+ : { basis: 'informational', wallClock: null, estimateMethod };
127
+ }
128
+
129
+ /** The estimate() of an entry whose findings have no waste model at all. */
130
+ export function noWasteModel() {
131
+ return costOnly('none');
132
+ }
133
+
134
+ export function singleStageImpact(
135
+ wasteMs ,
136
+ stageId ,
137
+ ctx ,
138
+ estimateMethod ,
139
+ rawWaste ,
140
+ opts ,
141
+ ) {
142
+ const est = estimateSingleStage(wasteMs, stageId, ctx.stages , ctx.occupancy, opts);
143
+ if (est) return { basis: est.basis, wallClock: est.wallClock, estimateMethod, rawWaste };
144
+ return costOnly(estimateMethod, rawWaste); // stage excluded from the sweep (duration <= 0)
145
+ }
146
+
147
+ /** estimateMultiStage over the finding's own stages, or null when every one was excluded from
148
+ * the sweep. */
149
+ export function multiStageImpact(
150
+ stageIds ,
151
+ wasteMsByStage ,
152
+ ctx ,
153
+ estimateMethod ,
154
+ rawWaste ,
155
+ ) {
156
+ const est = estimateMultiStage(stageIds, wasteMsByStage, ctx.stages , ctx.occupancy);
157
+ return est ? { basis: est.basis, wallClock: est.wallClock, estimateMethod, rawWaste } : null;
158
+ }
159
+
160
+ export function stageMappableWasteOrCostOnly(
161
+ wasteMs ,
162
+ stageIds ,
163
+ ctx ,
164
+ ) {
165
+ const rawWaste = wasteMs > 0 ? { value: wasteMs, unit: 'ms' } : undefined;
166
+ if (!stageIds || stageIds.length === 0) {
167
+ return costOnly('modeled', rawWaste);
168
+ }
169
+ // One waste event spread over a span of stages, not N independent wastes: apportion evenly so
170
+ // estimateMultiStage's union cap doesn't absorb the same amount claimed once per stage.
171
+ const perStageWasteMs = wasteMs / stageIds.length;
172
+ const wasteMsByStage = new Map(stageIds.map((id) => [id, perStageWasteMs]));
173
+ // null: every stage excluded from the sweep
174
+ return multiStageImpact(stageIds, wasteMsByStage, ctx, 'modeled', rawWaste) ?? costOnly('modeled', rawWaste);
175
+ }
@@ -13,6 +13,7 @@
13
13
 
14
14
 
15
15
 
16
+
16
17
 
17
18
 
18
19
 
@@ -37,6 +38,7 @@ export function routeMessage(
37
38
  case 'job': handlers.onJob?.(data.data); break;
38
39
  case 'runAggregates': handlers.onRunAggregates?.(data.data); break;
39
40
  case 'stageExecutorMetrics': handlers.onStageExecutorMetrics?.(data.data); break;
41
+ case 'stageSpeculationWaste': handlers.onStageSpeculationWaste?.(data.data); break;
40
42
  case 'done': {
41
43
  const { skippedLines } = data;
42
44
  handlers.onDone?.({ skippedLines });
@@ -0,0 +1,13 @@
1
+ // Interval union shared by the wall-clock breakdown and the savings rollups.
2
+ export function mergeIntervals(intervals ) {
3
+ if (intervals.length === 0) return [];
4
+ const sorted = [...intervals].sort((a, b) => a[0] - b[0]);
5
+ const out = [[sorted[0][0], sorted[0][1]]];
6
+ for (let i = 1; i < sorted.length; i++) {
7
+ const last = out[out.length - 1];
8
+ const cur = sorted[i];
9
+ if (cur[0] <= last[1]) last[1] = Math.max(last[1], cur[1]);
10
+ else out.push([cur[0], cur[1]]);
11
+ }
12
+ return out;
13
+ }
@@ -82,9 +82,8 @@ export function applyFiltersAndCap(entries , filters
82
82
  return filters.redact ? { ...capped, runs: redactRunListEntries(capped.runs) } : capped;
83
83
  }
84
84
 
85
- // Local counterpart to redact.ts's redactAppIdentity, sized for a listing of many distinct apps
86
- // rather than the single app that tool works over: redactAppIdentity has no way to keep two
87
- // entries sharing an appId in sync, so this builds its own stable per-distinct-appId pseudonym map
85
+ // Listing-wide app identity redaction: redact.ts's redactors each work over one run, with no way
86
+ // to keep two listing entries sharing an appId in sync, so this builds its own stable per-distinct-appId pseudonym map
88
87
  // (numeric-aware sort, same scheme as redact.ts's buildMap) across the whole result set, two
89
88
  // attempts of the same app redact to the same identity, and applies it to every identity-bearing
90
89
  // field: appId, name (a listing's human-readable name is as identifying as the id itself), and the
@@ -1,16 +1,81 @@
1
1
  // Plain JS: scripts/vendor-core.mjs copies it byte-for-byte into vendor-core/, so this file exists
2
2
  // at both vendor-core/load-vendored.js (published) and core/src/load-vendored.js (dev). Each
3
- // package's bin bootstraps by locating this file first, then uses the exports below for every other
4
- // core module, so the resolution logic lives in one place instead of per entry point.
5
- import { existsSync } from 'node:fs';
6
- import { join } from 'node:path';
3
+ // package's bin bootstraps by locating this file first (the core/src copy when it exists, so the
4
+ // staleness check below is always current code), then uses the exports below for every other core
5
+ // module, so the resolution logic lives in one place instead of per entry point.
6
+ import { createHash } from 'node:crypto';
7
+ import { existsSync, readdirSync, readFileSync } from 'node:fs';
8
+ import { join, relative } from 'node:path';
7
9
  import { pathToFileURL } from 'node:url';
8
10
 
11
+ // Written into vendor-core/ by scripts/vendor-core.mjs: the coreSourceHash of the core/src it was
12
+ // built from, so a monorepo run can tell a leftover vendored copy from a current one.
13
+ export const SOURCE_HASH_FILE = 'core-source-hash.txt';
14
+
15
+ // docs-content/ is generated from a pinned upstream commit, not analysis code.
16
+ const UNHASHED_TOP_LEVEL = new Set(['docs-content']);
17
+
18
+ /** A content hash of every analysis source file under core/src (paths and bytes, sorted). */
19
+ export function coreSourceHash(coreSrcDir) {
20
+ const hash = createHash('sha256');
21
+ const walk = (dir) => {
22
+ const entries = readdirSync(dir, { withFileTypes: true }).sort((a, b) => (a.name < b.name ? -1 : a.name > b.name ? 1 : 0));
23
+ for (const entry of entries) {
24
+ if (dir === coreSrcDir && UNHASHED_TOP_LEVEL.has(entry.name)) continue;
25
+ const path = join(dir, entry.name);
26
+ if (entry.isDirectory()) walk(path);
27
+ else if (entry.isFile()) hash.update(`${relative(coreSrcDir, path)}\0`).update(readFileSync(path)).update('\0');
28
+ }
29
+ };
30
+ walk(coreSrcDir);
31
+ return hash.digest('hex');
32
+ }
33
+
34
+ // One decision per package per process, so the hash runs and the warning prints once.
35
+ const useVendoredByPkg = new Map();
36
+
37
+ // A published install has only vendor-core/. In the monorepo, vendor-core/ is a leftover from a
38
+ // local `npm pack` (it is rebuilt only at prepack), so it is used only while it still matches
39
+ // core/src; otherwise the run says so on stderr and loads core/src, rather than silently running
40
+ // outdated detectors.
41
+ function useVendored(pkgDir) {
42
+ if (useVendoredByPkg.has(pkgDir)) return useVendoredByPkg.get(pkgDir);
43
+ const vendorDir = join(pkgDir, 'vendor-core');
44
+ const srcDir = join(pkgDir, '..', 'core', 'src');
45
+ let use = existsSync(vendorDir);
46
+ if (use && existsSync(join(srcDir, 'load-vendored.js'))) {
47
+ const stampPath = join(vendorDir, SOURCE_HASH_FILE);
48
+ const stamp = existsSync(stampPath) ? readFileSync(stampPath, 'utf8').trim() : null;
49
+ if (stamp !== coreSourceHash(srcDir)) {
50
+ process.stderr.write(
51
+ `sparkforensics: ${vendorDir} was built from older packages/core sources, so it would run outdated analysis. `
52
+ + `Using packages/core/src instead; rebuild it with \`node scripts/vendor-core.mjs ${pkgDir}\` or delete it.\n`,
53
+ );
54
+ use = false;
55
+ }
56
+ }
57
+ useVendoredByPkg.set(pkgDir, use);
58
+ return use;
59
+ }
60
+
9
61
  // moduleName is a path relative to core/src without extension, e.g. 'cli/collect-run'. Pass
10
62
  // srcExt: 'js' for modules already plain JS in core/src (e.g. 'proxy') rather than TypeScript.
11
63
  export function resolveVendored(pkgDir, moduleName, { srcExt = 'ts' } = {}) {
12
64
  const vendored = join(pkgDir, 'vendor-core', `${moduleName}.js`);
13
- return existsSync(vendored) ? vendored : join(pkgDir, '..', 'core', 'src', `${moduleName}.${srcExt}`);
65
+ return useVendored(pkgDir) && existsSync(vendored) ? vendored : join(pkgDir, '..', 'core', 'src', `${moduleName}.${srcExt}`);
66
+ }
67
+
68
+ /** The build id of the core this package runs: the stamped source hash of its vendor-core/ when
69
+ * that copy is in use, else the hash of core/src computed now. The same `coreSourceHash` the web
70
+ * build stamps into its bundle, so equal ids mean the same analysis code. 'dev' when neither
71
+ * exists. */
72
+ export function coreBuildId(pkgDir) {
73
+ if (useVendored(pkgDir)) {
74
+ const stampPath = join(pkgDir, 'vendor-core', SOURCE_HASH_FILE);
75
+ if (existsSync(stampPath)) return readFileSync(stampPath, 'utf8').trim();
76
+ }
77
+ const srcDir = join(pkgDir, '..', 'core', 'src');
78
+ return existsSync(srcDir) ? coreSourceHash(srcDir) : 'dev';
14
79
  }
15
80
 
16
81
  export async function loadVendored(pkgDir, moduleName, opts) {
@@ -5,6 +5,7 @@ import {
5
5
  resolveOrCreateRun, diagnoseRun, getRunSummary, compareRuns, getFindingEvidence, getFindingDocumentation, getReferenceDoc, evaluateBudgetsForRun,
6
6
  } from './mcp-tools.js';
7
7
  import { listRuns } from './list-runs.js';
8
+
8
9
 
9
10
  const sourceSchema = z.union([
10
11
  z.object({ path: z.string() }),
@@ -59,7 +60,9 @@ function toolResult (promise )
59
60
  return promise.then((value) => toCallToolResult(JSON.stringify(value), value ), toolErrorResult);
60
61
  }
61
62
 
62
- export function createMcpServer() {
63
+ /** `thresholds`: the bin's --thresholds overrides, applied to every analyzing tool for the life
64
+ * of the server. Omitted, every tool runs the specification defaults. */
65
+ export function createMcpServer({ thresholds } = {}) {
63
66
  const server = new McpServer({ name: 'sparkforensics', version: '1.0.0' });
64
67
 
65
68
  server.registerTool('list_runs', {
@@ -68,7 +71,7 @@ export function createMcpServer() {
68
71
  }, (params) => toolResult(listRuns(params)));
69
72
 
70
73
  server.registerTool('diagnose_run', {
71
- description: 'Diagnose a Spark run: thresholded findings with remediation text, an impact-ranked fix recommendation rollup, and clean-check status.',
74
+ description: 'Diagnose a Spark run: the dashboard verdict (title, summary, and the top places to look, ranked by potential savings), thresholded findings with remediation text, an impact-ranked fix recommendation rollup, clean-check status, and the checks the log lacked the data to run. When the server was started with --thresholds, findings and clean checks from a tuned detector carry tunedThresholds, and their impact estimates are uncalibrated.',
72
75
  inputSchema: {
73
76
  ...runRefSchema, redact: z.boolean().optional(),
74
77
  include: z.array(z.enum(['summary', 'evidenceAvailability', 'detectors'])).optional(),
@@ -77,14 +80,14 @@ export function createMcpServer() {
77
80
  },
78
81
  }, ({ source, runId, redact, include, format, impactBand, type, stageId }) => toolResultWithMarkdown(
79
82
  resolveOrCreateRun({ source, runId }).then(({ runId: id }) =>
80
- diagnoseRun(id, { redact, include, markdown: format === 'md', impactBand, type, stageId })),
83
+ diagnoseRun(id, { redact, include, markdown: format === 'md', impactBand, type, stageId, thresholds })),
81
84
  ));
82
85
 
83
86
  server.registerTool('get_run_summary', {
84
- description: 'App/stage/job/sql counts and duration for a run, no findings.',
87
+ description: 'App/stage/job/sql counts, duration, and how the run ended (failed/total jobs and Spark\'s first-line failure reason) for a run, no findings.',
85
88
  inputSchema: { ...runRefSchema, redact: z.boolean().optional() },
86
89
  }, ({ source, runId, redact }) => toolResult(
87
- resolveOrCreateRun({ source, runId }).then(({ runId: id }) => getRunSummary(id, { redact })),
90
+ resolveOrCreateRun({ source, runId }).then(({ runId: id }) => getRunSummary(id, { redact, thresholds })),
88
91
  ));
89
92
 
90
93
  server.registerTool('compare_runs', {
@@ -96,17 +99,17 @@ export function createMcpServer() {
96
99
  ...formatSchema,
97
100
  },
98
101
  }, ({ runIdA, sourceA, runIdB, sourceB, redact, format }) => toolResultWithMarkdown(
99
- compareRuns({ runId: runIdA, source: sourceA }, { runId: runIdB, source: sourceB }, { redact, markdown: format === 'md' }),
102
+ compareRuns({ runId: runIdA, source: sourceA }, { runId: runIdB, source: sourceB }, { redact, markdown: format === 'md', thresholds }),
100
103
  ));
101
104
 
102
105
  server.registerTool('evaluate_budgets', {
103
- description: 'Evaluate a run (optionally against a second run for regression budgets) against pass/fail thresholds.',
106
+ description: 'Evaluate a run against pass/fail thresholds, optionally against a baseline run for regression budgets. With two runs, absolute budgets apply to the candidate (sourceB/runIdB), matching the CLI. A run with no ApplicationEnd always adds an inconclusive run-complete result.',
104
107
  inputSchema: {
105
- source: sourceSchema.optional().describe('The run to evaluate. Also the regression baseline when sourceB/runIdB is given.'),
108
+ source: sourceSchema.optional().describe('The run to evaluate. When sourceB/runIdB is given, this is the regression baseline instead, and the absolute budgets apply to sourceB/runIdB.'),
106
109
  runId: runRefSchema.runId.describe('Same as `source`, referencing an already-resolved run by id.'),
107
110
  maxRuntimeMs: z.number().optional(), maxSpillGb: z.number().optional(), maxSkewRatio: z.number().optional(),
108
111
  maxFailedTaskRatePct: z.number().optional(), minEfficiencyPct: z.number().optional(),
109
- sourceB: sourceSchema.optional().describe('Optional candidate run, compared against source/runId as the regression baseline (maxRegressionPct/failOnIntroduced).'),
112
+ sourceB: sourceSchema.optional().describe('Optional candidate run, compared against source/runId as the regression baseline (maxRegressionPct/failOnIntroduced). When given, the absolute budgets (maxRuntimeMs etc.) are evaluated on this run.'),
110
113
  runIdB: secondRunRefSchema.runIdB.describe('Same as `sourceB`, referencing an already-resolved run by id.'),
111
114
  maxRegressionPct: z.number().optional(), regressionMetric: z.string().optional(), failOnIntroduced: z.string().optional(),
112
115
  },
@@ -118,13 +121,14 @@ export function createMcpServer() {
118
121
  { source, runId },
119
122
  { maxRuntimeMs, maxSpillGb, maxSkewRatio, maxFailedTaskRatePct, minEfficiencyPct, maxRegressionPct, regressionMetric, failOnIntroduced },
120
123
  (runIdB || sourceB) ? { runId: runIdB, source: sourceB } : undefined,
124
+ { thresholds },
121
125
  )));
122
126
 
123
127
  server.registerTool('get_finding_evidence', {
124
128
  description: 'Raw evidence bundle backing one finding, for drill-down after diagnose_run.',
125
129
  inputSchema: { runId: z.string(), findingId: z.string(), redact: z.boolean().optional() },
126
130
  }, ({ runId, findingId, redact }) => toolResult(
127
- Promise.resolve().then(() => getFindingEvidence(runId, findingId, { redact })),
131
+ Promise.resolve().then(() => getFindingEvidence(runId, findingId, { redact, thresholds })),
128
132
  ));
129
133
 
130
134
  server.registerTool('get_finding_documentation', {