sparkforensics-mcp 0.3.0 → 0.5.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (92) hide show
  1. package/bin/sparkforensics-mcp.mjs +41 -10
  2. package/package.json +1 -1
  3. package/vendor-core/allocation.js +106 -0
  4. package/vendor-core/analyzer.js +168 -60
  5. package/vendor-core/check-coverage.js +88 -0
  6. package/vendor-core/cli/budgets.js +54 -27
  7. package/vendor-core/cli/collect-run.js +84 -32
  8. package/vendor-core/cli/regression-budgets.js +83 -0
  9. package/vendor-core/cli/threshold-config.js +28 -0
  10. package/vendor-core/comparison-verdict.js +177 -0
  11. package/vendor-core/core-source-hash.txt +1 -0
  12. package/vendor-core/core-usage-locality.js +56 -2
  13. package/vendor-core/detector-docs.js +58 -0
  14. package/vendor-core/detectors.js +1094 -500
  15. package/vendor-core/docs-config.js +0 -36
  16. package/vendor-core/docs-content/chapters/nav-index.json +31 -0
  17. package/vendor-core/docs-content/detection/cache.md +3 -2
  18. package/vendor-core/docs-content/detection/cfg.md +9 -8
  19. package/vendor-core/docs-content/detection/chrn.md +1 -2
  20. package/vendor-core/docs-content/detection/cold.md +4 -2
  21. package/vendor-core/docs-content/detection/cstor.md +9 -0
  22. package/vendor-core/docs-content/detection/fail.md +3 -2
  23. package/vendor-core/docs-content/detection/gc.md +3 -2
  24. package/vendor-core/docs-content/detection/host.md +2 -1
  25. package/vendor-core/docs-content/detection/local.md +1 -1
  26. package/vendor-core/docs-content/detection/mem.md +5 -2
  27. package/vendor-core/docs-content/detection/plan.md +2 -1
  28. package/vendor-core/docs-content/detection/sfail.md +2 -1
  29. package/vendor-core/docs-content/detection/shape.md +5 -4
  30. package/vendor-core/docs-content/detection/skew.md +3 -1
  31. package/vendor-core/docs-content/detection/slow.md +2 -2
  32. package/vendor-core/docs-content/detection/spec.md +2 -3
  33. package/vendor-core/docs-content/detection/spill.md +1 -1
  34. package/vendor-core/docs-site-config.js +3 -0
  35. package/vendor-core/effective-conf.js +107 -0
  36. package/vendor-core/efficiency-model.js +8 -6
  37. package/vendor-core/event-handlers.js +321 -44
  38. package/vendor-core/event-schemas.js +23 -0
  39. package/vendor-core/evidence-report.js +432 -115
  40. package/vendor-core/export-data.js +79 -6
  41. package/vendor-core/finding-action-label.js +9 -88
  42. package/vendor-core/finding-filter-predicate.js +9 -0
  43. package/vendor-core/finding-generic-recommendation.js +26 -105
  44. package/vendor-core/finding-names.js +28 -45
  45. package/vendor-core/finding-presentation.js +368 -0
  46. package/vendor-core/finding-tag-help.js +110 -0
  47. package/vendor-core/finding-types.js +373 -0
  48. package/vendor-core/findings-of-type.js +11 -0
  49. package/vendor-core/format-utils.js +96 -30
  50. package/vendor-core/html-export.js +51 -0
  51. package/vendor-core/impact-band.js +21 -8
  52. package/vendor-core/impact-estimator.js +25 -520
  53. package/vendor-core/impact-format.js +115 -0
  54. package/vendor-core/impact-model.js +197 -0
  55. package/vendor-core/ingest.js +6 -2
  56. package/vendor-core/intervals.js +13 -0
  57. package/vendor-core/list-runs.js +7 -5
  58. package/vendor-core/load-vendored.js +70 -5
  59. package/vendor-core/mcp-server-factory.js +14 -10
  60. package/vendor-core/mcp-tools.js +105 -45
  61. package/vendor-core/model-assembler.js +35 -1
  62. package/vendor-core/occupancy.js +1 -1
  63. package/vendor-core/parser-worker.js +2 -2
  64. package/vendor-core/plan-graph-model.js +3 -2
  65. package/vendor-core/plan-node-detail.js +1 -1
  66. package/vendor-core/proxy.js +3 -1
  67. package/vendor-core/python-stage.js +25 -0
  68. package/vendor-core/recommendation-rollup.js +70 -3
  69. package/vendor-core/redact.js +96 -37
  70. package/vendor-core/remediation.js +20 -0
  71. package/vendor-core/run-comparison.js +73 -29
  72. package/vendor-core/run-interpretation.js +291 -0
  73. package/vendor-core/run-metrics.js +198 -0
  74. package/vendor-core/run-outcome.js +74 -0
  75. package/vendor-core/run-payload.js +17 -0
  76. package/vendor-core/run-shape.js +40 -0
  77. package/vendor-core/run-totals.js +24 -0
  78. package/vendor-core/run-verdict.js +352 -0
  79. package/vendor-core/scaling-sim.js +4 -5
  80. package/vendor-core/scorecard-estimates.js +63 -0
  81. package/vendor-core/session-snapshot.js +7 -0
  82. package/vendor-core/shs-schemas.js +2 -2
  83. package/vendor-core/spark-memory.js +17 -0
  84. package/vendor-core/sql-stages.js +11 -0
  85. package/vendor-core/stage-plan-nodes.js +18 -0
  86. package/vendor-core/stage-quantiles.js +6 -0
  87. package/vendor-core/threshold-overrides.js +160 -0
  88. package/vendor-core/threshold-summary.js +11 -33
  89. package/vendor-core/types.js +54 -42
  90. package/vendor-core/wall-clock.js +1 -12
  91. package/vendor-core/wasted-core-hours.js +12 -9
  92. package/vendor-core/write-targets.js +312 -0
@@ -0,0 +1,115 @@
1
+ // Savings figures as every surface prints them: the dashboard's widgets and verdict, and the
2
+ // CLI/MCP evidence report. One formatter set, so units and rounding never drift between paths.
3
+ import { fmtMs, formatRawWaste, formatWallClockRange, readsAsZero } from './format-utils.js';
4
+
5
+
6
+ /** A finding's one-line savings figure: the wall-clock range for a time-based
7
+ * finding, the raw resource figure for a `resourceOnly` one, or nothing for a
8
+ * purely informational estimate or a raw figure that rounds to zero ("0.0
9
+ * core-h" reads as a measured nothing). */
10
+ export function impactFigure(finding ) {
11
+ const estimate = finding.impactEstimate;
12
+ if (!estimate) return null;
13
+ if (estimate.wallClock) return formatWallClockRange(estimate.wallClock.low, estimate.wallClock.high);
14
+ if (estimate.rawWaste) {
15
+ const text = formatRawWaste(estimate.rawWaste);
16
+ return readsAsZero(text) ? null : text;
17
+ }
18
+ return null;
19
+ }
20
+
21
+ /** What a raw-waste figure counts, by its unit, as the words that follow it: an idle core
22
+ * figure is capacity no task ran on, not core time. */
23
+ export function rawWasteMeaning(rawWaste ) {
24
+ switch (rawWaste?.unit) {
25
+ case 'mbSeconds': return 'of unused executor memory';
26
+ case 'coreHours':
27
+ case 'coreMs': return rawWaste.idle ? 'of idle core capacity' : 'of core time';
28
+ case 'bytes': return 'of extra data written';
29
+ case 'ms': return 'of task time';
30
+ default: return null;
31
+ }
32
+ }
33
+
34
+ /** What a savings figure counts, as the words that follow it: run time for a
35
+ * wall-clock claim, or the resource a cost-only (`resourceOnly`) figure
36
+ * measures. A time figure and a capacity figure look alike ("58.6s",
37
+ * "0.7 core-h") but only the first shortens the run. Null when the finding
38
+ * shows no figure. */
39
+ export function savingsMeaning(finding ) {
40
+ const estimate = finding.impactEstimate;
41
+ if (!estimate) return null;
42
+ if (estimate.wallClock) return 'of run time';
43
+ return rawWasteMeaning(estimate.rawWaste);
44
+ }
45
+
46
+ /** A finding's "Potential savings" figure as the widget board shows it: the
47
+ * wall-clock range, or the raw waste only when there is no wall-clock claim,
48
+ * never both (they would read as two competing numbers). A zero-value
49
+ * estimate is suppressed like an informational one: "0s" reads as a measured
50
+ * figure. `meaning` says what the shown figure counts. */
51
+ export function impactEstimateFigure(estimate ) {
52
+ if (!estimate) return null;
53
+ const highText = estimate.wallClock && estimate.wallClock.high > 0 ? fmtMs(estimate.wallClock.high) : null;
54
+ if (highText && !readsAsZero(highText)) {
55
+ return { text: formatWallClockRange(estimate.wallClock .low, estimate.wallClock .high), meaning: 'of run time' };
56
+ }
57
+ const rawWasteText = estimate.rawWaste && estimate.rawWaste.value > 0 ? formatRawWaste(estimate.rawWaste) : null;
58
+ if (rawWasteText && !readsAsZero(rawWasteText)) return { text: rawWasteText, meaning: rawWasteMeaning(estimate.rawWaste) };
59
+ return null;
60
+ }
61
+
62
+ /** Compact single-value form for dense lists (the stage table's finding chips,
63
+ * the stage dialog): the high-end wall-clock figure, or the raw-waste figure
64
+ * when there's no wall-clock claim, or `null` for a purely informational or a
65
+ * zero-value estimate (a `0s` in the spot a real estimate goes would read as a
66
+ * measured nothing). */
67
+ export function impactEstimateCompact(estimate ) {
68
+ if (!estimate) return null;
69
+ if (estimate.wallClock) {
70
+ if (estimate.wallClock.high <= 0) return null;
71
+ const text = fmtMs(estimate.wallClock.high);
72
+ return readsAsZero(text) ? null : text;
73
+ }
74
+ if (estimate.rawWaste) {
75
+ if (estimate.rawWaste.value <= 0) return null;
76
+ const text = formatRawWaste(estimate.rawWaste);
77
+ return readsAsZero(text) ? null : text;
78
+ }
79
+ return null;
80
+ }
81
+
82
+ /** How a step's savings figure was derived, in one plain sentence for
83
+ * Advanced view: the estimate method, whether the stage ran alone (a
84
+ * near-point figure) or shared the cluster (a floor and an optimistic high),
85
+ * and the raw waste behind it. Null when the finding carries no estimate
86
+ * model (`estimateMethod: 'none'`), no estimate at all, or a figure that
87
+ * reads as zero (the step shows no savings then either). Uses the same
88
+ * formatting and zero rules as the step's own savings figure, which it does
89
+ * not repeat: the step already shows it. */
90
+ export function estimateProvenance(finding ) {
91
+ const estimate = finding.impactEstimate;
92
+ if (!estimate || estimate.estimateMethod === 'none') return null;
93
+ const method = `${estimate.estimateMethod[0].toUpperCase()}${estimate.estimateMethod.slice(1)}`;
94
+ const rawWaste = estimate.rawWaste && estimate.rawWaste.value > 0 ? estimate.rawWaste : null;
95
+ const raw = rawWaste && !readsAsZero(formatRawWaste(rawWaste)) ? formatRawWaste(rawWaste) : null;
96
+ const wallClock = estimate.wallClock;
97
+ if (estimate.basis === 'resourceOnly') {
98
+ return raw ? `${method}; ${raw} wasted, which may not shorten the run.` : null;
99
+ }
100
+ if (!wallClock || wallClock.high <= 0) return null;
101
+ const highText = formatWallClockRange(wallClock.high, wallClock.high);
102
+ if (readsAsZero(highText)) return null;
103
+ let rawNote = '';
104
+ if (raw && rawWaste .unit !== 'ms') rawNote = ` Resource waste measured: ${raw}.`;
105
+ else if (raw && rawWaste .value > wallClock.high && raw !== highText) rawNote = ` Raw waste before the floor clipped it: ${raw}.`;
106
+ if (estimate.basis === 'serial') {
107
+ return `${method}; the stage ran alone, so this is close to a point estimate.${rawNote}`;
108
+ }
109
+ if (estimate.basis === 'contended') {
110
+ const lowText = formatWallClockRange(wallClock.low, wallClock.low);
111
+ const spread = lowText === highText ? 'its floor and high agree' : `${lowText} is the floor, ${highText} if the fix fully lands`;
112
+ return `${method}; the stage shared the cluster: ${spread}.${rawNote}`;
113
+ }
114
+ return null;
115
+ }
@@ -0,0 +1,197 @@
1
+ // The waste models every detector entry's estimate() builds its ImpactEstimate from: the assumed
2
+ // throughputs, the per-stage measurements behind them and the occupancy clip wrappers. Each
3
+ // finding type's own composition of these lives on its DETECTORS entry, next to its detect().
4
+
5
+ import { isPythonStage } from './python-stage.js';
6
+ import { nsToMs } from './format-utils.js';
7
+ import {
8
+ estimateSingleStage, estimateMultiStage,
9
+
10
+ } from './occupancy.js';
11
+
12
+ /** What every estimate() is handed: built once per analyze(), and the same object detect() gates
13
+ * its runtime floors against (DetectorCtx.impact), so a floor and the savings displayed for it
14
+ * read one occupancy sweep. */
15
+
16
+
17
+
18
+
19
+
20
+
21
+
22
+
23
+
24
+ // Assumed shuffle-network throughput per executor link, ~1 Gbps. Starting assumption, unvalidated.
25
+ export const SHUFFLE_THROUGHPUT_BPS = 125_000_000;
26
+ // Assumed disk I/O throughput per executor for spilled data, ~200 MB/s (conservative HDD/SSD blend).
27
+ export const SPILL_IO_THROUGHPUT_BPS = 200_000_000;
28
+
29
+ // A stage's own per-task overhead: task wall time (launch to finish, summed per executor by
30
+ // finalizeStage) minus executorRunTime, i.e. deserialization, result serialization and the
31
+ // launch round trip that coalescing tasks removes. `concurrency` is the stage's achieved task
32
+ // concurrency (task time / stage duration). Null when the stage has no such data or no overhead.
33
+ export function measuredTaskOverhead(stage ) {
34
+ const executorStats = Array.isArray(stage.executorStats)
35
+ ? (stage.executorStats ) : [];
36
+ const taskTimeMs = executorStats.reduce((sum, e) => sum + (e.totalDuration ?? 0), 0);
37
+ const taskCount = stage.taskCount ?? 0;
38
+ const durationMs = (stage.completedAt ?? 0) - (stage.submittedAt ?? 0);
39
+ const overheadMs = taskTimeMs - (stage.executorRunTime ?? 0);
40
+ if (taskTimeMs <= 0 || taskCount <= 0 || durationMs <= 0 || overheadMs <= 0) return null;
41
+ return { perTaskMs: overheadMs / taskCount, concurrency: taskTimeMs / durationMs };
42
+ }
43
+
44
+ // Both constants above are single-device figures (one NIC, one local disk). A stage's shuffle
45
+ // reads and spills are spread over every executor that ran its tasks, each moving its own share
46
+ // in parallel, so the stage's aggregate bandwidth scales with that executor count. Dividing a
47
+ // stage-wide byte total by one device's bandwidth modeled the whole cluster as one link: on 14
48
+ // real logs that claimed up to 1,870s of shuffle time on stages whose tasks measured ~0s of
49
+ // shuffle fetch wait (222 of 284 shuffle findings). No executor data falls back to one device.
50
+ export function stageIoParallelism(stage ) {
51
+ const executors = Array.isArray(stage.executorStats) ? stage.executorStats.length : 0;
52
+ return Math.max(1, executors);
53
+ }
54
+
55
+ // Wall-clock the stage's tasks spent blocked fetching shuffle blocks: fetchWaitTime is a
56
+ // cross-task sum like executorRunTime, so dividing by the stage's average concurrency converts it
57
+ // (the gc estimate's conversion). Null when the stage has no run time or duration to convert with.
58
+ // On 14 real logs the link model claimed 2780s over 284 shuffle findings, 256s once capped at
59
+ // this, with about zero fetch wait on 5 of the 10 non-info ones: reads that overlapped compute
60
+ // stalled nothing.
61
+ export function fetchWaitWallClockMs(stage ) {
62
+ const fetchWaitMs = stage.fetchWaitTime;
63
+ const runTimeMs = stage.executorRunTime ?? 0;
64
+ const durationMs = (stage.completedAt ?? 0) - (stage.submittedAt ?? 0);
65
+ if (typeof fetchWaitMs !== 'number' || runTimeMs <= 0 || durationMs <= 0) return null;
66
+ return fetchWaitMs / (runTimeMs / durationMs);
67
+ }
68
+
69
+ // Wall-clock a stage's wasted (retried) attempts cost it. Each one delayed only its own task, and
70
+ // attempts of different tasks ran side by side: one lost executor fails every task it was running
71
+ // at once (4 wasted attempts of 36.6s each, all first attempts, on a real stage that ran 41 tasks
72
+ // at once, claimed as 146.6s). So the stage lost at most its longest retry chain, the most attempts
73
+ // one task wasted (its highest attempt number + 1) times the mean wasted attempt, or their summed
74
+ // time spread over its slots, whichever is larger. Without a sample of every wasted attempt
75
+ // (retryTaskSamples is capped), the chain isn't known: the summed time.
76
+ export function retryWallClockMs(stage ) {
77
+ const totalMs = (stage.retryWasteMs ) ?? 0;
78
+ const attempts = (stage.wastedAttempts ) ?? 0;
79
+ const samples = Array.isArray(stage.retryTaskSamples)
80
+ ? (stage.retryTaskSamples ) : [];
81
+ if (totalMs <= 0 || attempts <= 0 || samples.length < attempts) return totalMs;
82
+ const chain = samples.reduce((longest, s) => Math.max(longest, (s.attemptNumber ?? 0) + 1), 1);
83
+ const slots = Math.max(1, stage.peakConcurrentTasks ?? 1);
84
+ return Math.min(totalMs, Math.max((chain * totalMs) / attempts, totalMs / slots));
85
+ }
86
+ // Spark's classic recommended shuffle partition size.
87
+ export const IDEAL_BYTES_PER_PARTITION_TASK = 128 * 1024 * 1024;
88
+ // Assumed per-task scheduling/launch overhead: the fallback when a stage lacks the per-executor
89
+ // task-time sums measuredTaskOverhead needs.
90
+ export const TASK_SCHEDULING_OVERHEAD_MS = 50;
91
+ // Assumed per-file open latency (small-file overhead).
92
+ export const FILE_OPEN_OVERHEAD_MS = 10;
93
+ // Assumed broadcast-transfer bandwidth, shared with overBroadcast/underBroadcast.
94
+ export const BROADCAST_BANDWIDTH_BPS = 125_000_000;
95
+ // Assumed per-non-local-task network-fetch penalty, reported as extra core-time.
96
+ export const NETWORK_FETCH_PENALTY_MS = 20;
97
+ // Assumed executor JVM+container startup overhead.
98
+ export const EXECUTOR_STARTUP_OVERHEAD_MS = 15000;
99
+ // Assumed re-read throughput, shared by cachingOpportunity and cacheUtilization.
100
+ export const RE_READ_THROUGHPUT_BPS = 125_000_000;
101
+
102
+ // Below this share of executorRunTime spent on CPU, a stage's tasks were idle, waiting on something
103
+ // outside Spark: on 14 real logs (2026-09-23) every non-Python stage under 1% was a JDBC read, a
104
+ // file listing or a Delta log read, while file writes, which more partitions do parallelize,
105
+ // start at 2%.
106
+ const IDLE_CPU_SHARE_MAX = 0.01;
107
+
108
+ // True when the stage's tasks spent under IDLE_CPU_SHARE_MAX of their run time on CPU. False when
109
+ // the share can't be trusted: no CPU time recorded (older Spark logs omit the metric), or Python
110
+ // code run through a Python worker (isPythonStage: a PythonRDD stage or a Python UDF operator in
111
+ // its SQL plan), whose worker-process CPU executorCpuTime (the JVM task thread's) never counts
112
+ // (such stages read 0.1% on the same logs while computing).
113
+ export function tasksMostlyIdle(stage , sql = new Map()) {
114
+ const runMs = stage.executorRunTime ?? 0;
115
+ const cpuMs = nsToMs(stage.executorCpuTime ?? 0);
116
+ if (runMs <= 0 || cpuMs <= 0) return false;
117
+ if (isPythonStage(stage, sql)) return false;
118
+ return cpuMs / runMs < IDLE_CPU_SHARE_MAX;
119
+ }
120
+
121
+ // skew, straggler and stageSlowness claim time off the stage's longest task itself, so the
122
+ // occupancy clip must not floor them at that same task (see estimateSingleStage).
123
+ export const TAIL_CLAIM = { shortensLongestTask: true };
124
+
125
+ // No quantifiable magnitude -> 'informational'; a rawWaste figure with no stage window ->
126
+ // 'resourceOnly'. Never a fake {low:0, high:0}: wallClock is null in both cases.
127
+ export function costOnly(estimateMethod , rawWaste ) {
128
+ return rawWaste
129
+ ? { basis: 'resourceOnly', wallClock: null, estimateMethod, rawWaste }
130
+ : { basis: 'informational', wallClock: null, estimateMethod };
131
+ }
132
+
133
+ /** The estimate() of an entry whose findings have no waste model at all. */
134
+ export function noWasteModel() {
135
+ return costOnly('none');
136
+ }
137
+
138
+ export function singleStageImpact(
139
+ wasteMs ,
140
+ stageId ,
141
+ ctx ,
142
+ estimateMethod ,
143
+ rawWaste ,
144
+ opts ,
145
+ ) {
146
+ const est = estimateSingleStage(wasteMs, stageId, ctx.stages , ctx.occupancy, opts);
147
+ if (est) return { basis: est.basis, wallClock: est.wallClock, estimateMethod, rawWaste };
148
+ return costOnly(estimateMethod, rawWaste); // stage excluded from the sweep (duration <= 0)
149
+ }
150
+
151
+ /** estimateMultiStage over the finding's own stages, or null when every one was excluded from
152
+ * the sweep. */
153
+ export function multiStageImpact(
154
+ stageIds ,
155
+ wasteMsByStage ,
156
+ ctx ,
157
+ estimateMethod ,
158
+ rawWaste ,
159
+ ) {
160
+ const est = estimateMultiStage(stageIds, wasteMsByStage, ctx.stages , ctx.occupancy);
161
+ return est ? { basis: est.basis, wallClock: est.wallClock, estimateMethod, rawWaste } : null;
162
+ }
163
+
164
+ export function stageMappableWasteOrCostOnly(
165
+ wasteMs ,
166
+ stageIds ,
167
+ ctx ,
168
+ ) {
169
+ const rawWaste = wasteMs > 0 ? { value: wasteMs, unit: 'ms' } : undefined;
170
+ if (!stageIds || stageIds.length === 0) {
171
+ return costOnly('modeled', rawWaste);
172
+ }
173
+ // One waste event spread over a span of stages, not N independent wastes: apportion evenly so
174
+ // estimateMultiStage's union cap doesn't absorb the same amount claimed once per stage.
175
+ const perStageWasteMs = wasteMs / stageIds.length;
176
+ const wasteMsByStage = new Map(stageIds.map((id) => [id, perStageWasteMs]));
177
+ // null: every stage excluded from the sweep
178
+ return multiStageImpact(stageIds, wasteMsByStage, ctx, 'modeled', rawWaste) ?? costOnly('modeled', rawWaste);
179
+ }
180
+
181
+ // Finding types whose raw figure is busy core time read straight from the log: gc's jvmGCTime
182
+ // (coreMs) and the discarded speculative or retried attempts' run time ('ms' cross-task sums).
183
+ // Any other core figure is idle capacity, or modeled on an assumed constant (coreLocality's
184
+ // per-task fetch penalty, autoscalingChurn's executor-hours, jobFailureRate's job-hours).
185
+ const MEASURED_CORE_TIME_FIGURE = new Set(['gc', 'retryWaste', 'speculationWaste']);
186
+
187
+ /** The busy core time a finding's fix removes, in core-milliseconds, or null when the detector
188
+ * measures none. Only a measured figure counts: one its estimate() already set (skew and
189
+ * straggler's removed task time), or the raw figure of a MEASURED_CORE_TIME_FIGURE type. A
190
+ * wall-clock claim, an idle capacity figure and a modeled figure are never converted. It never
191
+ * reads executorCpuTime, which leaves out Python worker CPU. */
192
+ export function coreTimeFor(finding , estimate ) {
193
+ if (estimate.coreTimeMs !== undefined) return estimate.coreTimeMs;
194
+ const raw = estimate.rawWaste;
195
+ if (!raw || !MEASURED_CORE_TIME_FIGURE.has(finding.type)) return null;
196
+ return { low: raw.value, high: raw.value };
197
+ }
@@ -13,6 +13,8 @@
13
13
 
14
14
 
15
15
 
16
+
17
+
16
18
 
17
19
 
18
20
 
@@ -37,9 +39,11 @@ export function routeMessage(
37
39
  case 'job': handlers.onJob?.(data.data); break;
38
40
  case 'runAggregates': handlers.onRunAggregates?.(data.data); break;
39
41
  case 'stageExecutorMetrics': handlers.onStageExecutorMetrics?.(data.data); break;
42
+ case 'stageSpeculationWaste': handlers.onStageSpeculationWaste?.(data.data); break;
43
+ case 'stageLateAttemptWork': handlers.onStageLateAttemptWork?.(data.data); break;
40
44
  case 'done': {
41
- const { skippedLines } = data;
42
- handlers.onDone?.({ skippedLines });
45
+ const { skippedLines, unreadableSqlExecutions } = data;
46
+ handlers.onDone?.(unreadableSqlExecutions ? { skippedLines, unreadableSqlExecutions } : { skippedLines });
43
47
  break;
44
48
  }
45
49
  case 'error': handlers.onError?.(data); break;
@@ -0,0 +1,13 @@
1
+ // Interval union shared by the wall-clock breakdown and the savings rollups.
2
+ export function mergeIntervals(intervals ) {
3
+ if (intervals.length === 0) return [];
4
+ const sorted = [...intervals].sort((a, b) => a[0] - b[0]);
5
+ const out = [[sorted[0][0], sorted[0][1]]];
6
+ for (let i = 1; i < sorted.length; i++) {
7
+ const last = out[out.length - 1];
8
+ const cur = sorted[i];
9
+ if (cur[0] <= last[1]) last[1] = Math.max(last[1], cur[1]);
10
+ else out.push([cur[0], cur[1]]);
11
+ }
12
+ return out;
13
+ }
@@ -5,6 +5,7 @@ import { reassembleRollingEntries } from './parser-worker.js';
5
5
  import { peekLogHeader } from './log-header-peek.js';
6
6
  import { mcpError } from './mcp-error.js';
7
7
  import { normalizeBaseUrl } from './shs-request.js';
8
+ import { isConnectionFailure } from './proxy.js';
8
9
  import { DEFAULT_IDLE_TIMEOUT_MS, DEFAULT_MAX_ARCHIVE_BYTES } from './shs-load.js';
9
10
 
10
11
 
@@ -82,9 +83,8 @@ export function applyFiltersAndCap(entries , filters
82
83
  return filters.redact ? { ...capped, runs: redactRunListEntries(capped.runs) } : capped;
83
84
  }
84
85
 
85
- // Local counterpart to redact.ts's redactAppIdentity, sized for a listing of many distinct apps
86
- // rather than the single app that tool works over: redactAppIdentity has no way to keep two
87
- // entries sharing an appId in sync, so this builds its own stable per-distinct-appId pseudonym map
86
+ // Listing-wide app identity redaction: redact.ts's redactors each work over one run, with no way
87
+ // to keep two listing entries sharing an appId in sync, so this builds its own stable per-distinct-appId pseudonym map
88
88
  // (numeric-aware sort, same scheme as redact.ts's buildMap) across the whole result set, two
89
89
  // attempts of the same app redact to the same identity, and applies it to every identity-bearing
90
90
  // field: appId, name (a listing's human-readable name is as identifying as the id itself), and the
@@ -206,10 +206,12 @@ export async function listRunsShs(
206
206
  let res ;
207
207
  try {
208
208
  // An unresponsive SHS would otherwise hang the tool call forever. A timed-out signal rejects
209
- // the fetch with an AbortError, which this same catch turns into access-or-upstream-failure.
209
+ // the fetch with a TimeoutError, which this same catch turns into upstream-unreachable, the
210
+ // code a refused connection gets too.
210
211
  res = await fetchImpl(url.toString(), { signal: AbortSignal.timeout(DEFAULT_IDLE_TIMEOUT_MS) });
211
212
  } catch (e) {
212
- throw mcpError('access-or-upstream-failure', `Could not reach ${normalized}: ${e instanceof Error ? e.message : String(e)}`);
213
+ const code = isConnectionFailure(e) ? 'upstream-unreachable' : 'access-or-upstream-failure';
214
+ throw mcpError(code, `Could not reach ${normalized}: ${e instanceof Error ? e.message : String(e)}`);
213
215
  }
214
216
  if (!res.ok) {
215
217
  throw mcpError('access-or-upstream-failure', `SHS applications list request failed with status ${res.status}.`);
@@ -1,16 +1,81 @@
1
1
  // Plain JS: scripts/vendor-core.mjs copies it byte-for-byte into vendor-core/, so this file exists
2
2
  // at both vendor-core/load-vendored.js (published) and core/src/load-vendored.js (dev). Each
3
- // package's bin bootstraps by locating this file first, then uses the exports below for every other
4
- // core module, so the resolution logic lives in one place instead of per entry point.
5
- import { existsSync } from 'node:fs';
6
- import { join } from 'node:path';
3
+ // package's bin bootstraps by locating this file first (the core/src copy when it exists, so the
4
+ // staleness check below is always current code), then uses the exports below for every other core
5
+ // module, so the resolution logic lives in one place instead of per entry point.
6
+ import { createHash } from 'node:crypto';
7
+ import { existsSync, readdirSync, readFileSync } from 'node:fs';
8
+ import { join, relative } from 'node:path';
7
9
  import { pathToFileURL } from 'node:url';
8
10
 
11
+ // Written into vendor-core/ by scripts/vendor-core.mjs: the coreSourceHash of the core/src it was
12
+ // built from, so a monorepo run can tell a leftover vendored copy from a current one.
13
+ export const SOURCE_HASH_FILE = 'core-source-hash.txt';
14
+
15
+ // docs-content/ is generated from a pinned upstream commit, not analysis code.
16
+ const UNHASHED_TOP_LEVEL = new Set(['docs-content']);
17
+
18
+ /** A content hash of every analysis source file under core/src (paths and bytes, sorted). */
19
+ export function coreSourceHash(coreSrcDir) {
20
+ const hash = createHash('sha256');
21
+ const walk = (dir) => {
22
+ const entries = readdirSync(dir, { withFileTypes: true }).sort((a, b) => (a.name < b.name ? -1 : a.name > b.name ? 1 : 0));
23
+ for (const entry of entries) {
24
+ if (dir === coreSrcDir && UNHASHED_TOP_LEVEL.has(entry.name)) continue;
25
+ const path = join(dir, entry.name);
26
+ if (entry.isDirectory()) walk(path);
27
+ else if (entry.isFile()) hash.update(`${relative(coreSrcDir, path)}\0`).update(readFileSync(path)).update('\0');
28
+ }
29
+ };
30
+ walk(coreSrcDir);
31
+ return hash.digest('hex');
32
+ }
33
+
34
+ // One decision per package per process, so the hash runs and the warning prints once.
35
+ const useVendoredByPkg = new Map();
36
+
37
+ // A published install has only vendor-core/. In the monorepo, vendor-core/ is a leftover from a
38
+ // local `npm pack` (it is rebuilt only at prepack), so it is used only while it still matches
39
+ // core/src; otherwise the run says so on stderr and loads core/src, rather than silently running
40
+ // outdated detectors.
41
+ function useVendored(pkgDir) {
42
+ if (useVendoredByPkg.has(pkgDir)) return useVendoredByPkg.get(pkgDir);
43
+ const vendorDir = join(pkgDir, 'vendor-core');
44
+ const srcDir = join(pkgDir, '..', 'core', 'src');
45
+ let use = existsSync(vendorDir);
46
+ if (use && existsSync(join(srcDir, 'load-vendored.js'))) {
47
+ const stampPath = join(vendorDir, SOURCE_HASH_FILE);
48
+ const stamp = existsSync(stampPath) ? readFileSync(stampPath, 'utf8').trim() : null;
49
+ if (stamp !== coreSourceHash(srcDir)) {
50
+ process.stderr.write(
51
+ `sparkforensics: ${vendorDir} was built from older packages/core sources, so it would run outdated analysis. `
52
+ + `Using packages/core/src instead; rebuild it with \`node scripts/vendor-core.mjs ${pkgDir}\` or delete it.\n`,
53
+ );
54
+ use = false;
55
+ }
56
+ }
57
+ useVendoredByPkg.set(pkgDir, use);
58
+ return use;
59
+ }
60
+
9
61
  // moduleName is a path relative to core/src without extension, e.g. 'cli/collect-run'. Pass
10
62
  // srcExt: 'js' for modules already plain JS in core/src (e.g. 'proxy') rather than TypeScript.
11
63
  export function resolveVendored(pkgDir, moduleName, { srcExt = 'ts' } = {}) {
12
64
  const vendored = join(pkgDir, 'vendor-core', `${moduleName}.js`);
13
- return existsSync(vendored) ? vendored : join(pkgDir, '..', 'core', 'src', `${moduleName}.${srcExt}`);
65
+ return useVendored(pkgDir) && existsSync(vendored) ? vendored : join(pkgDir, '..', 'core', 'src', `${moduleName}.${srcExt}`);
66
+ }
67
+
68
+ /** The build id of the core this package runs: the stamped source hash of its vendor-core/ when
69
+ * that copy is in use, else the hash of core/src computed now. The same `coreSourceHash` the web
70
+ * build stamps into its bundle, so equal ids mean the same analysis code. 'dev' when neither
71
+ * exists. */
72
+ export function coreBuildId(pkgDir) {
73
+ if (useVendored(pkgDir)) {
74
+ const stampPath = join(pkgDir, 'vendor-core', SOURCE_HASH_FILE);
75
+ if (existsSync(stampPath)) return readFileSync(stampPath, 'utf8').trim();
76
+ }
77
+ const srcDir = join(pkgDir, '..', 'core', 'src');
78
+ return existsSync(srcDir) ? coreSourceHash(srcDir) : 'dev';
14
79
  }
15
80
 
16
81
  export async function loadVendored(pkgDir, moduleName, opts) {
@@ -5,6 +5,7 @@ import {
5
5
  resolveOrCreateRun, diagnoseRun, getRunSummary, compareRuns, getFindingEvidence, getFindingDocumentation, getReferenceDoc, evaluateBudgetsForRun,
6
6
  } from './mcp-tools.js';
7
7
  import { listRuns } from './list-runs.js';
8
+
8
9
 
9
10
  const sourceSchema = z.union([
10
11
  z.object({ path: z.string() }),
@@ -59,7 +60,9 @@ function toolResult (promise )
59
60
  return promise.then((value) => toCallToolResult(JSON.stringify(value), value ), toolErrorResult);
60
61
  }
61
62
 
62
- export function createMcpServer() {
63
+ /** `thresholds`: the bin's --thresholds overrides, applied to every analyzing tool for the life
64
+ * of the server. Omitted, every tool runs the specification defaults. */
65
+ export function createMcpServer({ thresholds } = {}) {
63
66
  const server = new McpServer({ name: 'sparkforensics', version: '1.0.0' });
64
67
 
65
68
  server.registerTool('list_runs', {
@@ -68,7 +71,7 @@ export function createMcpServer() {
68
71
  }, (params) => toolResult(listRuns(params)));
69
72
 
70
73
  server.registerTool('diagnose_run', {
71
- description: 'Diagnose a Spark run: thresholded findings with remediation text, an impact-ranked fix recommendation rollup, and clean-check status.',
74
+ description: 'Diagnose a Spark run: the dashboard verdict (title, summary, and the top places to look, ranked by potential savings), thresholded findings with remediation text, an impact-ranked fix recommendation rollup, clean-check status, and the checks the log lacked the data to run. When the server was started with --thresholds, findings and clean checks from a tuned detector carry tunedThresholds, and their impact estimates are uncalibrated.',
72
75
  inputSchema: {
73
76
  ...runRefSchema, redact: z.boolean().optional(),
74
77
  include: z.array(z.enum(['summary', 'evidenceAvailability', 'detectors'])).optional(),
@@ -77,14 +80,14 @@ export function createMcpServer() {
77
80
  },
78
81
  }, ({ source, runId, redact, include, format, impactBand, type, stageId }) => toolResultWithMarkdown(
79
82
  resolveOrCreateRun({ source, runId }).then(({ runId: id }) =>
80
- diagnoseRun(id, { redact, include, markdown: format === 'md', impactBand, type, stageId })),
83
+ diagnoseRun(id, { redact, include, markdown: format === 'md', impactBand, type, stageId, thresholds })),
81
84
  ));
82
85
 
83
86
  server.registerTool('get_run_summary', {
84
- description: 'App/stage/job/sql counts and duration for a run, no findings.',
87
+ description: 'App/stage/job/sql counts, duration, and how the run ended (failed/total jobs and Spark\'s first-line failure reason) for a run, no findings.',
85
88
  inputSchema: { ...runRefSchema, redact: z.boolean().optional() },
86
89
  }, ({ source, runId, redact }) => toolResult(
87
- resolveOrCreateRun({ source, runId }).then(({ runId: id }) => getRunSummary(id, { redact })),
90
+ resolveOrCreateRun({ source, runId }).then(({ runId: id }) => getRunSummary(id, { redact, thresholds })),
88
91
  ));
89
92
 
90
93
  server.registerTool('compare_runs', {
@@ -96,17 +99,17 @@ export function createMcpServer() {
96
99
  ...formatSchema,
97
100
  },
98
101
  }, ({ runIdA, sourceA, runIdB, sourceB, redact, format }) => toolResultWithMarkdown(
99
- compareRuns({ runId: runIdA, source: sourceA }, { runId: runIdB, source: sourceB }, { redact, markdown: format === 'md' }),
102
+ compareRuns({ runId: runIdA, source: sourceA }, { runId: runIdB, source: sourceB }, { redact, markdown: format === 'md', thresholds }),
100
103
  ));
101
104
 
102
105
  server.registerTool('evaluate_budgets', {
103
- description: 'Evaluate a run (optionally against a second run for regression budgets) against pass/fail thresholds.',
106
+ description: 'Evaluate a run against pass/fail thresholds, optionally against a baseline run for regression budgets. With two runs, absolute budgets apply to the candidate (sourceB/runIdB), matching the CLI. A run with no ApplicationEnd always adds an inconclusive run-complete result.',
104
107
  inputSchema: {
105
- source: sourceSchema.optional().describe('The run to evaluate. Also the regression baseline when sourceB/runIdB is given.'),
108
+ source: sourceSchema.optional().describe('The run to evaluate. When sourceB/runIdB is given, this is the regression baseline instead, and the absolute budgets apply to sourceB/runIdB.'),
106
109
  runId: runRefSchema.runId.describe('Same as `source`, referencing an already-resolved run by id.'),
107
110
  maxRuntimeMs: z.number().optional(), maxSpillGb: z.number().optional(), maxSkewRatio: z.number().optional(),
108
111
  maxFailedTaskRatePct: z.number().optional(), minEfficiencyPct: z.number().optional(),
109
- sourceB: sourceSchema.optional().describe('Optional candidate run, compared against source/runId as the regression baseline (maxRegressionPct/failOnIntroduced).'),
112
+ sourceB: sourceSchema.optional().describe('Optional candidate run, compared against source/runId as the regression baseline (maxRegressionPct/failOnIntroduced). When given, the absolute budgets (maxRuntimeMs etc.) are evaluated on this run.'),
110
113
  runIdB: secondRunRefSchema.runIdB.describe('Same as `sourceB`, referencing an already-resolved run by id.'),
111
114
  maxRegressionPct: z.number().optional(), regressionMetric: z.string().optional(), failOnIntroduced: z.string().optional(),
112
115
  },
@@ -118,13 +121,14 @@ export function createMcpServer() {
118
121
  { source, runId },
119
122
  { maxRuntimeMs, maxSpillGb, maxSkewRatio, maxFailedTaskRatePct, minEfficiencyPct, maxRegressionPct, regressionMetric, failOnIntroduced },
120
123
  (runIdB || sourceB) ? { runId: runIdB, source: sourceB } : undefined,
124
+ { thresholds },
121
125
  )));
122
126
 
123
127
  server.registerTool('get_finding_evidence', {
124
128
  description: 'Raw evidence bundle backing one finding, for drill-down after diagnose_run.',
125
129
  inputSchema: { runId: z.string(), findingId: z.string(), redact: z.boolean().optional() },
126
130
  }, ({ runId, findingId, redact }) => toolResult(
127
- Promise.resolve().then(() => getFindingEvidence(runId, findingId, { redact })),
131
+ Promise.resolve().then(() => getFindingEvidence(runId, findingId, { redact, thresholds })),
128
132
  ));
129
133
 
130
134
  server.registerTool('get_finding_documentation', {