sparkforensics-mcp 0.3.0 → 0.5.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (92) hide show
  1. package/bin/sparkforensics-mcp.mjs +41 -10
  2. package/package.json +1 -1
  3. package/vendor-core/allocation.js +106 -0
  4. package/vendor-core/analyzer.js +168 -60
  5. package/vendor-core/check-coverage.js +88 -0
  6. package/vendor-core/cli/budgets.js +54 -27
  7. package/vendor-core/cli/collect-run.js +84 -32
  8. package/vendor-core/cli/regression-budgets.js +83 -0
  9. package/vendor-core/cli/threshold-config.js +28 -0
  10. package/vendor-core/comparison-verdict.js +177 -0
  11. package/vendor-core/core-source-hash.txt +1 -0
  12. package/vendor-core/core-usage-locality.js +56 -2
  13. package/vendor-core/detector-docs.js +58 -0
  14. package/vendor-core/detectors.js +1094 -500
  15. package/vendor-core/docs-config.js +0 -36
  16. package/vendor-core/docs-content/chapters/nav-index.json +31 -0
  17. package/vendor-core/docs-content/detection/cache.md +3 -2
  18. package/vendor-core/docs-content/detection/cfg.md +9 -8
  19. package/vendor-core/docs-content/detection/chrn.md +1 -2
  20. package/vendor-core/docs-content/detection/cold.md +4 -2
  21. package/vendor-core/docs-content/detection/cstor.md +9 -0
  22. package/vendor-core/docs-content/detection/fail.md +3 -2
  23. package/vendor-core/docs-content/detection/gc.md +3 -2
  24. package/vendor-core/docs-content/detection/host.md +2 -1
  25. package/vendor-core/docs-content/detection/local.md +1 -1
  26. package/vendor-core/docs-content/detection/mem.md +5 -2
  27. package/vendor-core/docs-content/detection/plan.md +2 -1
  28. package/vendor-core/docs-content/detection/sfail.md +2 -1
  29. package/vendor-core/docs-content/detection/shape.md +5 -4
  30. package/vendor-core/docs-content/detection/skew.md +3 -1
  31. package/vendor-core/docs-content/detection/slow.md +2 -2
  32. package/vendor-core/docs-content/detection/spec.md +2 -3
  33. package/vendor-core/docs-content/detection/spill.md +1 -1
  34. package/vendor-core/docs-site-config.js +3 -0
  35. package/vendor-core/effective-conf.js +107 -0
  36. package/vendor-core/efficiency-model.js +8 -6
  37. package/vendor-core/event-handlers.js +321 -44
  38. package/vendor-core/event-schemas.js +23 -0
  39. package/vendor-core/evidence-report.js +432 -115
  40. package/vendor-core/export-data.js +79 -6
  41. package/vendor-core/finding-action-label.js +9 -88
  42. package/vendor-core/finding-filter-predicate.js +9 -0
  43. package/vendor-core/finding-generic-recommendation.js +26 -105
  44. package/vendor-core/finding-names.js +28 -45
  45. package/vendor-core/finding-presentation.js +368 -0
  46. package/vendor-core/finding-tag-help.js +110 -0
  47. package/vendor-core/finding-types.js +373 -0
  48. package/vendor-core/findings-of-type.js +11 -0
  49. package/vendor-core/format-utils.js +96 -30
  50. package/vendor-core/html-export.js +51 -0
  51. package/vendor-core/impact-band.js +21 -8
  52. package/vendor-core/impact-estimator.js +25 -520
  53. package/vendor-core/impact-format.js +115 -0
  54. package/vendor-core/impact-model.js +197 -0
  55. package/vendor-core/ingest.js +6 -2
  56. package/vendor-core/intervals.js +13 -0
  57. package/vendor-core/list-runs.js +7 -5
  58. package/vendor-core/load-vendored.js +70 -5
  59. package/vendor-core/mcp-server-factory.js +14 -10
  60. package/vendor-core/mcp-tools.js +105 -45
  61. package/vendor-core/model-assembler.js +35 -1
  62. package/vendor-core/occupancy.js +1 -1
  63. package/vendor-core/parser-worker.js +2 -2
  64. package/vendor-core/plan-graph-model.js +3 -2
  65. package/vendor-core/plan-node-detail.js +1 -1
  66. package/vendor-core/proxy.js +3 -1
  67. package/vendor-core/python-stage.js +25 -0
  68. package/vendor-core/recommendation-rollup.js +70 -3
  69. package/vendor-core/redact.js +96 -37
  70. package/vendor-core/remediation.js +20 -0
  71. package/vendor-core/run-comparison.js +73 -29
  72. package/vendor-core/run-interpretation.js +291 -0
  73. package/vendor-core/run-metrics.js +198 -0
  74. package/vendor-core/run-outcome.js +74 -0
  75. package/vendor-core/run-payload.js +17 -0
  76. package/vendor-core/run-shape.js +40 -0
  77. package/vendor-core/run-totals.js +24 -0
  78. package/vendor-core/run-verdict.js +352 -0
  79. package/vendor-core/scaling-sim.js +4 -5
  80. package/vendor-core/scorecard-estimates.js +63 -0
  81. package/vendor-core/session-snapshot.js +7 -0
  82. package/vendor-core/shs-schemas.js +2 -2
  83. package/vendor-core/spark-memory.js +17 -0
  84. package/vendor-core/sql-stages.js +11 -0
  85. package/vendor-core/stage-plan-nodes.js +18 -0
  86. package/vendor-core/stage-quantiles.js +6 -0
  87. package/vendor-core/threshold-overrides.js +160 -0
  88. package/vendor-core/threshold-summary.js +11 -33
  89. package/vendor-core/types.js +54 -42
  90. package/vendor-core/wall-clock.js +1 -12
  91. package/vendor-core/wasted-core-hours.js +12 -9
  92. package/vendor-core/write-targets.js +312 -0
@@ -1,5 +1,6 @@
1
1
  #!/usr/bin/env node
2
2
  import { existsSync } from 'node:fs';
3
+ import { parseArgs } from 'node:util';
3
4
  import { join, dirname } from 'node:path';
4
5
  import { fileURLToPath, pathToFileURL } from 'node:url';
5
6
  import { StdioServerTransport } from '@modelcontextprotocol/sdk/server/stdio.js';
@@ -7,16 +8,20 @@ import { StdioServerTransport } from '@modelcontextprotocol/sdk/server/stdio.js'
7
8
  const binDir = dirname(fileURLToPath(import.meta.url));
8
9
  const pkgDir = dirname(binDir);
9
10
 
10
- // vendor-core/ exists only in a published install (populated by vendor-core.mjs
11
- // at pack time); the monorepo falls back to the packages/core/src/ sibling.
11
+ // vendor-core/ is populated by vendor-core.mjs at pack time. In the monorepo the
12
+ // packages/core/src/ sibling's load-vendored.js is used, which prefers a leftover
13
+ // vendor-core/ only while it still matches core/src and warns when it does not.
12
14
  // load-vendored.js is the one module located by hand; the rest load through its
13
15
  // exported loadVendored().
14
- async function loadCreateMcpServer() {
15
- const vendoredHelper = join(pkgDir, 'vendor-core', 'load-vendored.js');
16
- const helperPath = existsSync(vendoredHelper) ? vendoredHelper : join(pkgDir, '..', 'core', 'src', 'load-vendored.js');
16
+ async function loadCoreModule(moduleName) {
17
+ const srcHelper = join(pkgDir, '..', 'core', 'src', 'load-vendored.js');
18
+ const helperPath = existsSync(srcHelper) ? srcHelper : join(pkgDir, 'vendor-core', 'load-vendored.js');
17
19
  const { loadVendored } = await import(pathToFileURL(helperPath).href);
18
- const mod = await loadVendored(pkgDir, 'mcp-server-factory');
19
- return mod.createMcpServer;
20
+ return loadVendored(pkgDir, moduleName);
21
+ }
22
+
23
+ async function loadCreateMcpServer() {
24
+ return (await loadCoreModule('mcp-server-factory')).createMcpServer;
20
25
  }
21
26
 
22
27
  // Tool names come from the server the bin actually builds, so --help can't drift
@@ -24,25 +29,51 @@ async function loadCreateMcpServer() {
24
29
  // no I/O. _registeredTools is the SDK's registry; the MCP package test checks that
25
30
  // this list matches what listTools reports.
26
31
  function usage(toolNames) {
27
- return `Usage: sparkforensics-mcp
32
+ return `Usage: sparkforensics-mcp [--thresholds <file>]
28
33
 
29
34
  Starts the SparkForensics MCP server, speaking the MCP protocol over
30
35
  stdio. Point an MCP client (Claude Desktop, Claude Code, etc.) at this
31
36
  command; it exposes ${toolNames.length} tools for diagnosing Apache Spark event logs:
32
37
  ${toolNames.join(', ')}.
33
38
 
39
+ --thresholds <file> Run every tool's detectors with the threshold overrides in
40
+ this JSON file ({"<detector>": {"<threshold>": value}}).
41
+ Findings a tuned detector produces carry tunedThresholds.
42
+ An unreadable or invalid file stops the server from starting.
43
+
34
44
  See https://github.com/shuffle-works/sparkforensics#readme for details.
35
45
  `;
36
46
  }
37
47
 
38
48
  async function main() {
39
- if (process.argv.includes('--help') || process.argv.includes('-h')) {
49
+ // Not strict: arguments this server never read before stay ignored, as they always were.
50
+ const { values } = parseArgs({
51
+ args: process.argv.slice(2), strict: false,
52
+ options: { thresholds: { type: 'string' }, help: { type: 'boolean', short: 'h' } },
53
+ });
54
+ if (values.help) {
40
55
  const createMcpServer = await loadCreateMcpServer();
41
56
  process.stderr.write(usage(Object.keys(createMcpServer()._registeredTools)));
42
57
  return;
43
58
  }
59
+ let thresholds;
60
+ if (values.thresholds !== undefined) {
61
+ if (typeof values.thresholds !== 'string') {
62
+ process.stderr.write('--thresholds requires a file path.\n');
63
+ process.exitCode = 2;
64
+ return;
65
+ }
66
+ // Refuse to start rather than serve default-threshold results the user meant to change.
67
+ try {
68
+ thresholds = (await loadCoreModule('cli/threshold-config')).loadThresholdOverrides(values.thresholds);
69
+ } catch (e) {
70
+ process.stderr.write(`--thresholds: ${e.message}\n`);
71
+ process.exitCode = 2;
72
+ return;
73
+ }
74
+ }
44
75
  const createMcpServer = await loadCreateMcpServer();
45
- const server = createMcpServer();
76
+ const server = createMcpServer({ thresholds });
46
77
  const transport = new StdioServerTransport();
47
78
  await server.connect(transport);
48
79
  }
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "sparkforensics-mcp",
3
- "version": "0.3.0",
3
+ "version": "0.5.0",
4
4
  "mcpName": "io.github.shuffle-works/sparkforensics-mcp",
5
5
  "type": "module",
6
6
  "license": "MIT",
@@ -0,0 +1,106 @@
1
+ import { parseSparkMemoryMB } from './spark-memory.js';
2
+
3
+
4
+ const MS_PER_HOUR = 3_600_000;
5
+ const MIB_PER_GIB = 1024;
6
+ // Spark's documented floor and factor for the default executor memory overhead
7
+ // (spark.executor.memoryOverhead = max(factor * executor memory, 384 MiB)).
8
+ const MIN_OVERHEAD_MIB = 384;
9
+ const DEFAULT_OVERHEAD_FACTOR = 0.1;
10
+
11
+
12
+  
13
+
14
+  
15
+
16
+
17
+
18
+
19
+
20
+
21
+
22
+
23
+
24
+
25
+ /** The latest timestamp the log records: the application start and end and every stage and
26
+ * executor event. Where a log with no ApplicationEnd stops. */
27
+ export function lastObservedTimestamp(input ) {
28
+ let last = null;
29
+ const take = (t ) => {
30
+ if (typeof t === 'number' && Number.isFinite(t) && t > 0 && (last == null || t > last)) last = t;
31
+ };
32
+ take(input.app?.startTime);
33
+ take(input.app?.endTime);
34
+ for (const s of input.stages.values()) { take(s.submittedAt); take(s.completedAt); }
35
+ for (const e of [...input.executors.added, ...input.executors.removed]) take(e.timestamp);
36
+ return last;
37
+ }
38
+
39
+ // Spark's default executor memory when spark.executor.memory is unset.
40
+ const DEFAULT_EXECUTOR_MEMORY_MIB = 1024;
41
+
42
+ // One container's memory in MiB, as Spark requests it from the cluster manager: executor heap,
43
+ // plus overhead, plus off-heap and PySpark worker memory when configured. Null when the log
44
+ // records no Spark properties (nothing to tell a default from a missing config) or a memory key it
45
+ // does record cannot be read.
46
+ function executorMemoryMiB(app ) {
47
+ const config = app?.config;
48
+ if (config == null) return null;
49
+ // undefined: key absent; null: present but unreadable.
50
+ const mib = (key ) => (config[key] == null ? undefined : parseSparkMemoryMB(config[key]));
51
+ const heap = mib('spark.executor.memory') ?? (config['spark.executor.memory'] == null ? DEFAULT_EXECUTOR_MEMORY_MIB : null);
52
+ if (heap == null) return null;
53
+
54
+ let overhead = mib('spark.executor.memoryOverhead');
55
+ if (overhead === undefined) overhead = mib('spark.yarn.executor.memoryOverhead'); // legacy key
56
+ if (overhead === undefined) {
57
+ const factor = Number.parseFloat(config['spark.executor.memoryOverheadFactor'] ?? '');
58
+ overhead = Math.max(MIN_OVERHEAD_MIB, Math.round(heap * (Number.isFinite(factor) && factor > 0 ? factor : DEFAULT_OVERHEAD_FACTOR)));
59
+ }
60
+ const offHeap = String(config['spark.memory.offHeap.enabled']).toLowerCase() === 'true' ? mib('spark.memory.offHeap.size') ?? 0 : 0;
61
+ const pyspark = mib('spark.executor.pyspark.memory') ?? 0;
62
+ if (overhead === null || offHeap === null || pyspark === null) return null;
63
+ return heap + overhead + offHeap + pyspark;
64
+ }
65
+
66
+ /** Allocated core-hours and memory GiB-hours from the executor lifecycle: each executor counts
67
+ * from its ExecutorAdded timestamp to its first later ExecutorRemoved timestamp. One with no
68
+ * removal closes at the application end when the log has one, else at the last timestamp the log
69
+ * records (a cut-off log). Cores are the ExecutorAdded event's Total Cores, else
70
+ * spark.executor.cores; memory per executor is spark.executor.memory (default 1g) plus the
71
+ * overhead (spark.executor.memoryOverhead, else the legacy spark.yarn.executor.memoryOverhead,
72
+ * else the larger of 384 MiB and spark.executor.memoryOverheadFactor, default 0.1, times the
73
+ * memory), plus spark.memory.offHeap.size when spark.memory.offHeap.enabled is true, plus
74
+ * spark.executor.pyspark.memory.
75
+ * Null, never 0, for a figure whose inputs the log lacks. */
76
+ export function computeAllocation(input ) {
77
+ const added = input.executors.added.filter((e) => e.kind === 'added');
78
+ if (added.length === 0) return { coreHours: null, memoryGbHours: null };
79
+ const removedAt = new Map ();
80
+ for (const e of input.executors.removed) {
81
+ if (e.kind !== 'removed') continue;
82
+ const times = removedAt.get(e.executorId);
83
+ if (times) times.push(e.timestamp); else removedAt.set(e.executorId, [e.timestamp]);
84
+ }
85
+ const closeAt = input.app?.endTime ?? lastObservedTimestamp(input);
86
+ const configuredCores = Number.parseInt(input.app?.config?.['spark.executor.cores'] ?? '', 10);
87
+ const memoryMiB = executorMemoryMiB(input.app);
88
+
89
+ let coreMs = 0;
90
+ let memoryMiBMs = 0;
91
+ let coresKnown = true;
92
+ const seen = new Set ();
93
+ for (const e of added) {
94
+ if (seen.has(e.executorId)) continue; // a replayed ExecutorAdded is the same executor
95
+ seen.add(e.executorId);
96
+ const removal = (removedAt.get(e.executorId) ?? []).filter((t) => t >= e.timestamp).sort((a, b) => a - b)[0];
97
+ const aliveMs = Math.max(0, (removal ?? closeAt ?? e.timestamp) - e.timestamp);
98
+ const cores = e.totalCores > 0 ? e.totalCores : Number.isFinite(configuredCores) ? configuredCores : null;
99
+ if (cores == null) coresKnown = false; else coreMs += cores * aliveMs;
100
+ memoryMiBMs += (memoryMiB ?? 0) * aliveMs;
101
+ }
102
+ return {
103
+ coreHours: coresKnown ? coreMs / MS_PER_HOUR : null,
104
+ memoryGbHours: memoryMiB != null ? memoryMiBMs / MIB_PER_GIB / MS_PER_HOUR : null,
105
+ };
106
+ }
@@ -1,14 +1,28 @@
1
- import { DETECTORS, } from './detectors.js';
1
+ import { DETECTORS, ENTRY_BY_TYPE, } from './detectors.js';
2
+ import { effectiveThresholds, findingTunedThresholds, overridesFor, tunedThresholdsNote } from './threshold-overrides.js';
2
3
  import { computePeakConcurrentCores } from './core-count.js';
3
4
  import { assertNever } from './assert-never.js';
4
- import { estimateImpact } from './impact-estimator.js';
5
+ import { estimateImpact, } from './impact-estimator.js';
5
6
  import { computeOccupancy, } from './occupancy.js';
6
- import { deriveImpactBand } from './impact-band.js';
7
+ import { deriveImpactBand, IMPACT_FLOOR_PCT_CRIT, IMPACT_FLOOR_PCT_WARN, } from './impact-band.js';
7
8
  import { IMPACT_BAND_ORDER } from './format-utils.js';
8
9
 
9
-
10
+
10
11
 
11
12
 
13
+ // The runner reads each entry through the Detector contract, not its own precise `as const` shape:
14
+ // the scope switch below calls each entry's bound detect with that scope's target.
15
+ const detectors = DETECTORS;
16
+
17
+
18
+
19
+
20
+
21
+
22
+
23
+
24
+
25
+
12
26
  // FNV-1a 32-bit stable string hash: deterministic finding id across runs, no timestamps/randomness.
13
27
  function fnv1a(str ) {
14
28
  let h = 0x811c9dc5;
@@ -19,66 +33,97 @@ function fnv1a(str ) {
19
33
  return (h >>> 0).toString(16).padStart(8, '0');
20
34
  }
21
35
 
36
+ // The id hash's discriminator slots, in the order it has always joined them: reordering or renaming
37
+ // one changes every finding id, which needs an EVIDENCE_SCHEMA_VERSION bump.
38
+ const DISCRIMINATOR_SLOTS = [
39
+ 'host', 'executorId', 'rule', 'variant', 'dimension',
40
+ 'direction', 'nodeName', 'rootName', 'subtreeSize', 'groupIndex', 'largerSideBytes',
41
+ 'rddId', 'relation', 'format', 'operator', 'executionIds',
42
+ ] ;
43
+
44
+
45
+ // Per type, the fields that tell apart sibling findings sharing one location and metric; without
46
+ // them those siblings hash to one id:
47
+ // slowHost (host), memoryUtilization (executorId), partitionSizing (rule);
48
+ // cacheUtilization (rddId+variant), cachingOpportunity (relation/format for leaf,
49
+ // operator+relation for composite, executionIds as last resort);
50
+ // smallFiles (direction/nodeName), duplicatePlanSubtree (groupIndex is the real
51
+ // uniqueness guarantee: rootName+subtreeSize can collide across groups),
52
+ // underBroadcast (value+largerSideBytes per node/side).
53
+ // memoryUtilization leaves out `rule`: one heap band per executor, so executorId is already
54
+ // unique and folding rule in risks id churn if band logic changes. partitionSizing keeps
55
+ // `rule`: a stage can emit several rules at once sharing stageId+metric.
56
+ const ID_DISCRIMINATORS = {
57
+ skew: [], stageShape: ['rule'], shuffle: [], partitionSizing: ['rule'], spill: [], gc: ['direction'],
58
+ slowHost: ['host', 'executorId', 'variant', 'dimension'], stageSlowness: [], stageFailed: ['variant'],
59
+ failures: [], straggler: [], speculationWaste: [], retryWaste: [], tinyTask: [],
60
+ incompleteRun: [], coldStart: [], utilization: [], memoryUtilization: ['executorId', 'variant'],
61
+ cacheUtilization: ['variant', 'rddId'], coreLocality: [], autoscalingChurn: [],
62
+ cachingOpportunity: ['variant', 'relation', 'format', 'operator', 'executionIds'],
63
+ jobFailureRate: [], configAudit: [],
64
+ duplicatePlanSubtree: ['rootName', 'subtreeSize', 'groupIndex'], smallFiles: ['direction', 'nodeName'],
65
+ underBroadcast: ['largerSideBytes'], overBroadcast: [],
66
+ };
67
+
68
+ // ID_DISCRIMINATORS as Sets, built once: findingId runs once per finding.
69
+ const ID_DISCRIMINATOR_SETS = new Map (
70
+ Object.entries(ID_DISCRIMINATORS).map(([type, slots]) => [type, new Set (slots)]),
71
+ );
72
+
73
+ // Location key: stage, else SQL execution, else audited config property.
74
+ function locationKey(f ) {
75
+ if (f.stageId != null) return f.stageId;
76
+ if ('executionId' in f) return f.executionId;
77
+ if (f.type === 'configAudit') return f.property;
78
+ return '';
79
+ }
80
+
22
81
  export function findingId(f ) {
23
- // Location key: stage, else SQL execution, else audited config property.
24
- const locKey = f.stageId ?? f.executionId ?? f.property ?? '';
25
- // Discriminators for detectors that emit multiple findings on the same
26
- // location+metric; without them these siblings hash to one id:
27
- // slowHost (host), memoryUtilization (executorId+dimension), partitionSizing (rule);
28
- // cacheUtilization (rddId+variant), cachingOpportunity (relation/format for leaf,
29
- // operator+relation for composite, executionIds as last resort);
30
- // smallFiles (direction/nodeName), duplicatePlanSubtree (groupIndex is the real
31
- // uniqueness guarantee: rootName+subtreeSize can collide across groups),
32
- // broadcastSizing (value+largerSideBytes per node/side).
33
- // memoryUtilization excludes `rule`: one heap band per executor, so executorId is already
34
- // unique and folding rule in risks id churn if band logic changes. partitionSizing keeps
35
- // `rule`: a stage can emit several rules at once sharing stageId+metric.
36
- const rule = f.type === 'memoryUtilization' ? undefined : f.rule;
37
- const disc = [
38
- f.host, f.executorId, rule, f.variant, f.dimension,
39
- f.direction, f.nodeName, f.rootName, f.subtreeSize, f.groupIndex, f.largerSideBytes,
40
- f.rddId, f.relation, f.format, f.operator,
41
- f.executionIds ? f.executionIds.join(',') : '',
42
- ].map((v) => v ?? '').join('|');
43
- return fnv1a(`${f.type}|${locKey}|${f.metric ?? ''}|${f.value ?? ''}|${disc}`);
82
+ const fields = f ;
83
+ const listed = ID_DISCRIMINATOR_SETS.get(f.type);
84
+ const disc = DISCRIMINATOR_SLOTS.map((slot) => {
85
+ const v = listed?.has(slot) ? fields[slot] : undefined;
86
+ return Array.isArray(v) ? v.join(',') : v ?? '';
87
+ }).join('|');
88
+ return fnv1a(`${f.type}|${locationKey(f)}|${f.metric ?? ''}|${f.value ?? f.valueText ?? ''}|${disc}`);
44
89
  }
45
90
 
46
- // skew's max/median branch (stage.taskCount below minTasksForP95) and straggler are both driven
47
- // by the identical (taskDurationMax - taskDurationP50) delta on the same stage: the same
48
- // dominant outlier task reported by two detectors, each independently clipped (see "Overlap
49
- // caveat: skew / straggler" in impact-estimation.md). skew's P95/median branch samples a
50
- // different task and stays independent. Flags both sides via validationRequired (rather than
51
- // suppressing either) so neither finding's own diagnostic value is lost; the flag rides the same
91
+ // skew (either branch) and straggler both claim the stage's replayed tail recovery
92
+ // (tailReplayRecoveryMs via tailRecoveryMs): the same slow-task tail reported by two detectors
93
+ // (see "Overlap caveat: skew / straggler" in impact-estimation.md). skew's branch only changes
94
+ // the fallback single-task delta on a stage without the replay, so every skew + straggler pair
95
+ // on a stage is flagged. Flags both sides via validationRequired (rather than suppressing
96
+ // either) so neither finding's own diagnostic value is lost; the flag rides the same
52
97
  // confidence-caveat UI a reader already sees before trusting either finding's magnitude.
53
98
  function overlapNote(otherType ) {
54
- return `This overlaps with the ${otherType} finding on this stage: both are driven by the same dominant outlier task, so don't add their recoverable-time figures together.`;
99
+ return `This overlaps with the ${otherType} finding on this stage: both measure the same slow-task tail, so don't add their recoverable-time figures together.`;
55
100
  }
56
101
 
57
102
  function flagSkewStragglerOverlap(findings ) {
58
- const maxMedianSkewStages = new Set(
59
- findings.filter((f) => f.type === 'skew' && f.metric === 'max/median' && f.stageId != null).map((f) => f.stageId),
103
+ const skewStages = new Set(
104
+ findings.filter((f) => f.type === 'skew' && f.stageId != null).map((f) => f.stageId),
60
105
  );
61
- if (maxMedianSkewStages.size === 0) return;
106
+ if (skewStages.size === 0) return;
62
107
  const stragglerStages = new Set(
63
108
  findings.filter((f) => f.type === 'straggler' && f.stageId != null).map((f) => f.stageId),
64
109
  );
65
- const overlapStages = new Set([...maxMedianSkewStages].filter((id) => stragglerStages.has(id)));
110
+ const overlapStages = new Set([...skewStages].filter((id) => stragglerStages.has(id)));
66
111
  if (overlapStages.size === 0) return;
67
112
  for (const f of findings) {
68
113
  if (f.stageId == null || !overlapStages.has(f.stageId)) continue;
69
114
  const note = f.type === 'skew' ? overlapNote('straggler') : f.type === 'straggler' ? overlapNote('skew') : null;
70
115
  if (!note) continue;
71
- f.validationRequired = f.validationRequired ? `${f.validationRequired} ${note}` : note;
116
+ f.validationRequired = [f.validationRequired, note].filter(Boolean).join(' ');
72
117
  }
73
118
  }
74
119
 
75
- function push(out , entry , result ) {
120
+ function push(out , entry , result , tuned = null) {
76
121
  if (!result) return;
77
122
  const detectorVersion = entry.version ?? 1;
78
123
  for (const f of (Array.isArray(result) ? result : [result])) {
79
124
  if (!f) continue;
80
- if (entry.suppressWhen && entry.suppressWhen(f, out)) continue;
81
125
  const stamped = { ...f, docAnchor: f.docAnchor ?? entry.docAnchor, detectorVersion };
126
+ if (tuned) stamped.tunedThresholds = tuned;
82
127
  const id = findingId(stamped);
83
128
  // Dedup guard: same id => same finding, keep first. Correctness depends on
84
129
  // findingId's discriminators being unique per distinct finding, not on this line.
@@ -87,6 +132,50 @@ function push(out , entry , result
87
132
  }
88
133
  }
89
134
 
135
+ // A tuned finding's caveat, once its estimate is known: only a finding with an estimate figure
136
+ // (wall-clock or raw waste) says that figure is unvalidated.
137
+ function noteTunedThresholds(findings ) {
138
+ for (const f of findings) {
139
+ if (!f.tunedThresholds) continue;
140
+ const hasFigure = f.impactEstimate != null && f.impactEstimate.basis !== 'informational';
141
+ const note = tunedThresholdsNote(f.tunedThresholds, hasFigure);
142
+ f.validationRequired = f.validationRequired ? `${f.validationRequired} ${note}` : note;
143
+ }
144
+ }
145
+
146
+ // skew and straggler gate on floorPctWarn/floorPctCrit thresholds that default to the band's own
147
+ // floors, so a finding they admit at their warn floor grades at least warning. A tuned floor grades
148
+ // that entry's findings too; a type whose entry has no such threshold keeps the defaults.
149
+ function bandFloors(type , overrides ) {
150
+ const entry = ENTRY_BY_TYPE.get(type);
151
+ if (!entry) return null;
152
+ const { floorPctWarn, floorPctCrit } = effectiveThresholds(entry, overrides);
153
+ return {
154
+ warnPct: typeof floorPctWarn === 'number' ? floorPctWarn : IMPACT_FLOOR_PCT_WARN,
155
+ critPct: typeof floorPctCrit === 'number' ? floorPctCrit : IMPACT_FLOOR_PCT_CRIT,
156
+ };
157
+ }
158
+
159
+ // An entry's `suppressedBy` names another entry: drop its findings on every stage that entry
160
+ // flagged. Runs once every detector has, so neither declaration order matters, and reads the
161
+ // unsuppressed findings, so the result doesn't depend on which suppression is applied first.
162
+ function applySuppression(out ) {
163
+ const dropped = new Map ();
164
+ for (const entry of detectors) {
165
+ if (!entry.suppressedBy) continue;
166
+ const suppressorTypes = new Set (
167
+ detectors.filter((d) => d.type === entry.suppressedBy).flatMap((d) => d.emits),
168
+ );
169
+ for (const type of entry.emits) {
170
+ const stagesToDrop = dropped.get(type) ?? new Set ();
171
+ for (const f of out) if (suppressorTypes.has(f.type) && f.stageId != null) stagesToDrop.add(f.stageId);
172
+ dropped.set(type, stagesToDrop);
173
+ }
174
+ }
175
+ if (dropped.size === 0) return out;
176
+ return out.filter((f) => f.stageId == null || !dropped.get(f.type)?.has(f.stageId));
177
+ }
178
+
90
179
  // `app` widened to `SparkAppInfo | null` to match real callers (AppModel.app is
91
180
  // nullable at the type level); every detector below tolerates a null app.
92
181
  export function analyze(
@@ -97,6 +186,7 @@ export function analyze(
97
186
  jobs ,
98
187
  sql = new Map(),
99
188
  runAggregates = null,
189
+ { thresholds } = {},
100
190
  ) {
101
191
  // `app ?? {}`: detectors tolerate a null app (malformed logs), so this must too.
102
192
  // computePeakConcurrentCores (not computeTotalCores): the occupancy ceiling needs a
@@ -108,37 +198,55 @@ export function analyze(
108
198
  executorsAdded ,
109
199
  executorsRemoved ,
110
200
  );
111
- // Computed once so detectors gate impact band on the same occupancy-clipped waste
112
- // estimateImpact displays as savings, not a raw pre-clip delta the two passes would disagree on.
113
- const occupancy = computeOccupancy(stages , totalCores);
114
- const ctx = {
115
- app, stages, executorsAdded, executorsRemoved, jobs, sql, runAggregates, occupancy,
201
+ // One occupancy sweep per analysis, shared by the detectors' runtime floors and every entry's
202
+ // estimate(), so a floor gates on the same occupancy-clipped figure displayed as savings.
203
+ const impact = {
204
+ stages, totalCores, sql,
205
+ occupancy: computeOccupancy(stages , totalCores),
206
+ };
207
+ // The one cast from the posted-model types to the detector-side shapes: types.ts's Stage and
208
+ // SqlExecution carry a catch-all index signature, while every field DetectorStage/DetectorSqlExec
209
+ // declare is one finalizeStage and event-handlers.ts always set (see detectors.ts's header).
210
+ const ctx = {
211
+ app, jobs, executorsAdded, executorsRemoved, runAggregates, impact,
212
+ stages: stages ,
213
+ sql: sql ,
116
214
  };
117
215
  const out = [];
118
- for (const d of DETECTORS) {
216
+ for (const d of detectors) {
119
217
  if (d.inScorecard === false) continue;
218
+ const overrides = overridesFor(d, thresholds);
219
+ const tuned = findingTunedThresholds(d, thresholds);
120
220
  switch (d.scope) {
121
- case 'stage':
122
- for (const s of stages.values()) push(out, d, d.detect(s, ctx));
221
+ case 'stage': {
222
+ const detect = d.withThresholds(overrides);
223
+ for (const s of ctx.stages.values()) push(out, d, detect(s, ctx), tuned);
123
224
  break;
124
- case 'sql':
125
- for (const e of sql.values()) push(out, d, d.detect(e, ctx));
225
+ }
226
+ case 'sql': {
227
+ const detect = d.withThresholds(overrides);
228
+ for (const e of ctx.sql.values()) push(out, d, detect(e, ctx), tuned);
126
229
  break;
230
+ }
127
231
  case 'app':
232
+ push(out, d, d.withThresholds(overrides)(ctx), tuned);
233
+ break;
128
234
  case 'config':
129
- push(out, d, d.detect(ctx));
235
+ push(out, d, d.withThresholds(overrides)(ctx ), tuned);
130
236
  break;
131
237
  default:
132
- assertNever(d.scope);
238
+ assertNever(d);
133
239
  }
134
240
  }
135
- estimateImpact(out, stages, totalCores);
136
- deriveImpactBand(out, app);
137
- flagSkewStragglerOverlap(out);
241
+ const findings = applySuppression(out);
242
+ estimateImpact(findings, impact);
243
+ noteTunedThresholds(findings);
244
+ deriveImpactBand(findings, app, thresholds ? (type) => bandFloors(type, thresholds) : undefined);
245
+ flagSkewStragglerOverlap(findings);
138
246
  // Ascending IMPACT_BAND_ORDER (critical 0 -> info 2) puts the worst band first;
139
247
  // stable sort keeps DETECTORS declaration order within a band.
140
- out.sort((a, b) => IMPACT_BAND_ORDER[a.impactBand] - IMPACT_BAND_ORDER[b.impactBand]);
141
- return out;
248
+ findings.sort((a, b) => IMPACT_BAND_ORDER[a.impactBand] - IMPACT_BAND_ORDER[b.impactBand]);
249
+ return findings;
142
250
  }
143
251
 
144
252
  // Memoizes auditConfig by `app` identity (like evidence-report.ts's jsonCache) so config detectors
@@ -149,10 +257,10 @@ const auditConfigCache = new WeakMap ();
149
257
 
150
258
  function computeAuditConfig(app ) {
151
259
  const out = [];
152
- for (const d of DETECTORS) if (d.scope === 'config') push(out, d, d.detect({ app }));
153
- // configAudit's impact case is unconditionally costOnly('none'): needs no stages/totalCores,
154
- // an empty stages map gives parity with analyze().
155
- estimateImpact(out, new Map());
260
+ for (const d of detectors) if (d.scope === 'config') push(out, d, d.withThresholds()({ app }));
261
+ // configAudit's estimate is unconditionally costOnly('none'): needs no stages/totalCores, so an
262
+ // empty context gives parity with analyze().
263
+ estimateImpact(out, { stages: new Map(), occupancy: new Map(), totalCores: 0 });
156
264
  deriveImpactBand(out, app);
157
265
  return out.map((f) => ({ ...f, stageId: f.stageId ?? null }));
158
266
  }
@@ -0,0 +1,88 @@
1
+ // Which checks a log could actually run. Shared by the dashboard (verdict, top bar, Clean checks)
2
+ // and the CLI/MCP evidence report, so a check the log lacked the data for never reads as passed on
3
+ // one path and "not checked" on another.
4
+ import { detectorCatalog } from './detectors.js';
5
+ import { isRealFinding } from './recommendation-rollup.js';
6
+ import { summarizeRunOutcome } from './run-outcome.js';
7
+
8
+
9
+ export const NO_FINISHED_STAGE_GAP = 'No stage in this log recorded an end, so the stage checks had nothing to measure.';
10
+ export const INCOMPLETE_RUN_GAP =
11
+ 'The log has no end-of-run record, so the core usage, memory and executor churn checks had no run length to measure.';
12
+
13
+ /** True when at least one stage recorded both a start and an end, so the
14
+ * stage checks had something to measure. */
15
+ export function hasFinishedStage(stages ) {
16
+ return [...stages.values()].some((stage) => stage.submittedAt != null && stage.completedAt != null);
17
+ }
18
+
19
+ /** A finding that reports a check could not run for lack of evidence (cache
20
+ * storage without block updates, memory without executor metrics), rather
21
+ * than a problem found. Its recommendation names the setting to turn on. */
22
+ export function isEvidenceCaveat(finding ) {
23
+ return ('dataUnavailable' in finding && finding.dataUnavailable === true) || !isRealFinding(finding);
24
+ }
25
+
26
+ /** App-level checks measured over the run's full span, which a log with no
27
+ * ApplicationEnd (an `incompleteRun` finding) cannot give them. */
28
+ export const RUN_SPAN_CHECK_TYPES = new Set(['utilization', 'memoryUtilization', 'autoscalingChurn']);
29
+
30
+ /** Detector types that measure each stage, read from the detector catalog. */
31
+ export const PER_STAGE_CHECK_TYPES = new Set(
32
+ detectorCatalog().filter((entry) => entry.scope === 'stage').map((entry) => entry.type),
33
+ );
34
+
35
+ export function isIncompleteRun(allFindings ) {
36
+ return allFindings.some((finding) => finding.type === 'incompleteRun');
37
+ }
38
+
39
+ /** What this log could not check, in plain sentences, each saying what to
40
+ * turn on for the next run where the detector names it. */
41
+ export function verdictGaps(allFindings , noFinishedStages ) {
42
+ const gaps = new Set ();
43
+ if (noFinishedStages) gaps.add(NO_FINISHED_STAGE_GAP);
44
+ if (isIncompleteRun(allFindings)) gaps.add(INCOMPLETE_RUN_GAP);
45
+ for (const finding of allFindings) {
46
+ if (isEvidenceCaveat(finding) && finding.recommendation) gaps.add(finding.recommendation);
47
+ }
48
+ return [...gaps];
49
+ }
50
+
51
+ /** The one rule for calling a run clean, shared by the verdict, the top bar
52
+ * and the evidence report: no finding at all, no failed job, and nothing the
53
+ * log lacked to run a check. */
54
+ export function isCleanRun(appModel , allFindings ) {
55
+ if (summarizeRunOutcome(appModel.jobs, allFindings).failedJobs > 0) return false;
56
+ if (allFindings.some(isRealFinding)) return false;
57
+ return verdictGaps(allFindings, !hasFinishedStage(appModel.stages)).length === 0;
58
+ }
59
+
60
+
61
+
62
+
63
+
64
+
65
+
66
+
67
+
68
+ /** The not-run rule for one run: an evidence caveat of that type, a per-stage
69
+ * check on a log where no stage finished, or a run-span check on a log with
70
+ * no ApplicationEnd. */
71
+ export function checkCoverage(stages , allFindings ) {
72
+ const noFinishedStages = !hasFinishedStage(stages);
73
+ const incomplete = isIncompleteRun(allFindings);
74
+ const caveatReasons = new Map ();
75
+ for (const finding of allFindings) {
76
+ if (!isEvidenceCaveat(finding)) continue;
77
+ if (!caveatReasons.get(finding.type)) caveatReasons.set(finding.type, finding.recommendation ?? null);
78
+ }
79
+ const notRunReason = (type ) => {
80
+ const caveatReason = caveatReasons.get(type);
81
+ if (caveatReason) return caveatReason;
82
+ if (noFinishedStages && PER_STAGE_CHECK_TYPES.has(type)) return NO_FINISHED_STAGE_GAP;
83
+ if (incomplete && RUN_SPAN_CHECK_TYPES.has(type)) return INCOMPLETE_RUN_GAP;
84
+ // A caveat with no recommendation still means the check did not run.
85
+ return caveatReasons.has(type) ? 'The log lacked the data this check needs.' : null;
86
+ };
87
+ return { isNotRun: (type) => notRunReason(type) != null, notRunReason };
88
+ }