sparkforensics-mcp 0.4.0 → 0.5.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (64) hide show
  1. package/package.json +1 -1
  2. package/vendor-core/allocation.js +106 -0
  3. package/vendor-core/analyzer.js +13 -13
  4. package/vendor-core/cli/budgets.js +23 -9
  5. package/vendor-core/cli/collect-run.js +11 -4
  6. package/vendor-core/cli/regression-budgets.js +83 -0
  7. package/vendor-core/comparison-verdict.js +22 -22
  8. package/vendor-core/core-source-hash.txt +1 -1
  9. package/vendor-core/detectors.js +202 -68
  10. package/vendor-core/docs-content/detection/cache.md +3 -2
  11. package/vendor-core/docs-content/detection/cfg.md +9 -8
  12. package/vendor-core/docs-content/detection/chrn.md +1 -2
  13. package/vendor-core/docs-content/detection/cold.md +4 -2
  14. package/vendor-core/docs-content/detection/fail.md +3 -2
  15. package/vendor-core/docs-content/detection/gc.md +3 -2
  16. package/vendor-core/docs-content/detection/host.md +2 -1
  17. package/vendor-core/docs-content/detection/local.md +1 -1
  18. package/vendor-core/docs-content/detection/mem.md +5 -2
  19. package/vendor-core/docs-content/detection/plan.md +2 -1
  20. package/vendor-core/docs-content/detection/sfail.md +2 -1
  21. package/vendor-core/docs-content/detection/shape.md +5 -4
  22. package/vendor-core/docs-content/detection/skew.md +3 -1
  23. package/vendor-core/docs-content/detection/slow.md +2 -2
  24. package/vendor-core/docs-content/detection/spec.md +2 -3
  25. package/vendor-core/docs-content/detection/spill.md +1 -1
  26. package/vendor-core/docs-site-config.js +1 -1
  27. package/vendor-core/effective-conf.js +107 -0
  28. package/vendor-core/efficiency-model.js +8 -6
  29. package/vendor-core/event-handlers.js +160 -47
  30. package/vendor-core/event-schemas.js +2 -0
  31. package/vendor-core/evidence-report.js +18 -10
  32. package/vendor-core/finding-generic-recommendation.js +20 -1
  33. package/vendor-core/finding-names.js +7 -0
  34. package/vendor-core/finding-presentation.js +61 -26
  35. package/vendor-core/finding-tag-help.js +1 -1
  36. package/vendor-core/finding-types.js +12 -0
  37. package/vendor-core/format-utils.js +4 -3
  38. package/vendor-core/impact-estimator.js +20 -2
  39. package/vendor-core/impact-format.js +14 -13
  40. package/vendor-core/impact-model.js +27 -5
  41. package/vendor-core/ingest.js +4 -2
  42. package/vendor-core/list-runs.js +5 -2
  43. package/vendor-core/mcp-tools.js +1 -1
  44. package/vendor-core/model-assembler.js +23 -1
  45. package/vendor-core/parser-worker.js +1 -1
  46. package/vendor-core/proxy.js +3 -1
  47. package/vendor-core/python-stage.js +25 -0
  48. package/vendor-core/recommendation-rollup.js +16 -9
  49. package/vendor-core/redact.js +51 -10
  50. package/vendor-core/remediation.js +20 -0
  51. package/vendor-core/run-comparison.js +43 -24
  52. package/vendor-core/run-interpretation.js +2 -1
  53. package/vendor-core/run-metrics.js +198 -0
  54. package/vendor-core/run-totals.js +24 -0
  55. package/vendor-core/run-verdict.js +3 -4
  56. package/vendor-core/scorecard-estimates.js +1 -0
  57. package/vendor-core/session-snapshot.js +7 -0
  58. package/vendor-core/shs-schemas.js +2 -2
  59. package/vendor-core/spark-memory.js +17 -0
  60. package/vendor-core/stage-plan-nodes.js +18 -0
  61. package/vendor-core/stage-quantiles.js +4 -0
  62. package/vendor-core/types.js +49 -1
  63. package/vendor-core/wasted-core-hours.js +10 -7
  64. package/vendor-core/write-targets.js +312 -0
@@ -1,7 +1,8 @@
1
1
  ### `CACHE`: Caching opportunity {#cache}
2
2
 
3
- A reusable dataset (re-read via the same SQL relation more than once) may be
4
- worth persisting between stages. Self-flags a confidence that scales with
3
+ A reusable dataset (re-read via the same SQL relation more than once, or a
4
+ join/union result recomputed by two or more executions, matched by plan
5
+ shape) may be worth persisting between stages. Self-flags a confidence that scales with
5
6
  how many executions reuse the same relation: reuse is only inferred, from
6
7
  plan-scan identity across SQL executions, so confirm the reads really do
7
8
  hit the same data before you cache anything.
@@ -1,15 +1,16 @@
1
1
  ### `CFG`: Configuration audit {#cfg}
2
2
 
3
3
  Flags configuration settings that may cause reliability or efficiency
4
- problems, independent of any one stage's behavior. Four properties are
5
- audited today:
4
+ problems, independent of any one stage's behavior. Four checks run:
6
5
 
7
6
  - `spark.shuffle.service.enabled`: flagged when dynamic allocation is on
8
7
  but the external shuffle service is off, since shuffle data won't survive
9
8
  executor removal.
10
- - `spark.dynamicAllocation.maxExecutors`: flagged for inverted bounds or a
11
- missing upper bound.
12
- - `spark.serializer`: flagged when still on the default Java serializer;
13
- `org.apache.spark.serializer.KryoSerializer` is faster and produces
14
- smaller buffers.
15
- - `spark.executor.memoryOverhead`: flagged when set below a safe floor.
9
+ - `spark.dynamicAllocation.minExecutors`/`maxExecutors`: with dynamic
10
+ allocation on, flagged when min exceeds max (reported on `minExecutors`)
11
+ or when no max is set.
12
+ - `spark.serializer`: flagged when not set to Kryo (the default is the Java
13
+ serializer); `org.apache.spark.serializer.KryoSerializer` is faster and
14
+ produces smaller buffers.
15
+ - `spark.executor.memoryOverhead`: flagged when set below max(384 MiB, 10%
16
+ of executor memory).
@@ -5,5 +5,4 @@ re-provisioning churn rather than normal scale-down. Raise
5
5
  `spark.dynamicAllocation.executorIdleTimeout`, or widen the
6
6
  `minExecutors`/`maxExecutors` bounds to reduce flapping. Self-flags a
7
7
  confidence that scales with how far the short-lived-executor share sits
8
- past the threshold: these thresholds are still a design spike, not yet
9
- validated against real-world runs.
8
+ past the threshold.
@@ -1,4 +1,6 @@
1
1
  ### `COLD`: Executor cold start {#cold}
2
2
 
3
- New executors take time to become available for work. Pre-warm the cluster,
4
- or use dynamic allocation.
3
+ The first stage waited more than 30 s for an executor. Keep a warm pool of
4
+ executors, or, with dynamic allocation, raise
5
+ `spark.dynamicAllocation.minExecutors`/`initialExecutors` so the app doesn't
6
+ scale up from zero.
@@ -5,5 +5,6 @@ instability or data-driven errors. The finding names the dominant error: the
5
5
  exception class, or the executor loss reason (for example "Container killed
6
6
  by YARN for exceeding memory limits"). It lists up to five distinct failures,
7
7
  each with its message and a short stack excerpt. With redaction on, messages
8
- and the message text inside excerpts are replaced, since they can carry file
9
- paths and data values; class names and stack frames stay.
8
+ become `[redacted]` and message lines inside excerpts are dropped, since they
9
+ can carry file paths and data values; class names, stack frames and the
10
+ executor loss reason stay (hosts in it are pseudonymized).
@@ -1,6 +1,7 @@
1
1
  ### `GC`: Garbage collection pressure {#gc}
2
2
 
3
- Tasks spend an unusually large share of time reclaiming memory. Reduce
3
+ Tasks spend more than 10% of executor run time reclaiming memory. Reduce
4
4
  object creation: use primitive types, avoid UDFs, or raise executor memory.
5
- A stage with very little GC gets an informational note that executor memory
5
+ A stage with GC below 5% gets an informational note that executor memory
6
6
  may be over-provisioned, only on stages that take at least 0.5% of the run.
7
+ Both need at least 10 s of executor run time on the stage.
@@ -1,6 +1,7 @@
1
1
  ### `HOST`: Slow host {#host}
2
2
 
3
- One executor is much slower than its peers. It may just hold data locality
3
+ One executor is much slower than its peers, or carries most of the stage's
4
+ task time or bytes. It may just hold data locality
4
5
  for its tasks or carry one heavy stage, rather than a hardware fault.
5
6
  Enable `spark.speculation` to relaunch a lagging task automatically. Only
6
7
  flagged on stages that take at least 0.5% of the run.
@@ -1,6 +1,6 @@
1
1
  ### `LOCAL`: Core usage locality {#local}
2
2
 
3
3
  Tasks run without process- or node-local data placement more often than
4
- expected. Check `spark.locality.wait` settings and executor/data colocation.
4
+ expected. Check executor/data colocation.
5
5
  Self-flags a confidence that scales with the non-local ratio and sample
6
6
  size: the thresholds are our own noise floor for this metric.
@@ -1,10 +1,13 @@
1
1
  ### `MEM`: Memory utilization {#mem}
2
2
 
3
- Executor memory or core capacity may be over- or under-provisioned. Some
3
+ Executor memory or core capacity may be over- or under-provisioned: more
4
+ than 50% of available core time ran no task, an executor's heap peaked above
5
+ 95% of its allocation, or it stayed below 70%. Some
4
6
  detail here needs `spark.eventLog.logStageExecutorMetrics=true` on the run
5
7
  being analyzed; without it, per-executor memory usage can't be broken down.
6
8
  Review `spark.executor.memory` and executor count if allocated memory sat
7
9
  largely idle over the run. That idle-memory variant self-flags a confidence
8
10
  that scales with how far the estimated waste sits past a 1.5x buffer: it
9
- estimates waste from allocated-versus-used memory-time. Check it against
11
+ estimates waste from allocated memory-time versus task run time (not
12
+ measured heap usage). Check it against
10
13
  the Spark UI before resizing anything.
@@ -7,7 +7,8 @@ this tag:
7
7
  plan. When the repeats have the same shape but different filters, columns
8
8
  or tables, the finding stays informational and claims no time. Only flagged
9
9
  when the repeat's stages take at least 0.5% of the run.
10
- - Small files: reading an excessive number of small files.
10
+ - Small files: one plan node reads or writes more than 100 files averaging
11
+ under 3 MB. Compact upstream output, or coalesce before writing.
11
12
  - Under-broadcast: the smaller side of a Sort Merge Join looks well under
12
13
  the broadcast threshold; consider a `broadcast()` hint or raising
13
14
  `spark.sql.autoBroadcastJoinThreshold`.
@@ -2,4 +2,5 @@
2
2
 
3
3
  A stage attempt failed outright rather than losing individual tasks within
4
4
  it. Inspect the driver log for the failure reason and the job that triggered
5
- it.
5
+ it. With redaction on, the failure reason becomes `[redacted]`, since it can
6
+ carry file paths and data values.
@@ -1,6 +1,7 @@
1
1
  ### `SHAPE`: Stage shape {#shape}
2
2
 
3
- The stage has an inefficient task count, output shape, or task-to-stage
4
- balance: for example, one straggler task taking a large fraction of the
5
- stage's wall-clock time. A too-low task count is only flagged on stages that
6
- take at least 0.5% of the run.
3
+ The stage has an inefficient task count, output shape (output more than 10×
4
+ input), or task-to-stage balance: one straggler task running for more than
5
+ half the stage's wall-clock time and over 3× the median task, so it alone
6
+ sets when the stage ends. A too-low task count and a straggler are only
7
+ flagged on stages that take at least 0.5% of the run.
@@ -3,4 +3,6 @@
3
3
  A small number of tasks take much longer than their peers in the same
4
4
  stage. For join-driven skew, enable AQE skew-join handling
5
5
  (`spark.sql.adaptive.skewJoin.enabled`); otherwise salt the key or
6
- repartition on a better key.
6
+ repartition on a better key. Flagged when P95 task time (the longest task,
7
+ on a stage with fewer than 20 tasks) exceeds 3x the median and the
8
+ recoverable tail is at least 0.5% of the run.
@@ -1,6 +1,6 @@
1
1
  ### `SLOW`: Stage slowness {#slow}
2
2
 
3
- A stage ran long overall without a more specific cause getting flagged.
4
- Often a partition-count problem: raise parallelism via
3
+ A stage ran for 15 minutes or more and no slow host was flagged on it. It
4
+ can appear alongside other findings on the same stage. Often a partition-count problem: raise parallelism via
5
5
  `spark.sql.shuffle.partitions` or `spark.default.parallelism`, or check for a
6
6
  large per-task data volume driving heavy shuffle and spill.
@@ -2,7 +2,6 @@
2
2
 
3
3
  Speculative task attempts used a lot of executor time without confirming a
4
4
  genuine straggler. Self-flags a confidence that scales with how far the
5
- wasted time sits past the threshold: these thresholds are still a design
6
- spike, not yet validated against real-world runs. If task durations are
7
- just naturally variable rather than genuine stragglers, tune
5
+ wasted time sits past the threshold. If task durations are just naturally
6
+ variable rather than genuine stragglers, tune
8
7
  `spark.speculation.multiplier`/`spark.speculation.quantile`.
@@ -4,4 +4,4 @@ Tasks are writing data out of memory, which slows execution. Two spill
4
4
  patterns get flagged differently: skew spill, where a few heavy tasks spill
5
5
  while most don't (rebalance partitioning), and volume spill, where most
6
6
  tasks spill because the data genuinely exceeds available memory (add
7
- partitions). Only flagged on stages that take at least 0.5% of the run.
7
+ partitions or executor memory). Only flagged on stages that take at least 0.5% of the run.
@@ -15,5 +15,5 @@ export function findingGuideUrl(type ) {
15
15
  return `docs/user-guide/understanding-findings.html#${typeTag(type).toLowerCase()}`;
16
16
  }
17
17
 
18
- // Every other way to get a log (cloud consoles, bastions, copying from storage).
18
+ // How to find a Spark event log: enabling event logging, the History Server, managed platforms, SSH bastions.
19
19
  export const ALTERNATIVE_LOG_RETRIEVAL_URL = 'docs/user-guide/alternative-log-retrieval.html';
@@ -0,0 +1,107 @@
1
+ // The effective Spark configuration of a run for the CLI's JSON output, so a caller can check
2
+ // that a --conf overlay took effect. Every key is listed; a value is withheld when the key or the
3
+ // value matches a secret pattern, and credentials inside URL-like values are stripped. A withheld
4
+ // key is shown as present with no value, and no derivative of the value (no hash, length or
5
+ // prefix) is emitted.
6
+
7
+
8
+ export const EFFECTIVE_CONF_SCHEMA_VERSION = 1;
9
+
10
+ /** Spark's default spark.redaction.regex, in JS syntax (the JVM pattern is `(?i)` + this). */
11
+ export const DEFAULT_SECRET_PATTERN = 'secret|password|token|access[.]?key';
12
+ /** Further credential names no Spark default covers, tested on keys (Azure `fs.azure.account.key.*`,
13
+ * `apiKey`, `pwd`, a bare `pass` or `sas` segment, `credential`). */
14
+ export const EXTRA_SECRET_KEY_PATTERN = 'passwd|pwd|(?:^|[._-])pass(?:[._-]|$)|api[._-]?key|account[._-]?key|private[._-]?key|credential|(?:^|[._-])sas(?:[._-]|$)|(?:^|[._-])sig(?:nature)?(?:[._-]|$)';
15
+ const REDACTED = '[redacted]';
16
+
17
+
18
+
19
+
20
+
21
+
22
+
23
+
24
+
25
+
26
+
27
+
28
+
29
+
30
+
31
+
32
+
33
+
34
+ /** Compiles a JVM regex, which may start with inline flags such as (?i), as a JS RegExp. Null when
35
+ * it uses syntax JS lacks. */
36
+ export function compileJvmPattern(source ) {
37
+ const lead = /^\(\?([a-z]+)\)/.exec(source);
38
+ const flags = new Set ();
39
+ let body = source;
40
+ if (lead) {
41
+ body = source.slice(lead[0].length);
42
+ for (const f of lead[1]) {
43
+ if (f === 'i' || f === 'm' || f === 's') flags.add(f);
44
+ else if (f !== 'u') return null;
45
+ }
46
+ }
47
+ try { return new RegExp(body, [...flags].join('')); } catch { return null; }
48
+ }
49
+
50
+ // Sensitive query/parameter names in a URL-like value: signatures of pre-signed and SAS URLs.
51
+ const SIGNATURE_PARAMS = 'sig|signature|x-amz-signature|x-amz-credential|x-amz-security-token|x-goog-signature|x-goog-credential';
52
+
53
+ // A `name=value` credential parameter in any value (a JDBC or ODBC string, a JVM option, a query
54
+ // string): `password`, `pwd`, `pass`, `apikey`, `accountkey`, `sas`, `sig` and the like, alone or as
55
+ // the tail of a longer name (`-Djavax.net.ssl.keyStorePassword=`).
56
+ const CREDENTIAL_PARAM = new RegExp(
57
+ '((?:^|[\\s;&,?:"\'(]|-D)(?:[\\w.-]*?(?:password|passwd|pwd|secret|token|api[_.-]?key|account[_.-]?key|access[_.-]?key|private[_.-]?key)|pass|sas|credential|'
58
+ + `${SIGNATURE_PARAMS})\\s*=\\s*)("[^"]*"|'[^']*'|[^;&,\\s"']*)`, 'gi');
59
+
60
+ /** Strips credentials from a value: `name=value` credential parameters anywhere (JDBC
61
+ * `password=`/`pwd=`/`pass=`, `apiKey=`, signature parameters such as Azure SAS `sig=`), and, in a
62
+ * value that looks like a URL, userinfo (user:password@host) and the Oracle thin form
63
+ * (jdbc:oracle:thin:user/password@host). */
64
+ export function stripUrlCredentials(value ) {
65
+ const urlLike = /^[a-z][a-z0-9+.-]*:/i.test(value) || value.includes('://');
66
+ const withoutUserinfo = urlLike
67
+ ? value
68
+ .replace(/(:\/\/)[^/\s@?#,]*@/g, `$1${REDACTED}@`)
69
+ .replace(/^((?:[a-z][a-z0-9+.-]*:)+[^\s:/@]+\/)[^\s@]+@/i, `$1${REDACTED}@`)
70
+ : value;
71
+ return withoutUserinfo.replace(CREDENTIAL_PARAM, `$1${REDACTED}`);
72
+ }
73
+
74
+ /** The run's Spark Properties as the CLI reports them; null when the log recorded none. */
75
+ export function buildEffectiveConf(app , options = {}) {
76
+ const config = app?.config;
77
+ if (config == null) return null;
78
+ const defaultPattern = new RegExp(DEFAULT_SECRET_PATTERN, 'i');
79
+ const extraKeyPattern = new RegExp(EXTRA_SECRET_KEY_PATTERN, 'i');
80
+ const jobSource = config['spark.redaction.regex'] ?? null;
81
+ const jobPattern = jobSource != null ? compileJvmPattern(jobSource) : null;
82
+ // A job pattern this runtime cannot evaluate withholds every value rather than guess.
83
+ const jobPatternUsable = jobSource == null || jobPattern != null;
84
+ const userSource = options.userPattern ?? null;
85
+ const userPattern = userSource != null ? compileJvmPattern(userSource) : null;
86
+ if (userSource != null && userPattern == null) throw new Error(`invalid secret pattern: ${userSource}`);
87
+
88
+ const wanted = options.keys ? new Set(options.keys) : null;
89
+ const values = {};
90
+ const maskedKeys = [];
91
+ const matches = (pattern , key , value ) => (pattern?.test(key) ?? false) || (pattern?.test(value) ?? false);
92
+ for (const key of Object.keys(config).sort()) {
93
+ if (wanted && !wanted.has(key)) continue;
94
+ const value = config[key];
95
+ // Spark applies its redaction pattern to the key and the value.
96
+ const masked = !jobPatternUsable
97
+ || matches(defaultPattern, key, value) || extraKeyPattern.test(key) || matches(jobPattern, key, value) || matches(userPattern, key, value);
98
+ if (masked) maskedKeys.push(key);
99
+ else values[key] = stripUrlCredentials(value);
100
+ }
101
+ return {
102
+ schemaVersion: EFFECTIVE_CONF_SCHEMA_VERSION,
103
+ values,
104
+ maskedKeys,
105
+ absentKeys: wanted ? [...wanted].filter((k) => !(k in config)).sort() : [],
106
+ };
107
+ }
@@ -1,13 +1,14 @@
1
1
  // §5 Efficiency/wastage model. DESIGN SPIKE:
2
2
  // decomposes available compute-hours into driver-bound vs executor-bound waste, plus two floors.
3
3
  import { computeWallClock } from './wall-clock.js';
4
- import { computeTotalCores } from './core-count.js';
4
+ import { computePeakConcurrentCores } from './core-count.js';
5
5
 
6
6
 
7
- export function computeEfficiencyModel({ app, stages, executorsAdded, runAggregates }
7
+ export function computeEfficiencyModel({ app, stages, executorsAdded, executorsRemoved, runAggregates }
8
8
 
9
9
 
10
10
 
11
+
11
12
 
12
13
  )
13
14
 
@@ -17,10 +18,11 @@ export function computeEfficiencyModel({ app, stages, executorsAdded, runAggrega
17
18
 
18
19
 
19
20
  {
20
- // `app ?? {}`: computeTotalCores falls back to the executor core sum when resources is absent,
21
- // and callers tolerate a null app (malformed logs); `app!` would crash on app.resources.
22
- // Cast: computeTotalCores reads only totalCores, absent on ExecutorRemovedEvent, so the union mismatches.
23
- const totalCores = computeTotalCores(app ?? {}, executorsAdded );
21
+ // Capacity is the peak concurrent core count, the one the utilization and idle-cores detectors
22
+ // use, so the Unused core time tile matches the verdict's idle figure. Summing every addition
23
+ // (computeTotalCores) counts a replaced executor's cores alongside its replacement's.
24
+ // `app ?? {}`: callers tolerate a null app (malformed logs); `app!` would crash on app.resources.
25
+ const totalCores = computePeakConcurrentCores(app ?? {}, executorsAdded, executorsRemoved);
24
26
  const appDurationMs = (app?.endTime ?? 0) - (app?.startTime ?? 0);
25
27
  const wc = computeWallClock(app, stages );
26
28