sparkforensics-mcp 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (60) hide show
  1. package/bin/sparkforensics-mcp.mjs +31 -0
  2. package/package.json +30 -0
  3. package/vendor-core/analyzer.js +167 -0
  4. package/vendor-core/assert-never.js +3 -0
  5. package/vendor-core/cli/budgets.js +203 -0
  6. package/vendor-core/cli/collect-run.js +107 -0
  7. package/vendor-core/core-count.js +68 -0
  8. package/vendor-core/core-locality-ratio.js +54 -0
  9. package/vendor-core/core-time-series.js +92 -0
  10. package/vendor-core/core-usage-locality.js +70 -0
  11. package/vendor-core/detectors.js +1989 -0
  12. package/vendor-core/docs-config.js +72 -0
  13. package/vendor-core/docs-site-config.js +23 -0
  14. package/vendor-core/efficiency-model.js +62 -0
  15. package/vendor-core/etl-phases.js +28 -0
  16. package/vendor-core/event-handlers.js +906 -0
  17. package/vendor-core/event-schemas.js +405 -0
  18. package/vendor-core/evidence-availability.js +121 -0
  19. package/vendor-core/evidence-report.js +459 -0
  20. package/vendor-core/finding-action-label.js +97 -0
  21. package/vendor-core/finding-filter-predicate.js +38 -0
  22. package/vendor-core/format-utils.js +167 -0
  23. package/vendor-core/impact-band.js +50 -0
  24. package/vendor-core/impact-estimator.js +428 -0
  25. package/vendor-core/ingest.js +139 -0
  26. package/vendor-core/job-groups.js +30 -0
  27. package/vendor-core/load-vendored.js +24 -0
  28. package/vendor-core/lz4-block.js +135 -0
  29. package/vendor-core/mcp-error.js +3 -0
  30. package/vendor-core/mcp-server-factory.js +115 -0
  31. package/vendor-core/mcp-tools.js +331 -0
  32. package/vendor-core/model-assembler.js +76 -0
  33. package/vendor-core/occupancy.js +202 -0
  34. package/vendor-core/parser-worker.js +249 -0
  35. package/vendor-core/plan-dot.js +25 -0
  36. package/vendor-core/plan-duration-attribution.js +185 -0
  37. package/vendor-core/plan-graph-model.js +171 -0
  38. package/vendor-core/plan-node-detail.js +159 -0
  39. package/vendor-core/plan-summary.js +233 -0
  40. package/vendor-core/plan-tree-walk.js +29 -0
  41. package/vendor-core/proxy.js +157 -0
  42. package/vendor-core/recommendation-rollup.js +197 -0
  43. package/vendor-core/redact.js +175 -0
  44. package/vendor-core/rolling-log-reassembly.js +52 -0
  45. package/vendor-core/run-aggregates.js +44 -0
  46. package/vendor-core/run-comparison.js +458 -0
  47. package/vendor-core/scaling-sim.js +73 -0
  48. package/vendor-core/session-snapshot.js +79 -0
  49. package/vendor-core/shs-fetch.js +196 -0
  50. package/vendor-core/shs-load.js +121 -0
  51. package/vendor-core/shs-request.js +101 -0
  52. package/vendor-core/shs-schemas.js +13 -0
  53. package/vendor-core/snappy-block.js +140 -0
  54. package/vendor-core/stage-quantiles.js +199 -0
  55. package/vendor-core/threshold-summary.js +35 -0
  56. package/vendor-core/types.js +286 -0
  57. package/vendor-core/vendor/fflate.js +2695 -0
  58. package/vendor-core/vendor/fzstd.js +768 -0
  59. package/vendor-core/wall-clock.js +36 -0
  60. package/vendor-core/wasted-core-hours.js +68 -0
@@ -0,0 +1,54 @@
1
+ // Whole-run core-usage-locality ratio: non-local task share across every
2
+ // stage's `stage.localityStats`. Mirrors wasted-core-hours.js's shape (pure
3
+ // reducer, no detector or view coupling), importable by both the
4
+ // `coreLocality` detector (src/detectors.js) and CoreUsageArea.tsx's
5
+ // stage-breakdown list.
6
+ //
7
+ // Locality classification (Spark, best to worst): PROCESS_LOCAL > NODE_LOCAL
8
+ // > NO_PREF > RACK_LOCAL > ANY. NO_PREF is not a locality failure; it's what
9
+ // shuffle-read stages report because there's no location-preference concept
10
+ // for a shuffle fetch, so it stays in the denominator (diluting the ratio for
11
+ // shuffle-heavy stages, which is correct) but never in the numerator.
12
+ const NON_LOCAL_TIERS = new Set(['RACK_LOCAL', 'ANY']);
13
+ const TOP_N = 5;
14
+ const MIN_TASKS_PER_STAGE = 10;
15
+
16
+ const EMPTY = { totalTasks: null, nonLocalTasks: null, ratio: null, topStages: [] };
17
+
18
+ export function computeCoreLocalityRatio(stages , { minTasksPerStage = MIN_TASKS_PER_STAGE, topN = TOP_N } = {}) {
19
+ if (!Array.isArray(stages) || stages.length === 0) return EMPTY;
20
+
21
+ let totalTasks = 0;
22
+ let nonLocalTasks = 0;
23
+ const perStage = [];
24
+
25
+ for (const stage of stages) {
26
+ const localityStats = stage?.localityStats;
27
+ if (!Array.isArray(localityStats) || localityStats.length === 0) continue;
28
+
29
+ let stageTotal = 0;
30
+ let stageNonLocal = 0;
31
+ for (const { locality, count } of localityStats) {
32
+ if (!Number.isFinite(count)) continue;
33
+ stageTotal += count;
34
+ if (NON_LOCAL_TIERS.has(locality)) stageNonLocal += count;
35
+ }
36
+ totalTasks += stageTotal;
37
+ nonLocalTasks += stageNonLocal;
38
+
39
+ if (stageTotal >= minTasksPerStage) {
40
+ perStage.push({
41
+ stageId: stage.id,
42
+ nonLocalTasks: stageNonLocal,
43
+ taskCount: stageTotal,
44
+ ratio: stageNonLocal / stageTotal,
45
+ });
46
+ }
47
+ }
48
+
49
+ if (totalTasks === 0) return EMPTY;
50
+
51
+ const topStages = perStage.sort((a, b) => b.nonLocalTasks - a.nonLocalTasks).slice(0, topN);
52
+
53
+ return { totalTasks, nonLocalTasks, ratio: nonLocalTasks / totalTasks, topStages };
54
+ }
@@ -0,0 +1,92 @@
1
+ // Sweep-line "busy cores over time" computation. Pure and side-effect-free so
2
+ // it can run inside parser-worker.js (over retained task launch/finish
3
+ // timestamps) or on the main thread (scaling simulator). One core per task:
4
+ // Spark's default task-to-core mapping.
5
+ //
6
+ // `hypotheticalCores` clamps the concurrent busy-core count to N (the
7
+ // deterministic "utilization at N cores" curve). It does NOT re-schedule tasks;
8
+ // makespan re-estimation is the scaling simulator's job, layered on this signal.
9
+
10
+
11
+
12
+
13
+
14
+
15
+
16
+
17
+
18
+
19
+
20
+
21
+ function buildStepFunction(intervals , hypotheticalCores ) {
22
+ // Emit +1 at launch, -1 at finish. Skip empty/negative intervals.
23
+ const events = [];
24
+ for (const { launch, finish } of intervals) {
25
+ if (!(finish > launch)) continue;
26
+ events.push({ t: launch, delta: 1 });
27
+ events.push({ t: finish, delta: -1 });
28
+ }
29
+ // Sort by time; at equal time process -1 before +1 so [launch, finish) is
30
+ // half-open (a task finishing exactly as another launches does not overlap).
31
+ events.sort((a, b) => (a.t - b.t) || (a.delta - b.delta));
32
+
33
+ // Walk consecutive timestamps, emitting the busy level held on each interval.
34
+ const segments = [];
35
+ let busy = 0;
36
+ for (let i = 0; i < events.length; i++) {
37
+ const cur = events[i];
38
+ busy += cur.delta;
39
+ const next = events[i + 1];
40
+ if (!next || next.t === cur.t) continue;
41
+ const effective = hypotheticalCores != null ? Math.min(busy, hypotheticalCores) : busy;
42
+ segments.push({ tStart: cur.t, tEnd: next.t, busy: effective });
43
+ }
44
+ return segments;
45
+ }
46
+
47
+ export function computeCoreTimeSeries(intervals , {
48
+ bucketBy = 'time',
49
+ bucketWidthMs = 1000,
50
+ hypotheticalCores = null,
51
+ }
52
+
53
+
54
+
55
+ = {})
56
+
57
+
58
+
59
+
60
+
61
+ {
62
+ const segments = buildStepFunction(intervals, hypotheticalCores);
63
+
64
+ if (bucketBy === 'coreCount') {
65
+ const histogram = [];
66
+ for (const seg of segments) {
67
+ histogram[seg.busy] = (histogram[seg.busy] ?? 0) + (seg.tEnd - seg.tStart);
68
+ }
69
+ // Fill sparse holes with 0 so the array reads as a dense histogram.
70
+ for (let k = 0; k < histogram.length; k++) if (histogram[k] === undefined) histogram[k] = 0;
71
+ return { mode: 'coreCount', histogram };
72
+ }
73
+
74
+ // bucketBy === 'time'
75
+ if (segments.length === 0) {
76
+ return { mode: 'time', bucketWidthMs, startTime: null, endTime: null, buckets: [] };
77
+ }
78
+ const startTime = segments[0].tStart;
79
+ const endTime = segments[segments.length - 1].tEnd;
80
+ const buckets = [];
81
+ for (let tStart = startTime; tStart < endTime; tStart += bucketWidthMs) {
82
+ const tEnd = tStart + bucketWidthMs;
83
+ let busyCoreMs = 0;
84
+ for (const seg of segments) {
85
+ const lo = Math.max(seg.tStart, tStart);
86
+ const hi = Math.min(seg.tEnd, tEnd);
87
+ if (hi > lo) busyCoreMs += seg.busy * (hi - lo);
88
+ }
89
+ buckets.push({ tStart, tEnd, busyCoreMs, avgBusyCores: busyCoreMs / bucketWidthMs });
90
+ }
91
+ return { mode: 'time', bucketWidthMs, startTime, endTime, buckets };
92
+ }
@@ -0,0 +1,70 @@
1
+ // Stage-granular approximation of core-usage-by-locality over time. See the
2
+ // plan's Task 6 DEVIATION note: per-task locality is not retained, so this
3
+ // distributes each stage's executorRunTime across its wall-clock window,
4
+ // split by localityStats proportions. Not exact; approximate by construction.
5
+ export const LOCALITY_TIERS = ['PROCESS_LOCAL', 'NODE_LOCAL', 'RACK_LOCAL', 'NO_PREF', 'ANY'];
6
+
7
+
8
+
9
+
10
+
11
+
12
+
13
+
14
+
15
+
16
+
17
+
18
+
19
+
20
+
21
+
22
+
23
+
24
+ export function computeLocalityAreaSeries(
25
+ stages ,
26
+ { bucketWidthMs = 60_000, tiers = LOCALITY_TIERS } = {},
27
+ ) {
28
+ const valid = stages.filter(s => (s.completedAt ?? 0) > (s.submittedAt ?? 0) && (s.executorRunTime ?? 0) > 0);
29
+ if (valid.length === 0) return { labels: [], series: {} };
30
+ const startTime = valid.reduce((m, s) => Math.min(m, s.submittedAt ?? m), Infinity);
31
+ const endTime = valid.reduce((m, s) => Math.max(m, s.completedAt ?? m), -Infinity);
32
+ const nBuckets = Math.max(1, Math.ceil((endTime - startTime) / bucketWidthMs));
33
+
34
+ const series = {};
35
+ for (const tier of tiers) series[tier] = new Array(nBuckets).fill(0);
36
+ const other = new Array(nBuckets).fill(0); // localities not in the fixed tier list
37
+
38
+ for (const s of valid) {
39
+ const completedAt = s.completedAt ?? 0;
40
+ const submittedAt = s.submittedAt ?? 0;
41
+ const wall = completedAt - submittedAt;
42
+ const avgCores = (s.executorRunTime ?? 0) / wall; // core-time / wall-time = avg concurrent cores
43
+ const total = (s.localityStats ?? []).reduce((a, l) => a + l.count, 0) || 1;
44
+ const props = new Map((s.localityStats ?? []).map(l => [l.locality, l.count / total]));
45
+ // spread avgCores over the buckets this stage overlaps, weighted by overlap fraction
46
+ for (let b = 0; b < nBuckets; b++) {
47
+ const bStart = startTime + b * bucketWidthMs;
48
+ const bEnd = bStart + bucketWidthMs;
49
+ const overlap = Math.min(completedAt, bEnd) - Math.max(submittedAt, bStart);
50
+ if (overlap <= 0) continue;
51
+ const frac = overlap / bucketWidthMs; // portion of the bucket this stage covers
52
+ for (const [loc, p] of props) {
53
+ const add = avgCores * p * frac;
54
+ if (series[loc]) series[loc][b] += add;
55
+ else other[b] += add;
56
+ }
57
+ }
58
+ }
59
+ if (other.some(v => v > 0)) series.OTHER = other;
60
+
61
+ // idle = peak total busy across buckets, minus each bucket's total busy.
62
+ const totals = new Array(nBuckets).fill(0);
63
+ for (const tier of Object.keys(series)) for (let b = 0; b < nBuckets; b++) totals[b] += series[tier][b];
64
+ const peak = totals.reduce((m, v) => Math.max(m, v), 0);
65
+ series.idle = totals.map(v => Math.max(0, peak - v));
66
+
67
+ const labels = [];
68
+ for (let b = 0; b < nBuckets; b++) labels.push(startTime + b * bucketWidthMs);
69
+ return { labels, series };
70
+ }