sparkforensics-mcp 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/bin/sparkforensics-mcp.mjs +31 -0
- package/package.json +30 -0
- package/vendor-core/analyzer.js +167 -0
- package/vendor-core/assert-never.js +3 -0
- package/vendor-core/cli/budgets.js +203 -0
- package/vendor-core/cli/collect-run.js +107 -0
- package/vendor-core/core-count.js +68 -0
- package/vendor-core/core-locality-ratio.js +54 -0
- package/vendor-core/core-time-series.js +92 -0
- package/vendor-core/core-usage-locality.js +70 -0
- package/vendor-core/detectors.js +1989 -0
- package/vendor-core/docs-config.js +72 -0
- package/vendor-core/docs-site-config.js +23 -0
- package/vendor-core/efficiency-model.js +62 -0
- package/vendor-core/etl-phases.js +28 -0
- package/vendor-core/event-handlers.js +906 -0
- package/vendor-core/event-schemas.js +405 -0
- package/vendor-core/evidence-availability.js +121 -0
- package/vendor-core/evidence-report.js +459 -0
- package/vendor-core/finding-action-label.js +97 -0
- package/vendor-core/finding-filter-predicate.js +38 -0
- package/vendor-core/format-utils.js +167 -0
- package/vendor-core/impact-band.js +50 -0
- package/vendor-core/impact-estimator.js +428 -0
- package/vendor-core/ingest.js +139 -0
- package/vendor-core/job-groups.js +30 -0
- package/vendor-core/load-vendored.js +24 -0
- package/vendor-core/lz4-block.js +135 -0
- package/vendor-core/mcp-error.js +3 -0
- package/vendor-core/mcp-server-factory.js +115 -0
- package/vendor-core/mcp-tools.js +331 -0
- package/vendor-core/model-assembler.js +76 -0
- package/vendor-core/occupancy.js +202 -0
- package/vendor-core/parser-worker.js +249 -0
- package/vendor-core/plan-dot.js +25 -0
- package/vendor-core/plan-duration-attribution.js +185 -0
- package/vendor-core/plan-graph-model.js +171 -0
- package/vendor-core/plan-node-detail.js +159 -0
- package/vendor-core/plan-summary.js +233 -0
- package/vendor-core/plan-tree-walk.js +29 -0
- package/vendor-core/proxy.js +157 -0
- package/vendor-core/recommendation-rollup.js +197 -0
- package/vendor-core/redact.js +175 -0
- package/vendor-core/rolling-log-reassembly.js +52 -0
- package/vendor-core/run-aggregates.js +44 -0
- package/vendor-core/run-comparison.js +458 -0
- package/vendor-core/scaling-sim.js +73 -0
- package/vendor-core/session-snapshot.js +79 -0
- package/vendor-core/shs-fetch.js +196 -0
- package/vendor-core/shs-load.js +121 -0
- package/vendor-core/shs-request.js +101 -0
- package/vendor-core/shs-schemas.js +13 -0
- package/vendor-core/snappy-block.js +140 -0
- package/vendor-core/stage-quantiles.js +199 -0
- package/vendor-core/threshold-summary.js +35 -0
- package/vendor-core/types.js +286 -0
- package/vendor-core/vendor/fflate.js +2695 -0
- package/vendor-core/vendor/fzstd.js +768 -0
- package/vendor-core/wall-clock.js +36 -0
- package/vendor-core/wasted-core-hours.js +68 -0
|
@@ -0,0 +1,31 @@
|
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
// bin/sparkforensics-mcp.mjs
|
|
3
|
+
import { existsSync } from 'node:fs';
|
|
4
|
+
import { join, dirname } from 'node:path';
|
|
5
|
+
import { fileURLToPath, pathToFileURL } from 'node:url';
|
|
6
|
+
import { StdioServerTransport } from '@modelcontextprotocol/sdk/server/stdio.js';
|
|
7
|
+
|
|
8
|
+
const binDir = dirname(fileURLToPath(import.meta.url));
|
|
9
|
+
const pkgDir = dirname(binDir);
|
|
10
|
+
|
|
11
|
+
// vendor-core/ exists only in a published/standalone install (populated by
|
|
12
|
+
// scripts/vendor-core.mjs at pack time); in the monorepo it falls back to
|
|
13
|
+
// the real packages/core/src/ sibling. load-vendored.js itself is the one
|
|
14
|
+
// module this bootstrap has to locate by hand (see its own header comment);
|
|
15
|
+
// every other core module then loads through its exported loadVendored().
|
|
16
|
+
async function loadCreateMcpServer() {
|
|
17
|
+
const vendoredHelper = join(pkgDir, 'vendor-core', 'load-vendored.js');
|
|
18
|
+
const helperPath = existsSync(vendoredHelper) ? vendoredHelper : join(pkgDir, '..', 'core', 'src', 'load-vendored.js');
|
|
19
|
+
const { loadVendored } = await import(pathToFileURL(helperPath).href);
|
|
20
|
+
const mod = await loadVendored(pkgDir, 'mcp-server-factory');
|
|
21
|
+
return mod.createMcpServer;
|
|
22
|
+
}
|
|
23
|
+
|
|
24
|
+
async function main() {
|
|
25
|
+
const createMcpServer = await loadCreateMcpServer();
|
|
26
|
+
const server = createMcpServer();
|
|
27
|
+
const transport = new StdioServerTransport();
|
|
28
|
+
await server.connect(transport);
|
|
29
|
+
}
|
|
30
|
+
|
|
31
|
+
await main();
|
package/package.json
ADDED
|
@@ -0,0 +1,30 @@
|
|
|
1
|
+
{
|
|
2
|
+
"name": "sparkforensics-mcp",
|
|
3
|
+
"version": "0.1.0",
|
|
4
|
+
"mcpName": "io.github.shuffle-works/sparkforensics-mcp",
|
|
5
|
+
"type": "module",
|
|
6
|
+
"license": "MIT",
|
|
7
|
+
"description": "MCP server exposing Apache Spark event-log diagnostics (diagnose_run, get_run_summary, compare_runs, get_finding_evidence, evaluate_budgets) over stdio.",
|
|
8
|
+
"repository": {
|
|
9
|
+
"type": "git",
|
|
10
|
+
"url": "git+https://github.com/shuffle-works/sparkforensics.git"
|
|
11
|
+
},
|
|
12
|
+
"bin": {
|
|
13
|
+
"sparkforensics-mcp": "./bin/sparkforensics-mcp.mjs"
|
|
14
|
+
},
|
|
15
|
+
"engines": {
|
|
16
|
+
"node": ">=18"
|
|
17
|
+
},
|
|
18
|
+
"files": ["bin/", "vendor-core/"],
|
|
19
|
+
"scripts": {
|
|
20
|
+
"prepack": "node ../../scripts/vendor-core.mjs .",
|
|
21
|
+
"test": "vitest run"
|
|
22
|
+
},
|
|
23
|
+
"dependencies": {
|
|
24
|
+
"@modelcontextprotocol/sdk": "^1.30.0",
|
|
25
|
+
"zod": "^4.4.3"
|
|
26
|
+
},
|
|
27
|
+
"devDependencies": {
|
|
28
|
+
"vitest": "^4.1.10"
|
|
29
|
+
}
|
|
30
|
+
}
|
|
@@ -0,0 +1,167 @@
|
|
|
1
|
+
import { DETECTORS, } from './detectors.js';
|
|
2
|
+
import { computePeakConcurrentCores } from './core-count.js';
|
|
3
|
+
import { assertNever } from './assert-never.js';
|
|
4
|
+
import { estimateImpact } from './impact-estimator.js';
|
|
5
|
+
import { computeOccupancy, } from './occupancy.js';
|
|
6
|
+
import { deriveImpactBand } from './impact-band.js';
|
|
7
|
+
import { IMPACT_BAND_ORDER } from './format-utils.js';
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
// FNV-1a 32-bit: a small, dependency-free stable string hash. Used to derive
|
|
13
|
+
// a deterministic finding `id` at the single choke point below so the same
|
|
14
|
+
// logical finding keeps the same id across runs (no timestamps, no randomness).
|
|
15
|
+
function fnv1a(str ) {
|
|
16
|
+
let h = 0x811c9dc5;
|
|
17
|
+
for (let i = 0; i < str.length; i++) {
|
|
18
|
+
h ^= str.charCodeAt(i);
|
|
19
|
+
h = Math.imul(h, 0x01000193);
|
|
20
|
+
}
|
|
21
|
+
return (h >>> 0).toString(16).padStart(8, '0');
|
|
22
|
+
}
|
|
23
|
+
|
|
24
|
+
export function findingId(f ) {
|
|
25
|
+
// Location key mirrors the report's stable identity: stage, else SQL
|
|
26
|
+
// execution, else the audited config property.
|
|
27
|
+
const locKey = f.stageId ?? f.executionId ?? f.property ?? '';
|
|
28
|
+
// Discriminators for detectors that intentionally emit multiple findings on
|
|
29
|
+
// the same location+metric; without them these siblings hash to one id:
|
|
30
|
+
// stage-scope: slowHost (host), memoryUtilization (executorId + dimension),
|
|
31
|
+
// partitionSizing (rule);
|
|
32
|
+
// app-scope: cacheUtilization (rddId + variant, stageId is always null),
|
|
33
|
+
// cachingOpportunity (stageId/executionId both null too: `relation`/
|
|
34
|
+
// `format` distinguish leaf findings, `operator`+`relation` distinguish
|
|
35
|
+
// composite findings, `executionIds` as a last resort when everything
|
|
36
|
+
// else matches);
|
|
37
|
+
// SQL plan-advisor: smallFiles (direction/nodeName),
|
|
38
|
+
// duplicatePlanSubtree (rootName/subtreeSize/groupIndex: two unrelated
|
|
39
|
+
// duplicate-subtree groups can share rootName+subtreeSize, e.g. the same
|
|
40
|
+
// "BroadcastExchange over a Project/Filter/Scan" shape repeated once per
|
|
41
|
+
// dimension table; groupIndex (the group's deterministic position within
|
|
42
|
+
// one execution's plan) is the actual uniqueness guarantee, since the
|
|
43
|
+
// informational `sampleRelation` field can be null or coincide for two
|
|
44
|
+
// groups),
|
|
45
|
+
// broadcastSizing (per node/side, distinguished by value + largerSideBytes).
|
|
46
|
+
// The metric `value` folds in the per-node magnitude that broadcastSizing
|
|
47
|
+
// siblings carry no other field for; it is deterministic for a given log, so
|
|
48
|
+
// the id stays stable across re-parses (no timestamp/randomness).
|
|
49
|
+
// memoryUtilization's `rule` (heapNearCapacity/heapOverProvisioned) is excluded here: an
|
|
50
|
+
// executor falls in exactly one heap band, so `executorId` alone already guarantees
|
|
51
|
+
// uniqueness and folding `rule` in too would just add gratuitous id-rotation risk if the
|
|
52
|
+
// band logic ever changes. partitionSizing's `rule` (maxPartitionTooBig/
|
|
53
|
+
// shufflePartitionSkew/lowShuffleParallelism) IS load-bearing here: a stage can emit more
|
|
54
|
+
// than one of those rules at once, sharing the same stageId+metric.
|
|
55
|
+
const rule = f.type === 'memoryUtilization' ? undefined : f.rule;
|
|
56
|
+
const disc = [
|
|
57
|
+
f.host, f.executorId, rule, f.variant, f.dimension,
|
|
58
|
+
f.direction, f.nodeName, f.rootName, f.subtreeSize, f.groupIndex, f.largerSideBytes,
|
|
59
|
+
f.rddId, f.relation, f.format, f.operator,
|
|
60
|
+
f.executionIds ? f.executionIds.join(',') : '',
|
|
61
|
+
].map((v) => v ?? '').join('|');
|
|
62
|
+
return fnv1a(`${f.type}|${locKey}|${f.metric ?? ''}|${f.value ?? ''}|${disc}`);
|
|
63
|
+
}
|
|
64
|
+
|
|
65
|
+
function push(out , entry , result ) {
|
|
66
|
+
if (!result) return;
|
|
67
|
+
const detectorVersion = entry.version ?? 1;
|
|
68
|
+
for (const f of (Array.isArray(result) ? result : [result])) {
|
|
69
|
+
if (!f) continue;
|
|
70
|
+
if (entry.suppressWhen && entry.suppressWhen(f, out)) continue;
|
|
71
|
+
const stamped = { ...f, docAnchor: f.docAnchor ?? entry.docAnchor, detectorVersion };
|
|
72
|
+
const id = findingId(stamped);
|
|
73
|
+
// Guardrail against any detector (this one or a future one) emitting two
|
|
74
|
+
// structurally-identical findings for what should be one logical
|
|
75
|
+
// occurrence: same id => same finding, keep only the first. Added after a
|
|
76
|
+
// real-log bug where duplicatePlanSubtree emitted two distinct findings
|
|
77
|
+
// sharing one id (see findingId's groupIndex discriminator above and
|
|
78
|
+
// tests/detectors-plan.test.js's duplicatePlanSubtree regression tests);
|
|
79
|
+
// this guard's correctness depends on findingId's discriminator list
|
|
80
|
+
// actually being unique per distinct finding, not on this line itself.
|
|
81
|
+
if (out.some((existing) => existing.id === id)) continue;
|
|
82
|
+
out.push({ ...stamped, id });
|
|
83
|
+
}
|
|
84
|
+
}
|
|
85
|
+
|
|
86
|
+
// `app` is widened to `SparkAppInfo | null` rather than the plan's literal
|
|
87
|
+
// non-nullable `SparkAppInfo`: real callers (evidence-report.ts's
|
|
88
|
+
// `buildJson`, useIngest.ts's `runDone`/`snapshotParsedRun`) pass
|
|
89
|
+
// `AppModel.app`, which is genuinely `SparkAppInfo | null` at the type
|
|
90
|
+
// level even though by the time `analyze()` runs in practice a full parse
|
|
91
|
+
// has completed and `app` is always populated (the same reasoning
|
|
92
|
+
// detectors.ts's `DetectorCtx.app` comment gives for treating its own,
|
|
93
|
+
// separate `app` field as non-nullable). Widened to match the real caller
|
|
94
|
+
// type rather than forcing a cast at every call site.
|
|
95
|
+
export function analyze(
|
|
96
|
+
app ,
|
|
97
|
+
stages ,
|
|
98
|
+
executorsAdded ,
|
|
99
|
+
executorsRemoved ,
|
|
100
|
+
jobs ,
|
|
101
|
+
sql = new Map(),
|
|
102
|
+
runAggregates = null,
|
|
103
|
+
) {
|
|
104
|
+
// `app ?? {}`: every detector below tolerates a null `app` (malformed logs
|
|
105
|
+
// with no ApplicationStart), so this call must too; computePeakConcurrentCores
|
|
106
|
+
// just falls back to the executor-derived core sum when `resources` is absent.
|
|
107
|
+
// computePeakConcurrentCores (not computeTotalCores) here specifically: the
|
|
108
|
+
// ceiling this feeds (src/occupancy.ts's computeCeiling) needs a genuine
|
|
109
|
+
// concurrent-capacity bound, and computeTotalCores's cumulative sum over
|
|
110
|
+
// every addition overstates that under dynamic allocation/executor
|
|
111
|
+
// replacement (a churned-through executor's cores were never actually
|
|
112
|
+
// concurrent with its replacement's).
|
|
113
|
+
// `executorsAdded`/`executorsRemoved` casts: computePeakConcurrentCores only
|
|
114
|
+
// reads `executorId`/`timestamp`/`totalCores`, all present on
|
|
115
|
+
// ExecutorAddedEvent/ExecutorRemovedEvent, but `totalCores` isn't on
|
|
116
|
+
// ExecutorRemovedEvent, so the `ExecutorEvent` union as a whole is a
|
|
117
|
+
// structural mismatch against its parameter shapes.
|
|
118
|
+
const totalCores = computePeakConcurrentCores(
|
|
119
|
+
app ?? {},
|
|
120
|
+
executorsAdded ,
|
|
121
|
+
executorsRemoved ,
|
|
122
|
+
);
|
|
123
|
+
// Computed once, up front, so detectors can gate impact band on the same
|
|
124
|
+
// occupancy-clipped waste figure estimateImpact() below displays as that
|
|
125
|
+
// finding's savings (src/detectors.ts's `clippedWasteMs`), not a raw
|
|
126
|
+
// pre-clip delta the two passes would otherwise disagree on.
|
|
127
|
+
const occupancy = computeOccupancy(stages , totalCores);
|
|
128
|
+
const ctx = {
|
|
129
|
+
app, stages, executorsAdded, executorsRemoved, jobs, sql, runAggregates, occupancy,
|
|
130
|
+
};
|
|
131
|
+
const out = [];
|
|
132
|
+
for (const d of DETECTORS) {
|
|
133
|
+
if (d.inScorecard === false) continue;
|
|
134
|
+
switch (d.scope) {
|
|
135
|
+
case 'stage':
|
|
136
|
+
for (const s of stages.values()) push(out, d, d.detect(s, ctx));
|
|
137
|
+
break;
|
|
138
|
+
case 'sql':
|
|
139
|
+
for (const e of sql.values()) push(out, d, d.detect(e, ctx));
|
|
140
|
+
break;
|
|
141
|
+
case 'app':
|
|
142
|
+
case 'config':
|
|
143
|
+
push(out, d, d.detect(ctx));
|
|
144
|
+
break;
|
|
145
|
+
default:
|
|
146
|
+
assertNever(d.scope);
|
|
147
|
+
}
|
|
148
|
+
}
|
|
149
|
+
estimateImpact(out, stages, totalCores);
|
|
150
|
+
deriveImpactBand(out, app);
|
|
151
|
+
// Ascending IMPACT_BAND_ORDER (critical 0 → info 2) puts the worst band
|
|
152
|
+
// first, matching the old descending 3/2/1 rank this replaced; `sort` is
|
|
153
|
+
// stable, so findings sharing a band keep their DETECTORS declaration order.
|
|
154
|
+
out.sort((a, b) => IMPACT_BAND_ORDER[a.impactBand] - IMPACT_BAND_ORDER[b.impactBand]);
|
|
155
|
+
return out;
|
|
156
|
+
}
|
|
157
|
+
|
|
158
|
+
export function auditConfig(app ) {
|
|
159
|
+
const out = [];
|
|
160
|
+
for (const d of DETECTORS) if (d.scope === 'config') push(out, d, d.detect({ app }));
|
|
161
|
+
// configAudit's own impact-estimator case (src/impact-estimator.ts) needs neither
|
|
162
|
+
// `stages` nor `totalCores`: it's unconditionally `costOnly('none')`. An empty stages
|
|
163
|
+
// map is enough for parity with the same finding type produced via analyze().
|
|
164
|
+
estimateImpact(out, new Map());
|
|
165
|
+
deriveImpactBand(out, app);
|
|
166
|
+
return out.map((f) => ({ ...f, stageId: f.stageId ?? null }));
|
|
167
|
+
}
|
|
@@ -0,0 +1,203 @@
|
|
|
1
|
+
import { computeEfficiencyModel } from '../efficiency-model.js';
|
|
2
|
+
import { computeSkewRatio, DETECTORS } from '../detectors.js';
|
|
3
|
+
import { IMPACT_BAND_ORDER } from '../format-utils.js';
|
|
4
|
+
|
|
5
|
+
|
|
6
|
+
|
|
7
|
+
const IMPACT_BANDS = Object.keys(IMPACT_BAND_ORDER) ;
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
const skewDetector = DETECTORS.find((d) => d.type === 'skew');
|
|
22
|
+
const SKEW_MIN_TASKS_FOR_P95 = skewDetector .thresholds .minTasksForP95 ;
|
|
23
|
+
|
|
24
|
+
function taskDataTrusted(appModel ) {
|
|
25
|
+
const entry = appModel.evidenceAvailability?.entries?.find((e) => e.key === 'taskCoreTime');
|
|
26
|
+
return entry?.state === 'present';
|
|
27
|
+
}
|
|
28
|
+
|
|
29
|
+
// `Finding.value` is `number | string` (some detectors, e.g. stageFailed/
|
|
30
|
+
// configAudit, put human-readable text there instead of a magnitude; see
|
|
31
|
+
// types.ts). The 'spill' findings this is used for are always numeric; the
|
|
32
|
+
// typeof guard below narrows without changing behavior for real input.
|
|
33
|
+
function maxFindingValue(catalog , type ) {
|
|
34
|
+
const values = catalog
|
|
35
|
+
.filter((f) => f.type === type)
|
|
36
|
+
.map((f) => f.value ?? 0)
|
|
37
|
+
.filter((v) => typeof v === 'number');
|
|
38
|
+
return values.length > 0 ? Math.max(...values) : null;
|
|
39
|
+
}
|
|
40
|
+
|
|
41
|
+
function checkRuntime(appModel , maxRuntimeMs ) {
|
|
42
|
+
const { startTime, endTime } = appModel.app ?? {};
|
|
43
|
+
if (startTime == null || endTime == null) {
|
|
44
|
+
return { name: 'max-runtime', status: 'inconclusive', detail: 'App start/end time not observed (run may not have finished).' };
|
|
45
|
+
}
|
|
46
|
+
const runtimeMs = endTime - startTime;
|
|
47
|
+
return runtimeMs > maxRuntimeMs
|
|
48
|
+
? { name: 'max-runtime', status: 'violation', detail: `Runtime ${runtimeMs}ms exceeds budget ${maxRuntimeMs}ms.` }
|
|
49
|
+
: { name: 'max-runtime', status: 'pass', detail: `Runtime ${runtimeMs}ms within budget ${maxRuntimeMs}ms.` };
|
|
50
|
+
}
|
|
51
|
+
|
|
52
|
+
function checkSpill(appModel , catalog , maxSpillGb ) {
|
|
53
|
+
if (!taskDataTrusted(appModel)) {
|
|
54
|
+
return { name: 'max-spill', status: 'inconclusive', detail: 'No trustworthy task-level evidence to measure spill.' };
|
|
55
|
+
}
|
|
56
|
+
const maxBytes = maxFindingValue(catalog, 'spill') ?? 0;
|
|
57
|
+
const budgetBytes = maxSpillGb * 1024 ** 3;
|
|
58
|
+
return maxBytes > budgetBytes
|
|
59
|
+
? { name: 'max-spill', status: 'violation', detail: `Peak stage spill ${maxBytes} bytes exceeds budget ${budgetBytes} bytes.` }
|
|
60
|
+
: { name: 'max-spill', status: 'pass', detail: `Peak stage spill ${maxBytes} bytes within budget ${budgetBytes} bytes.` };
|
|
61
|
+
}
|
|
62
|
+
|
|
63
|
+
function checkSkew(appModel , maxSkewRatio ) {
|
|
64
|
+
if (!taskDataTrusted(appModel)) {
|
|
65
|
+
return { name: 'max-skew', status: 'inconclusive', detail: 'No trustworthy task-level evidence to measure skew.' };
|
|
66
|
+
}
|
|
67
|
+
const stages = [...(appModel.stages?.values() ?? [])];
|
|
68
|
+
if (stages.length === 0) {
|
|
69
|
+
return { name: 'max-skew', status: 'inconclusive', detail: 'No stage data observed in this event log.' };
|
|
70
|
+
}
|
|
71
|
+
// Recompute the true ratio per stage (not from the finding catalog): the
|
|
72
|
+
// skew detector floors its findings at thresholds.ratioWarn (3), so a
|
|
73
|
+
// budget stricter than that floor could never be enforced by reading
|
|
74
|
+
// catalog findings alone.
|
|
75
|
+
const ratios = stages
|
|
76
|
+
.map((stage) => computeSkewRatio(stage, SKEW_MIN_TASKS_FOR_P95))
|
|
77
|
+
.filter((r) => r !== null)
|
|
78
|
+
.map((r) => r.ratio);
|
|
79
|
+
if (ratios.length === 0) {
|
|
80
|
+
return { name: 'max-skew', status: 'inconclusive', detail: 'No stage has a measurable task-duration median.' };
|
|
81
|
+
}
|
|
82
|
+
const maxRatio = Math.max(...ratios);
|
|
83
|
+
return maxRatio > maxSkewRatio
|
|
84
|
+
? { name: 'max-skew', status: 'violation', detail: `Peak stage skew ratio ${maxRatio} exceeds budget ${maxSkewRatio}.` }
|
|
85
|
+
: { name: 'max-skew', status: 'pass', detail: `Peak stage skew ratio ${maxRatio} within budget ${maxSkewRatio}.` };
|
|
86
|
+
}
|
|
87
|
+
|
|
88
|
+
function checkFailedTaskRate(appModel , catalog , maxPct ) {
|
|
89
|
+
const finding = catalog.find((f) => f.type === 'jobFailureRate');
|
|
90
|
+
if (finding) {
|
|
91
|
+
const rate = (finding.taskFailureRate ) ?? 0;
|
|
92
|
+
return rate > maxPct
|
|
93
|
+
? { name: 'max-failed-task-rate', status: 'violation', detail: `Task failure rate ${rate}% exceeds budget ${maxPct}%.` }
|
|
94
|
+
: { name: 'max-failed-task-rate', status: 'pass', detail: `Task failure rate ${rate}% within budget ${maxPct}%.` };
|
|
95
|
+
}
|
|
96
|
+
if ((appModel.jobs?.size ?? 0) === 0) {
|
|
97
|
+
return { name: 'max-failed-task-rate', status: 'inconclusive', detail: 'No job data observed in this event log.' };
|
|
98
|
+
}
|
|
99
|
+
// Jobs completed but the jobFailureRate detector never fired: job failure
|
|
100
|
+
// rate is below its own 10% info floor (src/detectors.js), so the task
|
|
101
|
+
// failure rate is implicitly low too. Known v1 limitation: a budget
|
|
102
|
+
// stricter than that floor cannot be enforced.
|
|
103
|
+
return { name: 'max-failed-task-rate', status: 'pass', detail: `No job-failure-rate finding: task failure rate is below the detector's reporting floor.` };
|
|
104
|
+
}
|
|
105
|
+
|
|
106
|
+
function checkEfficiency(appModel , minPct ) {
|
|
107
|
+
if (!taskDataTrusted(appModel)) {
|
|
108
|
+
return { name: 'min-efficiency', status: 'inconclusive', detail: 'No trustworthy task-level evidence to measure efficiency.' };
|
|
109
|
+
}
|
|
110
|
+
const model = computeEfficiencyModel({
|
|
111
|
+
app: appModel.app, stages: appModel.stages,
|
|
112
|
+
executorsAdded: appModel.executors.added, runAggregates: appModel.runAggregates,
|
|
113
|
+
});
|
|
114
|
+
if (model.wastagePct == null) {
|
|
115
|
+
return { name: 'min-efficiency', status: 'inconclusive', detail: 'Efficiency could not be computed (no available compute hours).' };
|
|
116
|
+
}
|
|
117
|
+
const efficiencyPct = 100 - model.wastagePct;
|
|
118
|
+
return efficiencyPct < minPct
|
|
119
|
+
? { name: 'min-efficiency', status: 'violation', detail: `Efficiency ${efficiencyPct}% below budget ${minPct}%.` }
|
|
120
|
+
: { name: 'min-efficiency', status: 'pass', detail: `Efficiency ${efficiencyPct}% meets budget ${minPct}%.` };
|
|
121
|
+
}
|
|
122
|
+
|
|
123
|
+
function checkRegression(comparison , maxRegressionPct , regressionMetric ) {
|
|
124
|
+
const row = comparison.metrics.find((m) => m.key === regressionMetric);
|
|
125
|
+
if (!row || row.direction === 'unavailable' || row.baseline == null || row.delta == null) {
|
|
126
|
+
return { name: 'max-regression', status: 'inconclusive', detail: `Metric "${regressionMetric}" is unavailable for this comparison.` };
|
|
127
|
+
}
|
|
128
|
+
// A neutral-direction metric (inputBytes, outputBytes, taskCount,
|
|
129
|
+
// executorsAdded — see NEUTRAL_METRIC_KEYS in run-comparison.ts) measures
|
|
130
|
+
// workload volume, not performance: an increase isn't a regression, so a
|
|
131
|
+
// regression budget can't be meaningfully evaluated against it either way.
|
|
132
|
+
if (row.direction === 'neutral') {
|
|
133
|
+
return { name: 'max-regression', status: 'inconclusive', detail: `Metric "${regressionMetric}" measures workload volume, not performance: it has no regression direction to check.` };
|
|
134
|
+
}
|
|
135
|
+
if (row.direction !== 'regression') {
|
|
136
|
+
return { name: 'max-regression', status: 'pass', detail: `Metric "${regressionMetric}" did not regress (${row.direction}).` };
|
|
137
|
+
}
|
|
138
|
+
const pct = row.baseline === 0 ? Infinity : Math.abs(row.delta / row.baseline) * 100;
|
|
139
|
+
const pctLabel = row.baseline === 0
|
|
140
|
+
? `regressed from 0 to ${row.delta} (was absent/zero in baseline)`
|
|
141
|
+
: `regressed ${pct.toFixed(1)}%`;
|
|
142
|
+
return pct > maxRegressionPct
|
|
143
|
+
? { name: 'max-regression', status: 'violation', detail: `Metric "${regressionMetric}" ${pctLabel}, exceeding budget ${maxRegressionPct}%.` }
|
|
144
|
+
: { name: 'max-regression', status: 'pass', detail: `Metric "${regressionMetric}" ${pctLabel}, within budget ${maxRegressionPct}%.` };
|
|
145
|
+
}
|
|
146
|
+
|
|
147
|
+
function checkFailOnIntroduced(comparison , band ) {
|
|
148
|
+
if (band !== 'all' && !IMPACT_BANDS.includes(band )) {
|
|
149
|
+
return { name: 'fail-on-introduced', status: 'inconclusive', detail: `Impact band "${band}" is not recognized (expected "all" or one of ${IMPACT_BANDS.join(', ')}).` };
|
|
150
|
+
}
|
|
151
|
+
const matches = band === 'all'
|
|
152
|
+
? comparison.findings.introduced
|
|
153
|
+
: comparison.findings.introduced.filter((f) => f.impactBand === band);
|
|
154
|
+
return matches.length > 0
|
|
155
|
+
? { name: 'fail-on-introduced', status: 'violation', detail: `${matches.length} introduced finding(s) match "${band}".` }
|
|
156
|
+
: { name: 'fail-on-introduced', status: 'pass', detail: `No introduced findings match "${band}".` };
|
|
157
|
+
}
|
|
158
|
+
|
|
159
|
+
// Shared by both comparison-dependent budgets below: each needs the same
|
|
160
|
+
// "no comparison yet -> inconclusive" fallback instead of actually checking.
|
|
161
|
+
function pushComparisonBudget(
|
|
162
|
+
results ,
|
|
163
|
+
comparison ,
|
|
164
|
+
name ,
|
|
165
|
+
check ,
|
|
166
|
+
) {
|
|
167
|
+
results.push(comparison ? check(comparison) : { name, status: 'inconclusive', detail: 'No baseline comparison available to evaluate this budget.' });
|
|
168
|
+
}
|
|
169
|
+
|
|
170
|
+
export function evaluateBudgets({ appModel, catalog, budgets, comparison }
|
|
171
|
+
|
|
172
|
+
) {
|
|
173
|
+
const results = [];
|
|
174
|
+
if (Number.isFinite(budgets.maxRuntimeMs)) results.push(checkRuntime(appModel, budgets.maxRuntimeMs ));
|
|
175
|
+
if (Number.isFinite(budgets.maxSpillGb)) results.push(checkSpill(appModel, catalog, budgets.maxSpillGb ));
|
|
176
|
+
if (Number.isFinite(budgets.maxSkewRatio)) results.push(checkSkew(appModel, budgets.maxSkewRatio ));
|
|
177
|
+
if (Number.isFinite(budgets.maxFailedTaskRatePct)) results.push(checkFailedTaskRate(appModel, catalog, budgets.maxFailedTaskRatePct ));
|
|
178
|
+
if (Number.isFinite(budgets.minEfficiencyPct)) results.push(checkEfficiency(appModel, budgets.minEfficiencyPct ));
|
|
179
|
+
// Guarded here (not just at the CLI/mcp-tools call sites) so any caller of
|
|
180
|
+
// evaluateBudgets() gets this for free: regressionMetric with no
|
|
181
|
+
// maxRegressionPct would otherwise skip the whole `if` below silently,
|
|
182
|
+
// reporting nothing at all instead of a visible inconclusive result.
|
|
183
|
+
if (budgets.regressionMetric !== undefined && budgets.maxRegressionPct === undefined) {
|
|
184
|
+
results.push({ name: 'max-regression', status: 'inconclusive', detail: `regressionMetric "${budgets.regressionMetric}" was set without maxRegressionPct; the regression budget was not evaluated.` });
|
|
185
|
+
} else if (budgets.maxRegressionPct !== undefined) {
|
|
186
|
+
// `!== undefined`, not `Number.isFinite`: a zero-baseline regression's pct
|
|
187
|
+
// is itself `Infinity` (see checkRegression), so an "unlimited" budget is a
|
|
188
|
+
// legitimate finite-typed-as-number input here (CLI's own flag validation
|
|
189
|
+
// already rejects non-finite --max-regression-pct input, so this only
|
|
190
|
+
// widens what a direct evaluateBudgets caller, e.g. a test, can express).
|
|
191
|
+
pushComparisonBudget(results, comparison, 'max-regression',
|
|
192
|
+
(c) => checkRegression(c, budgets.maxRegressionPct , budgets.regressionMetric ?? 'wallClock'));
|
|
193
|
+
}
|
|
194
|
+
if (budgets.failOnIntroduced !== undefined) {
|
|
195
|
+
pushComparisonBudget(results, comparison, 'fail-on-introduced',
|
|
196
|
+
(c) => checkFailOnIntroduced(c, budgets.failOnIntroduced ));
|
|
197
|
+
}
|
|
198
|
+
return {
|
|
199
|
+
results,
|
|
200
|
+
violated: results.some((r) => r.status === 'violation'),
|
|
201
|
+
inconclusive: results.some((r) => r.status === 'inconclusive'),
|
|
202
|
+
};
|
|
203
|
+
}
|
|
@@ -0,0 +1,107 @@
|
|
|
1
|
+
import { readFileSync, readdirSync, statSync } from 'node:fs';
|
|
2
|
+
import { join, basename } from 'node:path';
|
|
3
|
+
import { createState, runParse, runParseFiles, reassembleRollingEntries } from '../parser-worker.js';
|
|
4
|
+
import { createModelCallbacks } from '../model-assembler.js';
|
|
5
|
+
import { routeMessage, } from '../ingest.js';
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
export function emptyAppModel() {
|
|
15
|
+
return {
|
|
16
|
+
app: null,
|
|
17
|
+
stages: new Map(),
|
|
18
|
+
executors: { added: [], removed: [] },
|
|
19
|
+
sql: new Map(),
|
|
20
|
+
jobs: new Map(),
|
|
21
|
+
runAggregates: null,
|
|
22
|
+
evidenceAvailability: null,
|
|
23
|
+
};
|
|
24
|
+
}
|
|
25
|
+
|
|
26
|
+
// Mirrors tests/parser-worker.test.js's fakeFile: the proven File-like shape
|
|
27
|
+
// runParse/runParseFiles need (name, size, slice().arrayBuffer(), arrayBuffer()).
|
|
28
|
+
export function nodeFileFromPath(path ) {
|
|
29
|
+
const bytes = readFileSync(path);
|
|
30
|
+
const u8 = new Uint8Array(bytes.buffer, bytes.byteOffset, bytes.byteLength);
|
|
31
|
+
return {
|
|
32
|
+
name: basename(path),
|
|
33
|
+
size: u8.length,
|
|
34
|
+
slice(start , end ) {
|
|
35
|
+
const view = u8.subarray(start, end);
|
|
36
|
+
return { async arrayBuffer() { return view.slice().buffer; } };
|
|
37
|
+
},
|
|
38
|
+
async arrayBuffer() { return u8.slice().buffer; },
|
|
39
|
+
};
|
|
40
|
+
}
|
|
41
|
+
|
|
42
|
+
// Delegates to src/ingest.js's routeMessage: the single source of truth for
|
|
43
|
+
// worker-message-type -> handler-callback wiring, shared by the browser
|
|
44
|
+
// Worker context and this synchronous-in-Node CLI path. No pendingTaskRequests
|
|
45
|
+
// map: the CLI never sends a 'getTaskData' request, so it never receives a
|
|
46
|
+
// 'taskData' message back.
|
|
47
|
+
export function dispatch(msg , handlers ) {
|
|
48
|
+
routeMessage(msg , handlers);
|
|
49
|
+
}
|
|
50
|
+
|
|
51
|
+
function isRollingLogDirectory(dirPath ) {
|
|
52
|
+
const names = readdirSync(dirPath);
|
|
53
|
+
return names.some((n ) => /^events_\d+_/.test(n));
|
|
54
|
+
}
|
|
55
|
+
|
|
56
|
+
// Shared ingest scaffold for both consumers of routeMessage's onDone/onError
|
|
57
|
+
// dispatch pattern (this file's collectRun, over a local file/dir, and
|
|
58
|
+
// shs-load.ts's collectShsAppModel, over fetched archive bytes): builds a
|
|
59
|
+
// fresh AppModel wired to the shared handlers, then hands `run` a
|
|
60
|
+
// (state, emit, reject) triple so each caller only supplies its own decode
|
|
61
|
+
// step (and its own error type via `onDecodeError` — a plain Error here, an
|
|
62
|
+
// mcpError over in shs-load.ts).
|
|
63
|
+
export function collectViaDispatch(
|
|
64
|
+
run ,
|
|
65
|
+
onDecodeError ,
|
|
66
|
+
) {
|
|
67
|
+
const appModel = emptyAppModel();
|
|
68
|
+
const cb = createModelCallbacks(appModel, { onProgress() {}, onDone() {}, onError() {} });
|
|
69
|
+
|
|
70
|
+
return new Promise((resolve, reject) => {
|
|
71
|
+
const handlers = {
|
|
72
|
+
...cb,
|
|
73
|
+
onDone: (msg ) => resolve({ appModel, skippedLines: (msg )?.skippedLines ?? 0 }),
|
|
74
|
+
onError: (msg ) => reject(onDecodeError(msg)),
|
|
75
|
+
};
|
|
76
|
+
const emit = (msg ) => dispatch(msg, handlers);
|
|
77
|
+
const state = createState();
|
|
78
|
+
run(state, emit, reject);
|
|
79
|
+
});
|
|
80
|
+
}
|
|
81
|
+
|
|
82
|
+
export async function collectRun(inputPath ) {
|
|
83
|
+
const stat = statSync(inputPath);
|
|
84
|
+
return collectViaDispatch((state, emit, reject) => {
|
|
85
|
+
if (stat.isDirectory()) {
|
|
86
|
+
if (!isRollingLogDirectory(inputPath)) {
|
|
87
|
+
reject(new Error("This isn't a Spark rolling event-log directory. Pass a single event-log file instead."));
|
|
88
|
+
return;
|
|
89
|
+
}
|
|
90
|
+
const names = readdirSync(inputPath);
|
|
91
|
+
let ordered;
|
|
92
|
+
try {
|
|
93
|
+
ordered = reassembleRollingEntries(names);
|
|
94
|
+
} catch (e) {
|
|
95
|
+
reject(e);
|
|
96
|
+
return;
|
|
97
|
+
}
|
|
98
|
+
const files = ordered.map((name) => nodeFileFromPath(join(inputPath, name)));
|
|
99
|
+
// .catch(reject), not void: an unexpected throw past the parser's own
|
|
100
|
+
// guards (e.g. in decoder.flush) would otherwise leave this Promise
|
|
101
|
+
// pending forever and surface only as an unhandled rejection.
|
|
102
|
+
runParseFiles(files, state, { emit }).catch(reject);
|
|
103
|
+
} else {
|
|
104
|
+
runParse(nodeFileFromPath(inputPath), state, { emit }).catch(reject);
|
|
105
|
+
}
|
|
106
|
+
}, (msg) => new Error((msg ).message));
|
|
107
|
+
}
|
|
@@ -0,0 +1,68 @@
|
|
|
1
|
+
// Total cores computation: sum real executor totalCores, or fallback to
|
|
2
|
+
// peakExecutors × configured cores. Used by efficiency model, scaling simulator,
|
|
3
|
+
// wasted core-hours calculation, and utilization/memory-utilization detectors.
|
|
4
|
+
export function computeTotalCores(app , executorsAdded ) {
|
|
5
|
+
let total = executorsAdded.reduce((s, e) => s + (e.totalCores ?? 0), 0);
|
|
6
|
+
if (total <= 0) {
|
|
7
|
+
const cores = app.resources?.executor?.cores ?? null;
|
|
8
|
+
total = cores != null ? executorsAdded.length * cores : 0;
|
|
9
|
+
}
|
|
10
|
+
return total;
|
|
11
|
+
}
|
|
12
|
+
|
|
13
|
+
// Peak concurrently-alive core count: sweeps add/remove events by timestamp
|
|
14
|
+
// (add contributes +totalCores at its own executorId, remove contributes
|
|
15
|
+
// -totalCores looked up from that same executorId) and tracks the running
|
|
16
|
+
// total's max, rather than summing every addition regardless of overlap.
|
|
17
|
+
// computeTotalCores's cumulative sum overstates capacity under dynamic
|
|
18
|
+
// allocation/executor replacement, since a churned-through executor's cores
|
|
19
|
+
// are never actually concurrent with its replacement's; this is the
|
|
20
|
+
// concurrency-aware sibling used to bound the impact estimator's ceiling
|
|
21
|
+
// (src/occupancy.ts's computeCeiling) against that inflation.
|
|
22
|
+
// Tie-break same-timestamp events by delta ascending, so a removal is applied
|
|
23
|
+
// before a same-instant replacement's addition: without this, a seamless swap
|
|
24
|
+
// (old executor gone exactly when its replacement joins) would momentarily
|
|
25
|
+
// double-count both as concurrent.
|
|
26
|
+
function sweepPeak(events ) {
|
|
27
|
+
const sorted = [...events].sort((a, b) => a.time - b.time || a.delta - b.delta);
|
|
28
|
+
let running = 0;
|
|
29
|
+
let peak = 0;
|
|
30
|
+
for (const ev of sorted) {
|
|
31
|
+
running += ev.delta;
|
|
32
|
+
if (running > peak) peak = running;
|
|
33
|
+
}
|
|
34
|
+
return peak;
|
|
35
|
+
}
|
|
36
|
+
|
|
37
|
+
export function computePeakConcurrentCores(
|
|
38
|
+
app ,
|
|
39
|
+
executorsAdded ,
|
|
40
|
+
executorsRemoved ,
|
|
41
|
+
) {
|
|
42
|
+
const coresByExecutor = new Map ();
|
|
43
|
+
const coreEvents = [];
|
|
44
|
+
const countEvents = [];
|
|
45
|
+
for (const e of executorsAdded) {
|
|
46
|
+
const cores = e.totalCores ?? 0;
|
|
47
|
+
coresByExecutor.set(e.executorId, cores);
|
|
48
|
+
coreEvents.push({ time: e.timestamp, delta: cores });
|
|
49
|
+
countEvents.push({ time: e.timestamp, delta: 1 });
|
|
50
|
+
}
|
|
51
|
+
for (const e of executorsRemoved) {
|
|
52
|
+
const cores = coresByExecutor.get(e.executorId) ?? 0;
|
|
53
|
+
coreEvents.push({ time: e.timestamp, delta: -cores });
|
|
54
|
+
countEvents.push({ time: e.timestamp, delta: -1 });
|
|
55
|
+
}
|
|
56
|
+
const peak = sweepPeak(coreEvents);
|
|
57
|
+
if (peak > 0) return peak;
|
|
58
|
+
// Real totalCores data is missing/zero for every add event, so the cores
|
|
59
|
+
// sweep above can't tell us anything. Falling back to
|
|
60
|
+
// `executorsAdded.length * cores` here would reintroduce the exact
|
|
61
|
+
// cumulative-historical-additions overcount this function exists to avoid
|
|
62
|
+
// (a churned-through executor and its replacement both counted, even
|
|
63
|
+
// though they were never alive at once). Sweep peak *executor count*
|
|
64
|
+
// instead: still concurrency-aware, just cores-blind.
|
|
65
|
+
const peakExecutorCount = sweepPeak(countEvents);
|
|
66
|
+
const cores = app.resources?.executor?.cores ?? null;
|
|
67
|
+
return cores != null ? peakExecutorCount * cores : 0;
|
|
68
|
+
}
|