sparkforensics-mcp 0.3.0 → 0.5.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/bin/sparkforensics-mcp.mjs +41 -10
- package/package.json +1 -1
- package/vendor-core/allocation.js +106 -0
- package/vendor-core/analyzer.js +168 -60
- package/vendor-core/check-coverage.js +88 -0
- package/vendor-core/cli/budgets.js +54 -27
- package/vendor-core/cli/collect-run.js +84 -32
- package/vendor-core/cli/regression-budgets.js +83 -0
- package/vendor-core/cli/threshold-config.js +28 -0
- package/vendor-core/comparison-verdict.js +177 -0
- package/vendor-core/core-source-hash.txt +1 -0
- package/vendor-core/core-usage-locality.js +56 -2
- package/vendor-core/detector-docs.js +58 -0
- package/vendor-core/detectors.js +1094 -500
- package/vendor-core/docs-config.js +0 -36
- package/vendor-core/docs-content/chapters/nav-index.json +31 -0
- package/vendor-core/docs-content/detection/cache.md +3 -2
- package/vendor-core/docs-content/detection/cfg.md +9 -8
- package/vendor-core/docs-content/detection/chrn.md +1 -2
- package/vendor-core/docs-content/detection/cold.md +4 -2
- package/vendor-core/docs-content/detection/cstor.md +9 -0
- package/vendor-core/docs-content/detection/fail.md +3 -2
- package/vendor-core/docs-content/detection/gc.md +3 -2
- package/vendor-core/docs-content/detection/host.md +2 -1
- package/vendor-core/docs-content/detection/local.md +1 -1
- package/vendor-core/docs-content/detection/mem.md +5 -2
- package/vendor-core/docs-content/detection/plan.md +2 -1
- package/vendor-core/docs-content/detection/sfail.md +2 -1
- package/vendor-core/docs-content/detection/shape.md +5 -4
- package/vendor-core/docs-content/detection/skew.md +3 -1
- package/vendor-core/docs-content/detection/slow.md +2 -2
- package/vendor-core/docs-content/detection/spec.md +2 -3
- package/vendor-core/docs-content/detection/spill.md +1 -1
- package/vendor-core/docs-site-config.js +3 -0
- package/vendor-core/effective-conf.js +107 -0
- package/vendor-core/efficiency-model.js +8 -6
- package/vendor-core/event-handlers.js +321 -44
- package/vendor-core/event-schemas.js +23 -0
- package/vendor-core/evidence-report.js +432 -115
- package/vendor-core/export-data.js +79 -6
- package/vendor-core/finding-action-label.js +9 -88
- package/vendor-core/finding-filter-predicate.js +9 -0
- package/vendor-core/finding-generic-recommendation.js +26 -105
- package/vendor-core/finding-names.js +28 -45
- package/vendor-core/finding-presentation.js +368 -0
- package/vendor-core/finding-tag-help.js +110 -0
- package/vendor-core/finding-types.js +373 -0
- package/vendor-core/findings-of-type.js +11 -0
- package/vendor-core/format-utils.js +96 -30
- package/vendor-core/html-export.js +51 -0
- package/vendor-core/impact-band.js +21 -8
- package/vendor-core/impact-estimator.js +25 -520
- package/vendor-core/impact-format.js +115 -0
- package/vendor-core/impact-model.js +197 -0
- package/vendor-core/ingest.js +6 -2
- package/vendor-core/intervals.js +13 -0
- package/vendor-core/list-runs.js +7 -5
- package/vendor-core/load-vendored.js +70 -5
- package/vendor-core/mcp-server-factory.js +14 -10
- package/vendor-core/mcp-tools.js +105 -45
- package/vendor-core/model-assembler.js +35 -1
- package/vendor-core/occupancy.js +1 -1
- package/vendor-core/parser-worker.js +2 -2
- package/vendor-core/plan-graph-model.js +3 -2
- package/vendor-core/plan-node-detail.js +1 -1
- package/vendor-core/proxy.js +3 -1
- package/vendor-core/python-stage.js +25 -0
- package/vendor-core/recommendation-rollup.js +70 -3
- package/vendor-core/redact.js +96 -37
- package/vendor-core/remediation.js +20 -0
- package/vendor-core/run-comparison.js +73 -29
- package/vendor-core/run-interpretation.js +291 -0
- package/vendor-core/run-metrics.js +198 -0
- package/vendor-core/run-outcome.js +74 -0
- package/vendor-core/run-payload.js +17 -0
- package/vendor-core/run-shape.js +40 -0
- package/vendor-core/run-totals.js +24 -0
- package/vendor-core/run-verdict.js +352 -0
- package/vendor-core/scaling-sim.js +4 -5
- package/vendor-core/scorecard-estimates.js +63 -0
- package/vendor-core/session-snapshot.js +7 -0
- package/vendor-core/shs-schemas.js +2 -2
- package/vendor-core/spark-memory.js +17 -0
- package/vendor-core/sql-stages.js +11 -0
- package/vendor-core/stage-plan-nodes.js +18 -0
- package/vendor-core/stage-quantiles.js +6 -0
- package/vendor-core/threshold-overrides.js +160 -0
- package/vendor-core/threshold-summary.js +11 -33
- package/vendor-core/types.js +54 -42
- package/vendor-core/wall-clock.js +1 -12
- package/vendor-core/wasted-core-hours.js +12 -9
- package/vendor-core/write-targets.js +312 -0
|
@@ -1,8 +1,10 @@
|
|
|
1
1
|
import { computeEfficiencyModel } from '../efficiency-model.js';
|
|
2
|
-
import { computeSkewRatio,
|
|
2
|
+
import { computeSkewRatio, ENTRY_BY_TYPE, } from '../detectors.js';
|
|
3
|
+
import { effectiveThresholds } from '../threshold-overrides.js';
|
|
3
4
|
import { IMPACT_BAND_ORDER } from '../format-utils.js';
|
|
4
5
|
|
|
5
6
|
|
|
7
|
+
|
|
6
8
|
|
|
7
9
|
const IMPACT_BANDS = Object.keys(IMPACT_BAND_ORDER) ;
|
|
8
10
|
|
|
@@ -10,29 +12,33 @@ const IMPACT_BANDS = Object.keys(IMPACT_BAND_ORDER) ;
|
|
|
10
12
|
|
|
11
13
|
|
|
12
14
|
|
|
15
|
+
|
|
16
|
+
|
|
13
17
|
|
|
14
18
|
|
|
15
19
|
|
|
16
|
-
|
|
20
|
+
|
|
17
21
|
|
|
18
22
|
|
|
23
|
+
|
|
24
|
+
|
|
19
25
|
|
|
20
26
|
|
|
21
|
-
|
|
22
|
-
|
|
27
|
+
// The skew entry's minTasksForP95 under the run's overrides, so the budget measures the same
|
|
28
|
+
// ratio (P95/median or max/median) the skew finding reports.
|
|
29
|
+
function skewMinTasksForP95(thresholds ) {
|
|
30
|
+
return effectiveThresholds(ENTRY_BY_TYPE.get('skew') , thresholds).minTasksForP95 ;
|
|
31
|
+
}
|
|
23
32
|
|
|
24
33
|
function taskDataTrusted(appModel ) {
|
|
25
34
|
const entry = appModel.evidenceAvailability?.entries?.find((e) => e.key === 'taskCoreTime');
|
|
26
35
|
return entry?.state === 'present';
|
|
27
36
|
}
|
|
28
37
|
|
|
29
|
-
// Finding.value is number|string (some detectors put text there); spill findings are
|
|
30
|
-
// always numeric, so the typeof guard narrows without changing behavior for real input.
|
|
31
38
|
function maxFindingValue(catalog , type ) {
|
|
32
39
|
const values = catalog
|
|
33
40
|
.filter((f) => f.type === type)
|
|
34
|
-
.map((f) => f.value ?? 0)
|
|
35
|
-
.filter((v) => typeof v === 'number');
|
|
41
|
+
.map((f) => f.value ?? 0);
|
|
36
42
|
return values.length > 0 ? Math.max(...values) : null;
|
|
37
43
|
}
|
|
38
44
|
|
|
@@ -58,7 +64,7 @@ function checkSpill(appModel , catalog , maxSpillGb )
|
|
|
58
64
|
: { name: 'max-spill', status: 'pass', detail: `Peak stage spill ${maxBytes} bytes within budget ${budgetBytes} bytes.` };
|
|
59
65
|
}
|
|
60
66
|
|
|
61
|
-
function checkSkew(appModel , maxSkewRatio ) {
|
|
67
|
+
function checkSkew(appModel , maxSkewRatio , minTasksForP95 ) {
|
|
62
68
|
if (!taskDataTrusted(appModel)) {
|
|
63
69
|
return { name: 'max-skew', status: 'inconclusive', detail: 'No trustworthy task-level evidence to measure skew.' };
|
|
64
70
|
}
|
|
@@ -69,7 +75,7 @@ function checkSkew(appModel , maxSkewRatio ) {
|
|
|
69
75
|
// Recompute the true ratio per stage: the skew detector floors findings at
|
|
70
76
|
// thresholds.ratioWarn (3), so a stricter budget can't be enforced from catalog alone.
|
|
71
77
|
const ratios = stages
|
|
72
|
-
.map((stage) => computeSkewRatio(stage,
|
|
78
|
+
.map((stage) => computeSkewRatio(stage, minTasksForP95))
|
|
73
79
|
.filter((r) => r !== null)
|
|
74
80
|
.map((r) => r.ratio);
|
|
75
81
|
if (ratios.length === 0) {
|
|
@@ -97,43 +103,47 @@ function checkFailedTaskRate(appModel , catalog , maxPct
|
|
|
97
103
|
return { name: 'max-failed-task-rate', status: 'pass', detail: `No job-failure-rate finding: task failure rate is below the detector's reporting floor.` };
|
|
98
104
|
}
|
|
99
105
|
|
|
106
|
+
// Busy core time: the share of available executor core time that ran tasks, 100 minus the
|
|
107
|
+
// dashboard's "Unused core time". Not the dashboard's Efficiency tile (the share of wall-clock with
|
|
108
|
+
// a stage running), so the detail never calls it "Efficiency".
|
|
100
109
|
function checkEfficiency(appModel , minPct ) {
|
|
101
110
|
if (!taskDataTrusted(appModel)) {
|
|
102
|
-
return { name: 'min-efficiency', status: 'inconclusive', detail: 'No trustworthy task-level evidence to measure
|
|
111
|
+
return { name: 'min-efficiency', status: 'inconclusive', detail: 'No trustworthy task-level evidence to measure busy core time.' };
|
|
103
112
|
}
|
|
104
113
|
const model = computeEfficiencyModel({
|
|
105
114
|
app: appModel.app, stages: appModel.stages,
|
|
106
|
-
executorsAdded: appModel.executors.added,
|
|
115
|
+
executorsAdded: appModel.executors.added, executorsRemoved: appModel.executors.removed,
|
|
116
|
+
runAggregates: appModel.runAggregates,
|
|
107
117
|
});
|
|
108
118
|
if (model.wastagePct == null) {
|
|
109
|
-
return { name: 'min-efficiency', status: 'inconclusive', detail: '
|
|
119
|
+
return { name: 'min-efficiency', status: 'inconclusive', detail: 'Busy core time could not be computed (no available compute hours).' };
|
|
110
120
|
}
|
|
111
|
-
const
|
|
112
|
-
return
|
|
113
|
-
? { name: 'min-efficiency', status: 'violation', detail: `
|
|
114
|
-
: { name: 'min-efficiency', status: 'pass', detail: `
|
|
121
|
+
const busyCorePct = 100 - model.wastagePct;
|
|
122
|
+
return busyCorePct < minPct
|
|
123
|
+
? { name: 'min-efficiency', status: 'violation', detail: `Busy core time ${busyCorePct}% below budget ${minPct}%.` }
|
|
124
|
+
: { name: 'min-efficiency', status: 'pass', detail: `Busy core time ${busyCorePct}% meets budget ${minPct}%.` };
|
|
115
125
|
}
|
|
116
126
|
|
|
117
127
|
function checkRegression(comparison , maxRegressionPct , regressionMetric ) {
|
|
118
128
|
const row = comparison.metrics.find((m) => m.key === regressionMetric);
|
|
119
129
|
if (!row || row.direction === 'unavailable' || row.baseline == null || row.delta == null) {
|
|
120
|
-
return { name: 'max-regression', status: 'inconclusive', detail: `Metric "${regressionMetric}" is unavailable for this comparison.` };
|
|
130
|
+
return { name: 'max-regression', metric: regressionMetric, status: 'inconclusive', detail: `Metric "${regressionMetric}" is unavailable for this comparison.` };
|
|
121
131
|
}
|
|
122
132
|
// A neutral-direction metric (inputBytes/outputBytes/taskCount/executorsAdded) measures
|
|
123
133
|
// workload volume, not performance: an increase isn't a regression, so no direction to check.
|
|
124
134
|
if (row.direction === 'neutral') {
|
|
125
|
-
return { name: 'max-regression', status: 'inconclusive', detail: `Metric "${regressionMetric}" measures workload volume, not performance: it has no regression direction to check.` };
|
|
135
|
+
return { name: 'max-regression', metric: regressionMetric, status: 'inconclusive', detail: `Metric "${regressionMetric}" measures workload volume, not performance: it has no regression direction to check.` };
|
|
126
136
|
}
|
|
127
137
|
if (row.direction !== 'regression') {
|
|
128
|
-
return { name: 'max-regression', status: 'pass', detail: `Metric "${regressionMetric}" did not regress (${row.direction}).` };
|
|
138
|
+
return { name: 'max-regression', metric: regressionMetric, status: 'pass', detail: `Metric "${regressionMetric}" did not regress (${row.direction}).` };
|
|
129
139
|
}
|
|
130
140
|
const pct = row.baseline === 0 ? Infinity : Math.abs(row.delta / row.baseline) * 100;
|
|
131
141
|
const pctLabel = row.baseline === 0
|
|
132
142
|
? `regressed from 0 to ${row.delta} (was absent/zero in baseline)`
|
|
133
143
|
: `regressed ${pct.toFixed(1)}%`;
|
|
134
144
|
return pct > maxRegressionPct
|
|
135
|
-
? { name: 'max-regression', status: 'violation', detail: `Metric "${regressionMetric}" ${pctLabel}, exceeding budget ${maxRegressionPct}%.` }
|
|
136
|
-
: { name: 'max-regression', status: 'pass', detail: `Metric "${regressionMetric}" ${pctLabel}, within budget ${maxRegressionPct}%.` };
|
|
145
|
+
? { name: 'max-regression', metric: regressionMetric, status: 'violation', detail: `Metric "${regressionMetric}" ${pctLabel}, exceeding budget ${maxRegressionPct}%.` }
|
|
146
|
+
: { name: 'max-regression', metric: regressionMetric, status: 'pass', detail: `Metric "${regressionMetric}" ${pctLabel}, within budget ${maxRegressionPct}%.` };
|
|
137
147
|
}
|
|
138
148
|
|
|
139
149
|
function checkFailOnIntroduced(comparison , band ) {
|
|
@@ -154,33 +164,50 @@ function pushComparisonBudget(
|
|
|
154
164
|
comparison ,
|
|
155
165
|
name ,
|
|
156
166
|
check ,
|
|
167
|
+
metric ,
|
|
157
168
|
) {
|
|
158
|
-
results.push(comparison ? check(comparison) : {
|
|
169
|
+
results.push(comparison ? check(comparison) : {
|
|
170
|
+
name, status: 'inconclusive', detail: 'No baseline comparison available to evaluate this budget.',
|
|
171
|
+
...(metric !== undefined ? { metric } : {}),
|
|
172
|
+
});
|
|
159
173
|
}
|
|
160
174
|
|
|
161
|
-
|
|
175
|
+
/** `thresholds`: the overrides the catalog was analyzed with, so a budget that recomputes a
|
|
176
|
+
* detector's figure (--max-skew) uses the same thresholds. */
|
|
177
|
+
export function evaluateBudgets({ appModel, catalog, budgets, comparison, thresholds }
|
|
162
178
|
|
|
179
|
+
|
|
163
180
|
) {
|
|
164
181
|
const results = [];
|
|
165
182
|
if (Number.isFinite(budgets.maxRuntimeMs)) results.push(checkRuntime(appModel, budgets.maxRuntimeMs ));
|
|
166
183
|
if (Number.isFinite(budgets.maxSpillGb)) results.push(checkSpill(appModel, catalog, budgets.maxSpillGb ));
|
|
167
|
-
if (Number.isFinite(budgets.maxSkewRatio)) results.push(checkSkew(appModel, budgets.maxSkewRatio ));
|
|
184
|
+
if (Number.isFinite(budgets.maxSkewRatio)) results.push(checkSkew(appModel, budgets.maxSkewRatio , skewMinTasksForP95(thresholds)));
|
|
168
185
|
if (Number.isFinite(budgets.maxFailedTaskRatePct)) results.push(checkFailedTaskRate(appModel, catalog, budgets.maxFailedTaskRatePct ));
|
|
169
186
|
if (Number.isFinite(budgets.minEfficiencyPct)) results.push(checkEfficiency(appModel, budgets.minEfficiencyPct ));
|
|
170
187
|
// Guarded here so any evaluateBudgets caller benefits: regressionMetric without
|
|
171
188
|
// maxRegressionPct would otherwise skip the `if` silently, reporting nothing.
|
|
172
189
|
if (budgets.regressionMetric !== undefined && budgets.maxRegressionPct === undefined) {
|
|
173
|
-
results.push({ name: 'max-regression', status: 'inconclusive', detail: `regressionMetric "${budgets.regressionMetric}" was set without maxRegressionPct; the regression budget was not evaluated.` });
|
|
190
|
+
results.push({ name: 'max-regression', metric: budgets.regressionMetric, status: 'inconclusive', detail: `regressionMetric "${budgets.regressionMetric}" was set without maxRegressionPct; the regression budget was not evaluated.` });
|
|
174
191
|
} else if (budgets.maxRegressionPct !== undefined) {
|
|
175
192
|
// `!== undefined`, not Number.isFinite: a zero-baseline regression's pct is Infinity,
|
|
176
193
|
// so an "unlimited" budget is a legitimate input (the CLI already rejects non-finite flags).
|
|
177
194
|
pushComparisonBudget(results, comparison, 'max-regression',
|
|
178
|
-
(c) => checkRegression(c, budgets.maxRegressionPct , budgets.regressionMetric ?? 'wallClock')
|
|
195
|
+
(c) => checkRegression(c, budgets.maxRegressionPct , budgets.regressionMetric ?? 'wallClock'),
|
|
196
|
+
budgets.regressionMetric ?? 'wallClock');
|
|
197
|
+
}
|
|
198
|
+
for (const { metric, maxPct } of budgets.regressionBudgets ?? []) {
|
|
199
|
+
pushComparisonBudget(results, comparison, 'max-regression', (c) => checkRegression(c, maxPct, metric), metric);
|
|
179
200
|
}
|
|
180
201
|
if (budgets.failOnIntroduced !== undefined) {
|
|
181
202
|
pushComparisonBudget(results, comparison, 'fail-on-introduced',
|
|
182
203
|
(c) => checkFailOnIntroduced(c, budgets.failOnIntroduced ));
|
|
183
204
|
}
|
|
205
|
+
// Always checked, unlike the opt-in budgets above: a run with no ApplicationEnd is
|
|
206
|
+
// inconclusive by default, so a passing budget can't hide a truncated log.
|
|
207
|
+
const incompleteRunFinding = catalog.find((f) => f.type === 'incompleteRun');
|
|
208
|
+
if (incompleteRunFinding) {
|
|
209
|
+
results.push({ name: 'run-complete', status: 'inconclusive', detail: incompleteRunFinding.recommendation ?? 'Event log has no ApplicationEnd event.' });
|
|
210
|
+
}
|
|
184
211
|
return {
|
|
185
212
|
results,
|
|
186
213
|
violated: results.some((r) => r.status === 'violation'),
|
|
@@ -1,15 +1,19 @@
|
|
|
1
|
-
import {
|
|
1
|
+
import { readdirSync, statSync, openSync, readSync, closeSync } from 'node:fs';
|
|
2
2
|
import { join, basename } from 'node:path';
|
|
3
3
|
import { createState, runParse, runParseFiles, reassembleRollingEntries } from '../parser-worker.js';
|
|
4
4
|
import { nodeParseCodecs } from './native-zstd.js';
|
|
5
5
|
import { createModelCallbacks } from '../model-assembler.js';
|
|
6
6
|
import { routeMessage, } from '../ingest.js';
|
|
7
7
|
|
|
8
|
+
import { mcpError } from '../mcp-error.js';
|
|
8
9
|
|
|
10
|
+
// No whole-file arrayBuffer(): the parser only ever reads bounded slices, and a
|
|
11
|
+
// whole-file read is what capped local logs at 2 GiB.
|
|
9
12
|
|
|
10
13
|
|
|
11
14
|
|
|
12
|
-
|
|
15
|
+
|
|
16
|
+
|
|
13
17
|
|
|
14
18
|
|
|
15
19
|
export function emptyAppModel() {
|
|
@@ -24,18 +28,48 @@ export function emptyAppModel() {
|
|
|
24
28
|
};
|
|
25
29
|
}
|
|
26
30
|
|
|
27
|
-
// File-like shape runParse/runParseFiles need
|
|
31
|
+
// File-like shape runParse/runParseFiles need (name, size, slice().arrayBuffer()), read with
|
|
32
|
+
// positioned readSync so only the requested slice is ever in memory, whatever the file size.
|
|
33
|
+
// The descriptor opens on the first read and closes once a read reaches the end of the file,
|
|
34
|
+
// so a rolling directory's parts hold at most one open descriptor at a time while streaming.
|
|
35
|
+
// A later read (a zip archive reads its tail first) reopens it; callers must still call
|
|
36
|
+
// close() once parsing settles, to cover reads that stopped early on an error.
|
|
28
37
|
export function nodeFileFromPath(path ) {
|
|
29
|
-
const
|
|
30
|
-
const
|
|
38
|
+
const name = basename(path);
|
|
39
|
+
const size = statSync(path).size;
|
|
40
|
+
let fd = null;
|
|
41
|
+
const close = () => {
|
|
42
|
+
if (fd === null) return;
|
|
43
|
+
const open = fd;
|
|
44
|
+
fd = null;
|
|
45
|
+
closeSync(open);
|
|
46
|
+
};
|
|
31
47
|
return {
|
|
32
|
-
name
|
|
33
|
-
size
|
|
48
|
+
name,
|
|
49
|
+
size,
|
|
34
50
|
slice(start , end ) {
|
|
35
|
-
|
|
36
|
-
|
|
51
|
+
return {
|
|
52
|
+
async arrayBuffer() {
|
|
53
|
+
const from = Math.max(0, start);
|
|
54
|
+
const to = Math.min(end, size);
|
|
55
|
+
if (to <= from) return new ArrayBuffer(0);
|
|
56
|
+
const buf = new Uint8Array(to - from);
|
|
57
|
+
fd ??= openSync(path, 'r');
|
|
58
|
+
// readSync may return fewer bytes than asked; loop until the slice is full.
|
|
59
|
+
for (let filled = 0; filled < buf.length;) {
|
|
60
|
+
const n = readSync(fd, buf, filled, buf.length - filled, from + filled);
|
|
61
|
+
if (n === 0) {
|
|
62
|
+
close();
|
|
63
|
+
throw new Error(`"${name}" shrank while it was being read`);
|
|
64
|
+
}
|
|
65
|
+
filled += n;
|
|
66
|
+
}
|
|
67
|
+
if (to === size) close();
|
|
68
|
+
return buf.buffer;
|
|
69
|
+
},
|
|
70
|
+
};
|
|
37
71
|
},
|
|
38
|
-
|
|
72
|
+
close,
|
|
39
73
|
};
|
|
40
74
|
}
|
|
41
75
|
|
|
@@ -96,7 +130,10 @@ export function collectViaDispatch(
|
|
|
96
130
|
return new Promise((resolve, reject) => {
|
|
97
131
|
const handlers = {
|
|
98
132
|
...cb,
|
|
99
|
-
onDone: (msg ) =>
|
|
133
|
+
onDone: (msg ) => {
|
|
134
|
+
cb.onDone(msg); // records the parse gaps on the model, as the dashboard's ingest does
|
|
135
|
+
resolve({ appModel, skippedLines: (msg )?.skippedLines ?? 0 });
|
|
136
|
+
},
|
|
100
137
|
onError: (msg ) => reject(onDecodeError(msg)),
|
|
101
138
|
};
|
|
102
139
|
const emit = (msg ) => dispatch(msg, handlers);
|
|
@@ -107,26 +144,41 @@ export function collectViaDispatch(
|
|
|
107
144
|
|
|
108
145
|
export async function collectRun(inputPath ) {
|
|
109
146
|
const stat = statSync(inputPath);
|
|
110
|
-
|
|
111
|
-
|
|
112
|
-
|
|
113
|
-
|
|
114
|
-
|
|
115
|
-
|
|
116
|
-
|
|
117
|
-
|
|
118
|
-
|
|
119
|
-
|
|
120
|
-
|
|
121
|
-
|
|
122
|
-
|
|
147
|
+
// Every file opened for this run, closed once parsing settles on any path (done, parse
|
|
148
|
+
// error, decode error, or a throw past the parser's guards).
|
|
149
|
+
const opened = [];
|
|
150
|
+
const open = (path ) => {
|
|
151
|
+
const file = nodeFileFromPath(path);
|
|
152
|
+
opened.push(file);
|
|
153
|
+
return file;
|
|
154
|
+
};
|
|
155
|
+
try {
|
|
156
|
+
// A file or folder that isn't a decodable event log rejects with invalid-event-log, the code
|
|
157
|
+
// shs-load.ts gives an archive that fails to decode, so MCP reports both the same way. The
|
|
158
|
+
// CLI reads only the message.
|
|
159
|
+
return await collectViaDispatch((state, emit, reject) => {
|
|
160
|
+
if (stat.isDirectory()) {
|
|
161
|
+
if (!isRollingLogDirectory(inputPath)) {
|
|
162
|
+
reject(mcpError('invalid-event-log', "This isn't a Spark rolling event-log directory. Pass a single event-log file instead."));
|
|
163
|
+
return;
|
|
164
|
+
}
|
|
165
|
+
const names = readdirSync(inputPath);
|
|
166
|
+
let ordered;
|
|
167
|
+
try {
|
|
168
|
+
ordered = reassembleRollingEntries(names);
|
|
169
|
+
} catch (e) {
|
|
170
|
+
reject(mcpError('invalid-event-log', (e ).message));
|
|
171
|
+
return;
|
|
172
|
+
}
|
|
173
|
+
const files = ordered.map((name) => open(join(inputPath, name)));
|
|
174
|
+
// .catch(reject), not void: a throw past the parser's guards would otherwise leave
|
|
175
|
+
// this Promise pending forever, surfacing only as an unhandled rejection.
|
|
176
|
+
runParseFiles(files, state, { emit, ...nodeParseCodecs }).catch(reject);
|
|
177
|
+
} else {
|
|
178
|
+
runParse(open(inputPath), state, { emit, ...nodeParseCodecs }).catch(reject);
|
|
123
179
|
}
|
|
124
|
-
|
|
125
|
-
|
|
126
|
-
|
|
127
|
-
|
|
128
|
-
} else {
|
|
129
|
-
runParse(nodeFileFromPath(inputPath), state, { emit, ...nodeParseCodecs }).catch(reject);
|
|
130
|
-
}
|
|
131
|
-
}, (msg) => new Error((msg ).message));
|
|
180
|
+
}, (msg) => mcpError('invalid-event-log', (msg ).message));
|
|
181
|
+
} finally {
|
|
182
|
+
for (const file of opened) file.close();
|
|
183
|
+
}
|
|
132
184
|
}
|
|
@@ -0,0 +1,83 @@
|
|
|
1
|
+
// Parses the regression budgets a CLI caller lists: repeated `--regression-budget <metric>:<pct>`
|
|
2
|
+
// flags and a `--budgets <file.json>` file. Every failure throws an Error naming the problem, so
|
|
3
|
+
// the caller refuses to run (exit 2) instead of quietly dropping a budget the user meant to gate on.
|
|
4
|
+
import { readFileSync } from 'node:fs';
|
|
5
|
+
import { resolve } from 'node:path';
|
|
6
|
+
import { COMPARISON_METRIC_KEYS } from '../run-comparison.js';
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
// Plain non-negative decimals only: Number() would also accept '', ' ', '0x10' and '1e3'.
|
|
12
|
+
const PCT_PATTERN = /^\d+(\.\d+)?$/;
|
|
13
|
+
|
|
14
|
+
function assertKnownMetric(metric , origin ) {
|
|
15
|
+
if (!COMPARISON_METRIC_KEYS.includes(metric)) {
|
|
16
|
+
throw new Error(`${origin}: unknown metric "${metric}" (expected one of: ${COMPARISON_METRIC_KEYS.join(', ')}).`);
|
|
17
|
+
}
|
|
18
|
+
}
|
|
19
|
+
|
|
20
|
+
/** One `--regression-budget` value, `<metric>:<pct>`. */
|
|
21
|
+
export function parseRegressionBudgetFlag(spec ) {
|
|
22
|
+
const origin = `--regression-budget "${spec}"`;
|
|
23
|
+
const colon = spec.indexOf(':');
|
|
24
|
+
if (colon === -1) throw new Error(`${origin}: expected <metric>:<pct>.`);
|
|
25
|
+
const metric = spec.slice(0, colon);
|
|
26
|
+
const pct = spec.slice(colon + 1);
|
|
27
|
+
assertKnownMetric(metric, origin);
|
|
28
|
+
if (!PCT_PATTERN.test(pct)) throw new Error(`${origin}: "${pct}" is not a non-negative percentage.`);
|
|
29
|
+
return { metric, maxPct: Number(pct) };
|
|
30
|
+
}
|
|
31
|
+
|
|
32
|
+
/** The parsed `--budgets` JSON: `{"regression": {"<metric>": <pct>}}`. */
|
|
33
|
+
export function parseBudgetsFile(raw , origin ) {
|
|
34
|
+
if (raw === null || typeof raw !== 'object' || Array.isArray(raw)) {
|
|
35
|
+
throw new Error(`${origin}: expected a JSON object like {"regression": {"wallClock": 10}}.`);
|
|
36
|
+
}
|
|
37
|
+
const unknownKeys = Object.keys(raw).filter((k) => k !== 'regression');
|
|
38
|
+
if (unknownKeys.length > 0) {
|
|
39
|
+
throw new Error(`${origin}: unknown key ${unknownKeys.map((k) => `"${k}"`).join(', ')} (the only key is "regression").`);
|
|
40
|
+
}
|
|
41
|
+
const regression = (raw ).regression;
|
|
42
|
+
if (regression === null || typeof regression !== 'object' || Array.isArray(regression)) {
|
|
43
|
+
throw new Error(`${origin}: "regression" must be an object mapping a metric key to a percentage.`);
|
|
44
|
+
}
|
|
45
|
+
return Object.entries(regression).map(([metric, pct]) => {
|
|
46
|
+
assertKnownMetric(metric, origin);
|
|
47
|
+
if (typeof pct !== 'number' || !Number.isFinite(pct) || pct < 0) {
|
|
48
|
+
throw new Error(`${origin}: "${metric}" must be a non-negative number (a percentage), got ${JSON.stringify(pct)}.`);
|
|
49
|
+
}
|
|
50
|
+
return { metric, maxPct: pct };
|
|
51
|
+
});
|
|
52
|
+
}
|
|
53
|
+
|
|
54
|
+
export function loadBudgetsFile(path ) {
|
|
55
|
+
const fullPath = resolve(path);
|
|
56
|
+
let text ;
|
|
57
|
+
try {
|
|
58
|
+
text = readFileSync(fullPath, 'utf8');
|
|
59
|
+
} catch (e) {
|
|
60
|
+
throw new Error(`Cannot read budgets file ${fullPath}: ${(e ).message}`, { cause: e });
|
|
61
|
+
}
|
|
62
|
+
let raw ;
|
|
63
|
+
try {
|
|
64
|
+
raw = JSON.parse(text);
|
|
65
|
+
} catch (e) {
|
|
66
|
+
throw new Error(`Budgets file ${fullPath} is not valid JSON: ${(e ).message}`, { cause: e });
|
|
67
|
+
}
|
|
68
|
+
return parseBudgetsFile(raw, `Budgets file ${fullPath}`);
|
|
69
|
+
}
|
|
70
|
+
|
|
71
|
+
/** Checks that no metric is budgeted twice across the sources (the legacy
|
|
72
|
+
* `--max-regression-pct`/`--regression-metric` pair counts as one), and returns the budgets. */
|
|
73
|
+
export function combineRegressionBudgets(sources ) {
|
|
74
|
+
const seen = new Map ();
|
|
75
|
+
for (const { origin, budget } of sources) {
|
|
76
|
+
const first = seen.get(budget.metric);
|
|
77
|
+
if (first !== undefined) {
|
|
78
|
+
throw new Error(`Metric "${budget.metric}" has two regression budgets (${first} and ${origin}); give each metric one.`);
|
|
79
|
+
}
|
|
80
|
+
seen.set(budget.metric, origin);
|
|
81
|
+
}
|
|
82
|
+
return sources.map((s) => s.budget);
|
|
83
|
+
}
|
|
@@ -0,0 +1,28 @@
|
|
|
1
|
+
// Reads a --thresholds config file for the CLI and the MCP server. Every failure throws an Error
|
|
2
|
+
// naming the file and the problem, so the caller refuses to run instead of quietly falling back
|
|
3
|
+
// to the defaults the user meant to change.
|
|
4
|
+
import { readFileSync } from 'node:fs';
|
|
5
|
+
import { resolve } from 'node:path';
|
|
6
|
+
import { parseThresholdOverrides } from '../threshold-overrides.js';
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
export function loadThresholdOverrides(path ) {
|
|
10
|
+
const fullPath = resolve(path);
|
|
11
|
+
let text ;
|
|
12
|
+
try {
|
|
13
|
+
text = readFileSync(fullPath, 'utf8');
|
|
14
|
+
} catch (e) {
|
|
15
|
+
throw new Error(`Cannot read thresholds file ${fullPath}: ${(e ).message}`, { cause: e });
|
|
16
|
+
}
|
|
17
|
+
let raw ;
|
|
18
|
+
try {
|
|
19
|
+
raw = JSON.parse(text);
|
|
20
|
+
} catch (e) {
|
|
21
|
+
throw new Error(`Thresholds file ${fullPath} is not valid JSON: ${(e ).message}`, { cause: e });
|
|
22
|
+
}
|
|
23
|
+
try {
|
|
24
|
+
return parseThresholdOverrides(raw);
|
|
25
|
+
} catch (e) {
|
|
26
|
+
throw new Error(`Thresholds file ${fullPath}: ${(e ).message}`, { cause: e });
|
|
27
|
+
}
|
|
28
|
+
}
|
|
@@ -0,0 +1,177 @@
|
|
|
1
|
+
import { formatDuration, typeTag } from './format-utils.js';
|
|
2
|
+
import { TAG_HELP } from './finding-tag-help.js';
|
|
3
|
+
import { NEUTRAL_METRIC_KEYS, } from './run-comparison.js';
|
|
4
|
+
|
|
5
|
+
/** The slice of a comparison metric row the verdict reads. */
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
/** The slice of a finding-category delta the verdict reads. */
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
/** How many of a run's ended jobs failed, counted as `summarizeRunOutcome`
|
|
23
|
+
* counts them. */
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
/** A change under this share of the baseline's value reads as "about the same", for
|
|
41
|
+
* run time and cost metrics alike: run-to-run noise on a shared cluster easily
|
|
42
|
+
* moves a job a percent or two. */
|
|
43
|
+
export const SAME_CHANGE_SHARE = 0.02;
|
|
44
|
+
|
|
45
|
+
function measured(metric ) {
|
|
46
|
+
return metric.baseline != null && metric.candidate != null;
|
|
47
|
+
}
|
|
48
|
+
|
|
49
|
+
function movedPastNoise(metric ) {
|
|
50
|
+
if (metric.baseline === 0) return metric.candidate !== 0;
|
|
51
|
+
return Math.abs(metric.candidate - metric.baseline) / Math.abs(metric.baseline) >= SAME_CHANGE_SHARE;
|
|
52
|
+
}
|
|
53
|
+
|
|
54
|
+
/** Plain name for a finding type ("Memory and disk spill"), falling back to
|
|
55
|
+
* its tag when the tag has no help entry. */
|
|
56
|
+
function categoryName(type ) {
|
|
57
|
+
const tag = typeTag(type);
|
|
58
|
+
return TAG_HELP[tag]?.expansion ?? tag;
|
|
59
|
+
}
|
|
60
|
+
|
|
61
|
+
/** Net count change per displayed category name. Categories are tallied per
|
|
62
|
+
* (rule, impact band), and several rules share one name (every Plan Advisor
|
|
63
|
+
* type reads "Plan advisor", every stage-shape sub-rule "Stage shape"), so a
|
|
64
|
+
* rule whose findings moved from critical to warning, or two rules under one
|
|
65
|
+
* name moving opposite ways, would otherwise
|
|
66
|
+
* read as both "introduced" and "resolved". Order follows first appearance,
|
|
67
|
+
* introduced before resolved. */
|
|
68
|
+
function netByCategory(findings ) {
|
|
69
|
+
const net = new Map ();
|
|
70
|
+
for (const item of [...findings.introduced, ...findings.resolved]) {
|
|
71
|
+
const name = categoryName(item.type);
|
|
72
|
+
net.set(name, (net.get(name) ?? 0) + item.candCount - item.baseCount);
|
|
73
|
+
}
|
|
74
|
+
return net;
|
|
75
|
+
}
|
|
76
|
+
|
|
77
|
+
function namesWhere(net , keep ) {
|
|
78
|
+
return [...net].filter(([, change]) => keep(change)).map(([name]) => name);
|
|
79
|
+
}
|
|
80
|
+
|
|
81
|
+
/** "The candidate had 2 of 5 jobs fail", or, when every job failed, "The candidate's only
|
|
82
|
+
* job failed" / "All 3 of the candidate's jobs failed" rather than "1 of 1 jobs". */
|
|
83
|
+
function jobsFailed(run , { failedJobs, totalJobs } ) {
|
|
84
|
+
if (failedJobs < totalJobs) return `The ${run} had ${failedJobs} of ${totalJobs} jobs fail`;
|
|
85
|
+
return totalJobs === 1 ? `The ${run}'s only job failed` : `All ${totalJobs} of the ${run}'s jobs failed`;
|
|
86
|
+
}
|
|
87
|
+
|
|
88
|
+
/** The baseline's failures in the parenthesis after the candidate's. */
|
|
89
|
+
function baselineFailures({ failedJobs, totalJobs } ) {
|
|
90
|
+
if (failedJobs === 0) return 'none';
|
|
91
|
+
if (failedJobs < totalJobs) return `${failedJobs} of ${totalJobs}`;
|
|
92
|
+
return totalJobs === 1 ? 'its only job failed' : `all ${totalJobs} failed`;
|
|
93
|
+
}
|
|
94
|
+
|
|
95
|
+
/** The headline when either run had failed jobs, as the run verdict leads
|
|
96
|
+
* with a failure: a faster candidate that dropped work is not an improvement,
|
|
97
|
+
* and with equal failure counts the tone stays neutral since a failing run
|
|
98
|
+
* that ends sooner may just have failed earlier. Null when both runs
|
|
99
|
+
* completed. */
|
|
100
|
+
function failureHeadline(base , cand ) {
|
|
101
|
+
if (base.failedJobs === 0 && cand.failedJobs === 0) return null;
|
|
102
|
+
const tone = cand.failedJobs > base.failedJobs ? 'worse' : cand.failedJobs < base.failedJobs ? 'better' : 'same';
|
|
103
|
+
if (cand.failedJobs === 0) return { title: `${jobsFailed('baseline', base)}; the candidate completed`, tone };
|
|
104
|
+
return { title: `${jobsFailed('candidate', cand)} (baseline: ${baselineFailures(base)})`, tone };
|
|
105
|
+
}
|
|
106
|
+
|
|
107
|
+
/** One plain answer to "did the candidate get better or worse than the baseline", from the
|
|
108
|
+
* comparison's own whole-run metrics and finding-category tallies: run time
|
|
109
|
+
* first, then which cost metrics moved each way past run-to-run noise, then which finding
|
|
110
|
+
* categories appeared or went away. When either run had failed jobs, that
|
|
111
|
+
* leads instead and run time becomes the first sentence. When either log is
|
|
112
|
+
* incomplete, run time is stated as what each log covers, never as faster or
|
|
113
|
+
* slower, and the tone stays neutral. Volume and count metrics (input, output,
|
|
114
|
+
* tasks, executors) are left out because more or less of them is not
|
|
115
|
+
* inherently better or worse. */
|
|
116
|
+
export function summarizeComparison(
|
|
117
|
+
metrics ,
|
|
118
|
+
findings ,
|
|
119
|
+
jobs ,
|
|
120
|
+
) {
|
|
121
|
+
const wall = metrics.find((metric) => metric.key === 'wallClock');
|
|
122
|
+
let title = 'Run time could not be compared between the baseline and the candidate';
|
|
123
|
+
let tone = 'unknown';
|
|
124
|
+
// A log with no end-of-run record stops where the run was cut off, so its
|
|
125
|
+
// shorter time is not a speed-up: say how much each log covers, neutrally.
|
|
126
|
+
const incompleteRuns = jobs ? (['baseline', 'candidate'] ).filter((run) => jobs[run].incomplete) : [];
|
|
127
|
+
if (wall && wall.baseline != null && wall.baseline > 0 && wall.candidate != null && incompleteRuns.length > 0) {
|
|
128
|
+
const change = wall.candidate - wall.baseline;
|
|
129
|
+
title = Math.abs(change / wall.baseline) < SAME_CHANGE_SHARE
|
|
130
|
+
? "The candidate's log covers about as much run time as the baseline's"
|
|
131
|
+
: `The candidate's log covers ${formatDuration(Math.abs(change))} ${change < 0 ? 'less' : 'more'} run time than the baseline's`;
|
|
132
|
+
} else if (wall && wall.baseline != null && wall.baseline > 0 && wall.candidate != null) {
|
|
133
|
+
const change = wall.candidate - wall.baseline;
|
|
134
|
+
const share = change / wall.baseline;
|
|
135
|
+
if (Math.abs(share) < SAME_CHANGE_SHARE) {
|
|
136
|
+
title = 'The candidate took about as long as the baseline';
|
|
137
|
+
tone = 'same';
|
|
138
|
+
} else {
|
|
139
|
+
const faster = change < 0;
|
|
140
|
+
title = `The candidate finished ${formatDuration(Math.abs(change))} ${faster ? 'faster' : 'slower'} than the baseline (${Math.round(Math.abs(share) * 100)}%)`;
|
|
141
|
+
tone = faster ? 'better' : 'worse';
|
|
142
|
+
}
|
|
143
|
+
}
|
|
144
|
+
|
|
145
|
+
const cost = metrics.filter((metric) => metric.key !== 'wallClock' && !NEUTRAL_METRIC_KEYS.has(metric.key)).filter(measured);
|
|
146
|
+
const moved = cost.filter(movedPastNoise);
|
|
147
|
+
const worse = moved.filter((metric) => metric.direction === 'regression').map((metric) => metric.label);
|
|
148
|
+
const better = moved.filter((metric) => metric.direction === 'improvement').map((metric) => metric.label);
|
|
149
|
+
const sentences = [];
|
|
150
|
+
const failure = jobs ? failureHeadline(jobs.baseline, jobs.candidate) : null;
|
|
151
|
+
if (failure) {
|
|
152
|
+
sentences.push(`${title}.`);
|
|
153
|
+
title = failure.title;
|
|
154
|
+
tone = failure.tone;
|
|
155
|
+
}
|
|
156
|
+
if (incompleteRuns.length === 2) {
|
|
157
|
+
sentences.push('Neither log has an end-of-run record, so their times cover only what each log captured, not how long the runs took.');
|
|
158
|
+
} else if (incompleteRuns.length === 1) {
|
|
159
|
+
sentences.push(`The ${incompleteRuns[0]}'s log has no end-of-run record, so its time covers only what the log captured, not how long the run took.`);
|
|
160
|
+
}
|
|
161
|
+
if (worse.length > 0) sentences.push(`Worse in the candidate: ${worse.join(', ')}.`);
|
|
162
|
+
if (better.length > 0) sentences.push(`Better in the candidate: ${better.join(', ')}.`);
|
|
163
|
+
if (cost.length > 0 && worse.length === 0 && better.length === 0) sentences.push('Other measured cost metrics look about the same.');
|
|
164
|
+
|
|
165
|
+
const net = netByCategory(findings);
|
|
166
|
+
const introduced = namesWhere(net, (change) => change > 0);
|
|
167
|
+
const resolved = namesWhere(net, (change) => change < 0);
|
|
168
|
+
if (introduced.length > 0) sentences.push(`New or more frequent in the candidate: ${introduced.join(', ')}.`);
|
|
169
|
+
if (resolved.length > 0) sentences.push(`Less frequent in the candidate: ${resolved.join(', ')}.`);
|
|
170
|
+
return { title, tone, sentences };
|
|
171
|
+
}
|
|
172
|
+
|
|
173
|
+
/** The comparison verdict for a core comparison result: the one the dashboard's comparison page
|
|
174
|
+
* leads with, and the one the CLI's --baseline report and MCP compare_runs carry. */
|
|
175
|
+
export function comparisonVerdict(result ) {
|
|
176
|
+
return summarizeComparison(result.metrics, result.findings, result.jobOutcomes);
|
|
177
|
+
}
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
ab36c5fdcdc1a32a4e6206459e843e8d2267a90e6f1de29671827315d81bc374
|