sparkforensics-mcp 0.4.0 → 0.5.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/package.json +1 -1
- package/vendor-core/allocation.js +106 -0
- package/vendor-core/analyzer.js +13 -13
- package/vendor-core/cli/budgets.js +23 -9
- package/vendor-core/cli/collect-run.js +11 -4
- package/vendor-core/cli/regression-budgets.js +83 -0
- package/vendor-core/comparison-verdict.js +22 -22
- package/vendor-core/core-source-hash.txt +1 -1
- package/vendor-core/detectors.js +202 -68
- package/vendor-core/docs-content/detection/cache.md +3 -2
- package/vendor-core/docs-content/detection/cfg.md +9 -8
- package/vendor-core/docs-content/detection/chrn.md +1 -2
- package/vendor-core/docs-content/detection/cold.md +4 -2
- package/vendor-core/docs-content/detection/fail.md +3 -2
- package/vendor-core/docs-content/detection/gc.md +3 -2
- package/vendor-core/docs-content/detection/host.md +2 -1
- package/vendor-core/docs-content/detection/local.md +1 -1
- package/vendor-core/docs-content/detection/mem.md +5 -2
- package/vendor-core/docs-content/detection/plan.md +2 -1
- package/vendor-core/docs-content/detection/sfail.md +2 -1
- package/vendor-core/docs-content/detection/shape.md +5 -4
- package/vendor-core/docs-content/detection/skew.md +3 -1
- package/vendor-core/docs-content/detection/slow.md +2 -2
- package/vendor-core/docs-content/detection/spec.md +2 -3
- package/vendor-core/docs-content/detection/spill.md +1 -1
- package/vendor-core/docs-site-config.js +1 -1
- package/vendor-core/effective-conf.js +107 -0
- package/vendor-core/efficiency-model.js +8 -6
- package/vendor-core/event-handlers.js +160 -47
- package/vendor-core/event-schemas.js +2 -0
- package/vendor-core/evidence-report.js +18 -10
- package/vendor-core/finding-generic-recommendation.js +20 -1
- package/vendor-core/finding-names.js +7 -0
- package/vendor-core/finding-presentation.js +61 -26
- package/vendor-core/finding-tag-help.js +1 -1
- package/vendor-core/finding-types.js +12 -0
- package/vendor-core/format-utils.js +4 -3
- package/vendor-core/impact-estimator.js +20 -2
- package/vendor-core/impact-format.js +14 -13
- package/vendor-core/impact-model.js +27 -5
- package/vendor-core/ingest.js +4 -2
- package/vendor-core/list-runs.js +5 -2
- package/vendor-core/mcp-tools.js +1 -1
- package/vendor-core/model-assembler.js +23 -1
- package/vendor-core/parser-worker.js +1 -1
- package/vendor-core/proxy.js +3 -1
- package/vendor-core/python-stage.js +25 -0
- package/vendor-core/recommendation-rollup.js +16 -9
- package/vendor-core/redact.js +51 -10
- package/vendor-core/remediation.js +20 -0
- package/vendor-core/run-comparison.js +43 -24
- package/vendor-core/run-interpretation.js +2 -1
- package/vendor-core/run-metrics.js +198 -0
- package/vendor-core/run-totals.js +24 -0
- package/vendor-core/run-verdict.js +3 -4
- package/vendor-core/scorecard-estimates.js +1 -0
- package/vendor-core/session-snapshot.js +7 -0
- package/vendor-core/shs-schemas.js +2 -2
- package/vendor-core/spark-memory.js +17 -0
- package/vendor-core/stage-plan-nodes.js +18 -0
- package/vendor-core/stage-quantiles.js +4 -0
- package/vendor-core/types.js +49 -1
- package/vendor-core/wasted-core-hours.js +10 -7
- package/vendor-core/write-targets.js +312 -0
package/package.json
CHANGED
|
@@ -0,0 +1,106 @@
|
|
|
1
|
+
import { parseSparkMemoryMB } from './spark-memory.js';
|
|
2
|
+
|
|
3
|
+
|
|
4
|
+
const MS_PER_HOUR = 3_600_000;
|
|
5
|
+
const MIB_PER_GIB = 1024;
|
|
6
|
+
// Spark's documented floor and factor for the default executor memory overhead
|
|
7
|
+
// (spark.executor.memoryOverhead = max(factor * executor memory, 384 MiB)).
|
|
8
|
+
const MIN_OVERHEAD_MIB = 384;
|
|
9
|
+
const DEFAULT_OVERHEAD_FACTOR = 0.1;
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
/** The latest timestamp the log records: the application start and end and every stage and
|
|
26
|
+
* executor event. Where a log with no ApplicationEnd stops. */
|
|
27
|
+
export function lastObservedTimestamp(input ) {
|
|
28
|
+
let last = null;
|
|
29
|
+
const take = (t ) => {
|
|
30
|
+
if (typeof t === 'number' && Number.isFinite(t) && t > 0 && (last == null || t > last)) last = t;
|
|
31
|
+
};
|
|
32
|
+
take(input.app?.startTime);
|
|
33
|
+
take(input.app?.endTime);
|
|
34
|
+
for (const s of input.stages.values()) { take(s.submittedAt); take(s.completedAt); }
|
|
35
|
+
for (const e of [...input.executors.added, ...input.executors.removed]) take(e.timestamp);
|
|
36
|
+
return last;
|
|
37
|
+
}
|
|
38
|
+
|
|
39
|
+
// Spark's default executor memory when spark.executor.memory is unset.
|
|
40
|
+
const DEFAULT_EXECUTOR_MEMORY_MIB = 1024;
|
|
41
|
+
|
|
42
|
+
// One container's memory in MiB, as Spark requests it from the cluster manager: executor heap,
|
|
43
|
+
// plus overhead, plus off-heap and PySpark worker memory when configured. Null when the log
|
|
44
|
+
// records no Spark properties (nothing to tell a default from a missing config) or a memory key it
|
|
45
|
+
// does record cannot be read.
|
|
46
|
+
function executorMemoryMiB(app ) {
|
|
47
|
+
const config = app?.config;
|
|
48
|
+
if (config == null) return null;
|
|
49
|
+
// undefined: key absent; null: present but unreadable.
|
|
50
|
+
const mib = (key ) => (config[key] == null ? undefined : parseSparkMemoryMB(config[key]));
|
|
51
|
+
const heap = mib('spark.executor.memory') ?? (config['spark.executor.memory'] == null ? DEFAULT_EXECUTOR_MEMORY_MIB : null);
|
|
52
|
+
if (heap == null) return null;
|
|
53
|
+
|
|
54
|
+
let overhead = mib('spark.executor.memoryOverhead');
|
|
55
|
+
if (overhead === undefined) overhead = mib('spark.yarn.executor.memoryOverhead'); // legacy key
|
|
56
|
+
if (overhead === undefined) {
|
|
57
|
+
const factor = Number.parseFloat(config['spark.executor.memoryOverheadFactor'] ?? '');
|
|
58
|
+
overhead = Math.max(MIN_OVERHEAD_MIB, Math.round(heap * (Number.isFinite(factor) && factor > 0 ? factor : DEFAULT_OVERHEAD_FACTOR)));
|
|
59
|
+
}
|
|
60
|
+
const offHeap = String(config['spark.memory.offHeap.enabled']).toLowerCase() === 'true' ? mib('spark.memory.offHeap.size') ?? 0 : 0;
|
|
61
|
+
const pyspark = mib('spark.executor.pyspark.memory') ?? 0;
|
|
62
|
+
if (overhead === null || offHeap === null || pyspark === null) return null;
|
|
63
|
+
return heap + overhead + offHeap + pyspark;
|
|
64
|
+
}
|
|
65
|
+
|
|
66
|
+
/** Allocated core-hours and memory GiB-hours from the executor lifecycle: each executor counts
|
|
67
|
+
* from its ExecutorAdded timestamp to its first later ExecutorRemoved timestamp. One with no
|
|
68
|
+
* removal closes at the application end when the log has one, else at the last timestamp the log
|
|
69
|
+
* records (a cut-off log). Cores are the ExecutorAdded event's Total Cores, else
|
|
70
|
+
* spark.executor.cores; memory per executor is spark.executor.memory (default 1g) plus the
|
|
71
|
+
* overhead (spark.executor.memoryOverhead, else the legacy spark.yarn.executor.memoryOverhead,
|
|
72
|
+
* else the larger of 384 MiB and spark.executor.memoryOverheadFactor, default 0.1, times the
|
|
73
|
+
* memory), plus spark.memory.offHeap.size when spark.memory.offHeap.enabled is true, plus
|
|
74
|
+
* spark.executor.pyspark.memory.
|
|
75
|
+
* Null, never 0, for a figure whose inputs the log lacks. */
|
|
76
|
+
export function computeAllocation(input ) {
|
|
77
|
+
const added = input.executors.added.filter((e) => e.kind === 'added');
|
|
78
|
+
if (added.length === 0) return { coreHours: null, memoryGbHours: null };
|
|
79
|
+
const removedAt = new Map ();
|
|
80
|
+
for (const e of input.executors.removed) {
|
|
81
|
+
if (e.kind !== 'removed') continue;
|
|
82
|
+
const times = removedAt.get(e.executorId);
|
|
83
|
+
if (times) times.push(e.timestamp); else removedAt.set(e.executorId, [e.timestamp]);
|
|
84
|
+
}
|
|
85
|
+
const closeAt = input.app?.endTime ?? lastObservedTimestamp(input);
|
|
86
|
+
const configuredCores = Number.parseInt(input.app?.config?.['spark.executor.cores'] ?? '', 10);
|
|
87
|
+
const memoryMiB = executorMemoryMiB(input.app);
|
|
88
|
+
|
|
89
|
+
let coreMs = 0;
|
|
90
|
+
let memoryMiBMs = 0;
|
|
91
|
+
let coresKnown = true;
|
|
92
|
+
const seen = new Set ();
|
|
93
|
+
for (const e of added) {
|
|
94
|
+
if (seen.has(e.executorId)) continue; // a replayed ExecutorAdded is the same executor
|
|
95
|
+
seen.add(e.executorId);
|
|
96
|
+
const removal = (removedAt.get(e.executorId) ?? []).filter((t) => t >= e.timestamp).sort((a, b) => a - b)[0];
|
|
97
|
+
const aliveMs = Math.max(0, (removal ?? closeAt ?? e.timestamp) - e.timestamp);
|
|
98
|
+
const cores = e.totalCores > 0 ? e.totalCores : Number.isFinite(configuredCores) ? configuredCores : null;
|
|
99
|
+
if (cores == null) coresKnown = false; else coreMs += cores * aliveMs;
|
|
100
|
+
memoryMiBMs += (memoryMiB ?? 0) * aliveMs;
|
|
101
|
+
}
|
|
102
|
+
return {
|
|
103
|
+
coreHours: coresKnown ? coreMs / MS_PER_HOUR : null,
|
|
104
|
+
memoryGbHours: memoryMiB != null ? memoryMiBMs / MIB_PER_GIB / MS_PER_HOUR : null,
|
|
105
|
+
};
|
|
106
|
+
}
|
package/vendor-core/analyzer.js
CHANGED
|
@@ -88,32 +88,32 @@ export function findingId(f ) {
|
|
|
88
88
|
return fnv1a(`${f.type}|${locationKey(f)}|${f.metric ?? ''}|${f.value ?? f.valueText ?? ''}|${disc}`);
|
|
89
89
|
}
|
|
90
90
|
|
|
91
|
-
// skew
|
|
92
|
-
//
|
|
93
|
-
//
|
|
94
|
-
//
|
|
95
|
-
//
|
|
96
|
-
//
|
|
91
|
+
// skew (either branch) and straggler both claim the stage's replayed tail recovery
|
|
92
|
+
// (tailReplayRecoveryMs via tailRecoveryMs): the same slow-task tail reported by two detectors
|
|
93
|
+
// (see "Overlap caveat: skew / straggler" in impact-estimation.md). skew's branch only changes
|
|
94
|
+
// the fallback single-task delta on a stage without the replay, so every skew + straggler pair
|
|
95
|
+
// on a stage is flagged. Flags both sides via validationRequired (rather than suppressing
|
|
96
|
+
// either) so neither finding's own diagnostic value is lost; the flag rides the same
|
|
97
97
|
// confidence-caveat UI a reader already sees before trusting either finding's magnitude.
|
|
98
98
|
function overlapNote(otherType ) {
|
|
99
|
-
return `This overlaps with the ${otherType} finding on this stage: both
|
|
99
|
+
return `This overlaps with the ${otherType} finding on this stage: both measure the same slow-task tail, so don't add their recoverable-time figures together.`;
|
|
100
100
|
}
|
|
101
101
|
|
|
102
102
|
function flagSkewStragglerOverlap(findings ) {
|
|
103
|
-
const
|
|
104
|
-
findings.filter((f) => f.type === 'skew' && f.
|
|
103
|
+
const skewStages = new Set(
|
|
104
|
+
findings.filter((f) => f.type === 'skew' && f.stageId != null).map((f) => f.stageId),
|
|
105
105
|
);
|
|
106
|
-
if (
|
|
106
|
+
if (skewStages.size === 0) return;
|
|
107
107
|
const stragglerStages = new Set(
|
|
108
108
|
findings.filter((f) => f.type === 'straggler' && f.stageId != null).map((f) => f.stageId),
|
|
109
109
|
);
|
|
110
|
-
const overlapStages = new Set([...
|
|
110
|
+
const overlapStages = new Set([...skewStages].filter((id) => stragglerStages.has(id)));
|
|
111
111
|
if (overlapStages.size === 0) return;
|
|
112
112
|
for (const f of findings) {
|
|
113
113
|
if (f.stageId == null || !overlapStages.has(f.stageId)) continue;
|
|
114
114
|
const note = f.type === 'skew' ? overlapNote('straggler') : f.type === 'straggler' ? overlapNote('skew') : null;
|
|
115
115
|
if (!note) continue;
|
|
116
|
-
f.validationRequired = f.validationRequired
|
|
116
|
+
f.validationRequired = [f.validationRequired, note].filter(Boolean).join(' ');
|
|
117
117
|
}
|
|
118
118
|
}
|
|
119
119
|
|
|
@@ -201,7 +201,7 @@ export function analyze(
|
|
|
201
201
|
// One occupancy sweep per analysis, shared by the detectors' runtime floors and every entry's
|
|
202
202
|
// estimate(), so a floor gates on the same occupancy-clipped figure displayed as savings.
|
|
203
203
|
const impact = {
|
|
204
|
-
stages, totalCores,
|
|
204
|
+
stages, totalCores, sql,
|
|
205
205
|
occupancy: computeOccupancy(stages , totalCores),
|
|
206
206
|
};
|
|
207
207
|
// The one cast from the posted-model types to the detector-side shapes: types.ts's Stage and
|
|
@@ -4,6 +4,7 @@ import { effectiveThresholds } from '../threshold-overrides.js';
|
|
|
4
4
|
import { IMPACT_BAND_ORDER } from '../format-utils.js';
|
|
5
5
|
|
|
6
6
|
|
|
7
|
+
|
|
7
8
|
|
|
8
9
|
const IMPACT_BANDS = Object.keys(IMPACT_BAND_ORDER) ;
|
|
9
10
|
|
|
@@ -11,12 +12,16 @@ const IMPACT_BANDS = Object.keys(IMPACT_BAND_ORDER) ;
|
|
|
11
12
|
|
|
12
13
|
|
|
13
14
|
|
|
15
|
+
|
|
16
|
+
|
|
14
17
|
|
|
15
18
|
|
|
16
19
|
|
|
17
20
|
|
|
18
21
|
|
|
19
22
|
|
|
23
|
+
|
|
24
|
+
|
|
20
25
|
|
|
21
26
|
|
|
22
27
|
// The skew entry's minTasksForP95 under the run's overrides, so the budget measures the same
|
|
@@ -107,7 +112,8 @@ function checkEfficiency(appModel , minPct ) {
|
|
|
107
112
|
}
|
|
108
113
|
const model = computeEfficiencyModel({
|
|
109
114
|
app: appModel.app, stages: appModel.stages,
|
|
110
|
-
executorsAdded: appModel.executors.added,
|
|
115
|
+
executorsAdded: appModel.executors.added, executorsRemoved: appModel.executors.removed,
|
|
116
|
+
runAggregates: appModel.runAggregates,
|
|
111
117
|
});
|
|
112
118
|
if (model.wastagePct == null) {
|
|
113
119
|
return { name: 'min-efficiency', status: 'inconclusive', detail: 'Busy core time could not be computed (no available compute hours).' };
|
|
@@ -121,23 +127,23 @@ function checkEfficiency(appModel , minPct ) {
|
|
|
121
127
|
function checkRegression(comparison , maxRegressionPct , regressionMetric ) {
|
|
122
128
|
const row = comparison.metrics.find((m) => m.key === regressionMetric);
|
|
123
129
|
if (!row || row.direction === 'unavailable' || row.baseline == null || row.delta == null) {
|
|
124
|
-
return { name: 'max-regression', status: 'inconclusive', detail: `Metric "${regressionMetric}" is unavailable for this comparison.` };
|
|
130
|
+
return { name: 'max-regression', metric: regressionMetric, status: 'inconclusive', detail: `Metric "${regressionMetric}" is unavailable for this comparison.` };
|
|
125
131
|
}
|
|
126
132
|
// A neutral-direction metric (inputBytes/outputBytes/taskCount/executorsAdded) measures
|
|
127
133
|
// workload volume, not performance: an increase isn't a regression, so no direction to check.
|
|
128
134
|
if (row.direction === 'neutral') {
|
|
129
|
-
return { name: 'max-regression', status: 'inconclusive', detail: `Metric "${regressionMetric}" measures workload volume, not performance: it has no regression direction to check.` };
|
|
135
|
+
return { name: 'max-regression', metric: regressionMetric, status: 'inconclusive', detail: `Metric "${regressionMetric}" measures workload volume, not performance: it has no regression direction to check.` };
|
|
130
136
|
}
|
|
131
137
|
if (row.direction !== 'regression') {
|
|
132
|
-
return { name: 'max-regression', status: 'pass', detail: `Metric "${regressionMetric}" did not regress (${row.direction}).` };
|
|
138
|
+
return { name: 'max-regression', metric: regressionMetric, status: 'pass', detail: `Metric "${regressionMetric}" did not regress (${row.direction}).` };
|
|
133
139
|
}
|
|
134
140
|
const pct = row.baseline === 0 ? Infinity : Math.abs(row.delta / row.baseline) * 100;
|
|
135
141
|
const pctLabel = row.baseline === 0
|
|
136
142
|
? `regressed from 0 to ${row.delta} (was absent/zero in baseline)`
|
|
137
143
|
: `regressed ${pct.toFixed(1)}%`;
|
|
138
144
|
return pct > maxRegressionPct
|
|
139
|
-
? { name: 'max-regression', status: 'violation', detail: `Metric "${regressionMetric}" ${pctLabel}, exceeding budget ${maxRegressionPct}%.` }
|
|
140
|
-
: { name: 'max-regression', status: 'pass', detail: `Metric "${regressionMetric}" ${pctLabel}, within budget ${maxRegressionPct}%.` };
|
|
145
|
+
? { name: 'max-regression', metric: regressionMetric, status: 'violation', detail: `Metric "${regressionMetric}" ${pctLabel}, exceeding budget ${maxRegressionPct}%.` }
|
|
146
|
+
: { name: 'max-regression', metric: regressionMetric, status: 'pass', detail: `Metric "${regressionMetric}" ${pctLabel}, within budget ${maxRegressionPct}%.` };
|
|
141
147
|
}
|
|
142
148
|
|
|
143
149
|
function checkFailOnIntroduced(comparison , band ) {
|
|
@@ -158,8 +164,12 @@ function pushComparisonBudget(
|
|
|
158
164
|
comparison ,
|
|
159
165
|
name ,
|
|
160
166
|
check ,
|
|
167
|
+
metric ,
|
|
161
168
|
) {
|
|
162
|
-
results.push(comparison ? check(comparison) : {
|
|
169
|
+
results.push(comparison ? check(comparison) : {
|
|
170
|
+
name, status: 'inconclusive', detail: 'No baseline comparison available to evaluate this budget.',
|
|
171
|
+
...(metric !== undefined ? { metric } : {}),
|
|
172
|
+
});
|
|
163
173
|
}
|
|
164
174
|
|
|
165
175
|
/** `thresholds`: the overrides the catalog was analyzed with, so a budget that recomputes a
|
|
@@ -177,12 +187,16 @@ export function evaluateBudgets({ appModel, catalog, budgets, comparison, thresh
|
|
|
177
187
|
// Guarded here so any evaluateBudgets caller benefits: regressionMetric without
|
|
178
188
|
// maxRegressionPct would otherwise skip the `if` silently, reporting nothing.
|
|
179
189
|
if (budgets.regressionMetric !== undefined && budgets.maxRegressionPct === undefined) {
|
|
180
|
-
results.push({ name: 'max-regression', status: 'inconclusive', detail: `regressionMetric "${budgets.regressionMetric}" was set without maxRegressionPct; the regression budget was not evaluated.` });
|
|
190
|
+
results.push({ name: 'max-regression', metric: budgets.regressionMetric, status: 'inconclusive', detail: `regressionMetric "${budgets.regressionMetric}" was set without maxRegressionPct; the regression budget was not evaluated.` });
|
|
181
191
|
} else if (budgets.maxRegressionPct !== undefined) {
|
|
182
192
|
// `!== undefined`, not Number.isFinite: a zero-baseline regression's pct is Infinity,
|
|
183
193
|
// so an "unlimited" budget is a legitimate input (the CLI already rejects non-finite flags).
|
|
184
194
|
pushComparisonBudget(results, comparison, 'max-regression',
|
|
185
|
-
(c) => checkRegression(c, budgets.maxRegressionPct , budgets.regressionMetric ?? 'wallClock')
|
|
195
|
+
(c) => checkRegression(c, budgets.maxRegressionPct , budgets.regressionMetric ?? 'wallClock'),
|
|
196
|
+
budgets.regressionMetric ?? 'wallClock');
|
|
197
|
+
}
|
|
198
|
+
for (const { metric, maxPct } of budgets.regressionBudgets ?? []) {
|
|
199
|
+
pushComparisonBudget(results, comparison, 'max-regression', (c) => checkRegression(c, maxPct, metric), metric);
|
|
186
200
|
}
|
|
187
201
|
if (budgets.failOnIntroduced !== undefined) {
|
|
188
202
|
pushComparisonBudget(results, comparison, 'fail-on-introduced',
|
|
@@ -5,6 +5,7 @@ import { nodeParseCodecs } from './native-zstd.js';
|
|
|
5
5
|
import { createModelCallbacks } from '../model-assembler.js';
|
|
6
6
|
import { routeMessage, } from '../ingest.js';
|
|
7
7
|
|
|
8
|
+
import { mcpError } from '../mcp-error.js';
|
|
8
9
|
|
|
9
10
|
// No whole-file arrayBuffer(): the parser only ever reads bounded slices, and a
|
|
10
11
|
// whole-file read is what capped local logs at 2 GiB.
|
|
@@ -129,7 +130,10 @@ export function collectViaDispatch(
|
|
|
129
130
|
return new Promise((resolve, reject) => {
|
|
130
131
|
const handlers = {
|
|
131
132
|
...cb,
|
|
132
|
-
onDone: (msg ) =>
|
|
133
|
+
onDone: (msg ) => {
|
|
134
|
+
cb.onDone(msg); // records the parse gaps on the model, as the dashboard's ingest does
|
|
135
|
+
resolve({ appModel, skippedLines: (msg )?.skippedLines ?? 0 });
|
|
136
|
+
},
|
|
133
137
|
onError: (msg ) => reject(onDecodeError(msg)),
|
|
134
138
|
};
|
|
135
139
|
const emit = (msg ) => dispatch(msg, handlers);
|
|
@@ -149,10 +153,13 @@ export async function collectRun(inputPath )
|
|
|
149
153
|
return file;
|
|
150
154
|
};
|
|
151
155
|
try {
|
|
156
|
+
// A file or folder that isn't a decodable event log rejects with invalid-event-log, the code
|
|
157
|
+
// shs-load.ts gives an archive that fails to decode, so MCP reports both the same way. The
|
|
158
|
+
// CLI reads only the message.
|
|
152
159
|
return await collectViaDispatch((state, emit, reject) => {
|
|
153
160
|
if (stat.isDirectory()) {
|
|
154
161
|
if (!isRollingLogDirectory(inputPath)) {
|
|
155
|
-
reject(
|
|
162
|
+
reject(mcpError('invalid-event-log', "This isn't a Spark rolling event-log directory. Pass a single event-log file instead."));
|
|
156
163
|
return;
|
|
157
164
|
}
|
|
158
165
|
const names = readdirSync(inputPath);
|
|
@@ -160,7 +167,7 @@ export async function collectRun(inputPath )
|
|
|
160
167
|
try {
|
|
161
168
|
ordered = reassembleRollingEntries(names);
|
|
162
169
|
} catch (e) {
|
|
163
|
-
reject(e);
|
|
170
|
+
reject(mcpError('invalid-event-log', (e ).message));
|
|
164
171
|
return;
|
|
165
172
|
}
|
|
166
173
|
const files = ordered.map((name) => open(join(inputPath, name)));
|
|
@@ -170,7 +177,7 @@ export async function collectRun(inputPath )
|
|
|
170
177
|
} else {
|
|
171
178
|
runParse(open(inputPath), state, { emit, ...nodeParseCodecs }).catch(reject);
|
|
172
179
|
}
|
|
173
|
-
}, (msg) =>
|
|
180
|
+
}, (msg) => mcpError('invalid-event-log', (msg ).message));
|
|
174
181
|
} finally {
|
|
175
182
|
for (const file of opened) file.close();
|
|
176
183
|
}
|
|
@@ -0,0 +1,83 @@
|
|
|
1
|
+
// Parses the regression budgets a CLI caller lists: repeated `--regression-budget <metric>:<pct>`
|
|
2
|
+
// flags and a `--budgets <file.json>` file. Every failure throws an Error naming the problem, so
|
|
3
|
+
// the caller refuses to run (exit 2) instead of quietly dropping a budget the user meant to gate on.
|
|
4
|
+
import { readFileSync } from 'node:fs';
|
|
5
|
+
import { resolve } from 'node:path';
|
|
6
|
+
import { COMPARISON_METRIC_KEYS } from '../run-comparison.js';
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
// Plain non-negative decimals only: Number() would also accept '', ' ', '0x10' and '1e3'.
|
|
12
|
+
const PCT_PATTERN = /^\d+(\.\d+)?$/;
|
|
13
|
+
|
|
14
|
+
function assertKnownMetric(metric , origin ) {
|
|
15
|
+
if (!COMPARISON_METRIC_KEYS.includes(metric)) {
|
|
16
|
+
throw new Error(`${origin}: unknown metric "${metric}" (expected one of: ${COMPARISON_METRIC_KEYS.join(', ')}).`);
|
|
17
|
+
}
|
|
18
|
+
}
|
|
19
|
+
|
|
20
|
+
/** One `--regression-budget` value, `<metric>:<pct>`. */
|
|
21
|
+
export function parseRegressionBudgetFlag(spec ) {
|
|
22
|
+
const origin = `--regression-budget "${spec}"`;
|
|
23
|
+
const colon = spec.indexOf(':');
|
|
24
|
+
if (colon === -1) throw new Error(`${origin}: expected <metric>:<pct>.`);
|
|
25
|
+
const metric = spec.slice(0, colon);
|
|
26
|
+
const pct = spec.slice(colon + 1);
|
|
27
|
+
assertKnownMetric(metric, origin);
|
|
28
|
+
if (!PCT_PATTERN.test(pct)) throw new Error(`${origin}: "${pct}" is not a non-negative percentage.`);
|
|
29
|
+
return { metric, maxPct: Number(pct) };
|
|
30
|
+
}
|
|
31
|
+
|
|
32
|
+
/** The parsed `--budgets` JSON: `{"regression": {"<metric>": <pct>}}`. */
|
|
33
|
+
export function parseBudgetsFile(raw , origin ) {
|
|
34
|
+
if (raw === null || typeof raw !== 'object' || Array.isArray(raw)) {
|
|
35
|
+
throw new Error(`${origin}: expected a JSON object like {"regression": {"wallClock": 10}}.`);
|
|
36
|
+
}
|
|
37
|
+
const unknownKeys = Object.keys(raw).filter((k) => k !== 'regression');
|
|
38
|
+
if (unknownKeys.length > 0) {
|
|
39
|
+
throw new Error(`${origin}: unknown key ${unknownKeys.map((k) => `"${k}"`).join(', ')} (the only key is "regression").`);
|
|
40
|
+
}
|
|
41
|
+
const regression = (raw ).regression;
|
|
42
|
+
if (regression === null || typeof regression !== 'object' || Array.isArray(regression)) {
|
|
43
|
+
throw new Error(`${origin}: "regression" must be an object mapping a metric key to a percentage.`);
|
|
44
|
+
}
|
|
45
|
+
return Object.entries(regression).map(([metric, pct]) => {
|
|
46
|
+
assertKnownMetric(metric, origin);
|
|
47
|
+
if (typeof pct !== 'number' || !Number.isFinite(pct) || pct < 0) {
|
|
48
|
+
throw new Error(`${origin}: "${metric}" must be a non-negative number (a percentage), got ${JSON.stringify(pct)}.`);
|
|
49
|
+
}
|
|
50
|
+
return { metric, maxPct: pct };
|
|
51
|
+
});
|
|
52
|
+
}
|
|
53
|
+
|
|
54
|
+
export function loadBudgetsFile(path ) {
|
|
55
|
+
const fullPath = resolve(path);
|
|
56
|
+
let text ;
|
|
57
|
+
try {
|
|
58
|
+
text = readFileSync(fullPath, 'utf8');
|
|
59
|
+
} catch (e) {
|
|
60
|
+
throw new Error(`Cannot read budgets file ${fullPath}: ${(e ).message}`, { cause: e });
|
|
61
|
+
}
|
|
62
|
+
let raw ;
|
|
63
|
+
try {
|
|
64
|
+
raw = JSON.parse(text);
|
|
65
|
+
} catch (e) {
|
|
66
|
+
throw new Error(`Budgets file ${fullPath} is not valid JSON: ${(e ).message}`, { cause: e });
|
|
67
|
+
}
|
|
68
|
+
return parseBudgetsFile(raw, `Budgets file ${fullPath}`);
|
|
69
|
+
}
|
|
70
|
+
|
|
71
|
+
/** Checks that no metric is budgeted twice across the sources (the legacy
|
|
72
|
+
* `--max-regression-pct`/`--regression-metric` pair counts as one), and returns the budgets. */
|
|
73
|
+
export function combineRegressionBudgets(sources ) {
|
|
74
|
+
const seen = new Map ();
|
|
75
|
+
for (const { origin, budget } of sources) {
|
|
76
|
+
const first = seen.get(budget.metric);
|
|
77
|
+
if (first !== undefined) {
|
|
78
|
+
throw new Error(`Metric "${budget.metric}" has two regression budgets (${first} and ${origin}); give each metric one.`);
|
|
79
|
+
}
|
|
80
|
+
seen.set(budget.metric, origin);
|
|
81
|
+
}
|
|
82
|
+
return sources.map((s) => s.budget);
|
|
83
|
+
}
|
|
@@ -37,7 +37,7 @@ import { NEUTRAL_METRIC_KEYS, } from './run-comparison.js
|
|
|
37
37
|
|
|
38
38
|
|
|
39
39
|
|
|
40
|
-
/** A change under this share of
|
|
40
|
+
/** A change under this share of the baseline's value reads as "about the same", for
|
|
41
41
|
* run time and cost metrics alike: run-to-run noise on a shared cluster easily
|
|
42
42
|
* moves a job a percent or two. */
|
|
43
43
|
export const SAME_CHANGE_SHARE = 0.02;
|
|
@@ -78,14 +78,14 @@ function namesWhere(net , keep )
|
|
|
78
78
|
return [...net].filter(([, change]) => keep(change)).map(([name]) => name);
|
|
79
79
|
}
|
|
80
80
|
|
|
81
|
-
/** "
|
|
82
|
-
* job failed" / "All 3 of
|
|
83
|
-
function jobsFailed(run
|
|
84
|
-
if (failedJobs < totalJobs) return `
|
|
85
|
-
return totalJobs === 1 ? `
|
|
81
|
+
/** "The candidate had 2 of 5 jobs fail", or, when every job failed, "The candidate's only
|
|
82
|
+
* job failed" / "All 3 of the candidate's jobs failed" rather than "1 of 1 jobs". */
|
|
83
|
+
function jobsFailed(run , { failedJobs, totalJobs } ) {
|
|
84
|
+
if (failedJobs < totalJobs) return `The ${run} had ${failedJobs} of ${totalJobs} jobs fail`;
|
|
85
|
+
return totalJobs === 1 ? `The ${run}'s only job failed` : `All ${totalJobs} of the ${run}'s jobs failed`;
|
|
86
86
|
}
|
|
87
87
|
|
|
88
|
-
/**
|
|
88
|
+
/** The baseline's failures in the parenthesis after the candidate's. */
|
|
89
89
|
function baselineFailures({ failedJobs, totalJobs } ) {
|
|
90
90
|
if (failedJobs === 0) return 'none';
|
|
91
91
|
if (failedJobs < totalJobs) return `${failedJobs} of ${totalJobs}`;
|
|
@@ -93,18 +93,18 @@ function baselineFailures({ failedJobs, totalJobs } )
|
|
|
93
93
|
}
|
|
94
94
|
|
|
95
95
|
/** The headline when either run had failed jobs, as the run verdict leads
|
|
96
|
-
* with a failure: a faster
|
|
96
|
+
* with a failure: a faster candidate that dropped work is not an improvement,
|
|
97
97
|
* and with equal failure counts the tone stays neutral since a failing run
|
|
98
98
|
* that ends sooner may just have failed earlier. Null when both runs
|
|
99
99
|
* completed. */
|
|
100
100
|
function failureHeadline(base , cand ) {
|
|
101
101
|
if (base.failedJobs === 0 && cand.failedJobs === 0) return null;
|
|
102
102
|
const tone = cand.failedJobs > base.failedJobs ? 'worse' : cand.failedJobs < base.failedJobs ? 'better' : 'same';
|
|
103
|
-
if (cand.failedJobs === 0) return { title: `${jobsFailed('
|
|
104
|
-
return { title: `${jobsFailed('
|
|
103
|
+
if (cand.failedJobs === 0) return { title: `${jobsFailed('baseline', base)}; the candidate completed`, tone };
|
|
104
|
+
return { title: `${jobsFailed('candidate', cand)} (baseline: ${baselineFailures(base)})`, tone };
|
|
105
105
|
}
|
|
106
106
|
|
|
107
|
-
/** One plain answer to "did
|
|
107
|
+
/** One plain answer to "did the candidate get better or worse than the baseline", from the
|
|
108
108
|
* comparison's own whole-run metrics and finding-category tallies: run time
|
|
109
109
|
* first, then which cost metrics moved each way past run-to-run noise, then which finding
|
|
110
110
|
* categories appeared or went away. When either run had failed jobs, that
|
|
@@ -119,25 +119,25 @@ export function summarizeComparison(
|
|
|
119
119
|
jobs ,
|
|
120
120
|
) {
|
|
121
121
|
const wall = metrics.find((metric) => metric.key === 'wallClock');
|
|
122
|
-
let title = 'Run time could not be compared between
|
|
122
|
+
let title = 'Run time could not be compared between the baseline and the candidate';
|
|
123
123
|
let tone = 'unknown';
|
|
124
124
|
// A log with no end-of-run record stops where the run was cut off, so its
|
|
125
125
|
// shorter time is not a speed-up: say how much each log covers, neutrally.
|
|
126
|
-
const incompleteRuns = jobs ? (['
|
|
126
|
+
const incompleteRuns = jobs ? (['baseline', 'candidate'] ).filter((run) => jobs[run].incomplete) : [];
|
|
127
127
|
if (wall && wall.baseline != null && wall.baseline > 0 && wall.candidate != null && incompleteRuns.length > 0) {
|
|
128
128
|
const change = wall.candidate - wall.baseline;
|
|
129
129
|
title = Math.abs(change / wall.baseline) < SAME_CHANGE_SHARE
|
|
130
|
-
? "
|
|
131
|
-
: `
|
|
130
|
+
? "The candidate's log covers about as much run time as the baseline's"
|
|
131
|
+
: `The candidate's log covers ${formatDuration(Math.abs(change))} ${change < 0 ? 'less' : 'more'} run time than the baseline's`;
|
|
132
132
|
} else if (wall && wall.baseline != null && wall.baseline > 0 && wall.candidate != null) {
|
|
133
133
|
const change = wall.candidate - wall.baseline;
|
|
134
134
|
const share = change / wall.baseline;
|
|
135
135
|
if (Math.abs(share) < SAME_CHANGE_SHARE) {
|
|
136
|
-
title = '
|
|
136
|
+
title = 'The candidate took about as long as the baseline';
|
|
137
137
|
tone = 'same';
|
|
138
138
|
} else {
|
|
139
139
|
const faster = change < 0;
|
|
140
|
-
title = `
|
|
140
|
+
title = `The candidate finished ${formatDuration(Math.abs(change))} ${faster ? 'faster' : 'slower'} than the baseline (${Math.round(Math.abs(share) * 100)}%)`;
|
|
141
141
|
tone = faster ? 'better' : 'worse';
|
|
142
142
|
}
|
|
143
143
|
}
|
|
@@ -156,17 +156,17 @@ export function summarizeComparison(
|
|
|
156
156
|
if (incompleteRuns.length === 2) {
|
|
157
157
|
sentences.push('Neither log has an end-of-run record, so their times cover only what each log captured, not how long the runs took.');
|
|
158
158
|
} else if (incompleteRuns.length === 1) {
|
|
159
|
-
sentences.push(`
|
|
159
|
+
sentences.push(`The ${incompleteRuns[0]}'s log has no end-of-run record, so its time covers only what the log captured, not how long the run took.`);
|
|
160
160
|
}
|
|
161
|
-
if (worse.length > 0) sentences.push(`Worse in
|
|
162
|
-
if (better.length > 0) sentences.push(`Better in
|
|
161
|
+
if (worse.length > 0) sentences.push(`Worse in the candidate: ${worse.join(', ')}.`);
|
|
162
|
+
if (better.length > 0) sentences.push(`Better in the candidate: ${better.join(', ')}.`);
|
|
163
163
|
if (cost.length > 0 && worse.length === 0 && better.length === 0) sentences.push('Other measured cost metrics look about the same.');
|
|
164
164
|
|
|
165
165
|
const net = netByCategory(findings);
|
|
166
166
|
const introduced = namesWhere(net, (change) => change > 0);
|
|
167
167
|
const resolved = namesWhere(net, (change) => change < 0);
|
|
168
|
-
if (introduced.length > 0) sentences.push(`New or more frequent in
|
|
169
|
-
if (resolved.length > 0) sentences.push(`Less frequent in
|
|
168
|
+
if (introduced.length > 0) sentences.push(`New or more frequent in the candidate: ${introduced.join(', ')}.`);
|
|
169
|
+
if (resolved.length > 0) sentences.push(`Less frequent in the candidate: ${resolved.join(', ')}.`);
|
|
170
170
|
return { title, tone, sentences };
|
|
171
171
|
}
|
|
172
172
|
|
|
@@ -1 +1 @@
|
|
|
1
|
-
|
|
1
|
+
ab36c5fdcdc1a32a4e6206459e843e8d2267a90e6f1de29671827315d81bc374
|