sparkforensics-mcp 0.4.0 → 0.5.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/package.json +1 -1
- package/vendor-core/allocation.js +106 -0
- package/vendor-core/analyzer.js +13 -13
- package/vendor-core/cli/budgets.js +23 -9
- package/vendor-core/cli/collect-run.js +11 -4
- package/vendor-core/cli/regression-budgets.js +83 -0
- package/vendor-core/comparison-verdict.js +22 -22
- package/vendor-core/core-source-hash.txt +1 -1
- package/vendor-core/detectors.js +202 -68
- package/vendor-core/docs-content/detection/cache.md +3 -2
- package/vendor-core/docs-content/detection/cfg.md +9 -8
- package/vendor-core/docs-content/detection/chrn.md +1 -2
- package/vendor-core/docs-content/detection/cold.md +4 -2
- package/vendor-core/docs-content/detection/fail.md +3 -2
- package/vendor-core/docs-content/detection/gc.md +3 -2
- package/vendor-core/docs-content/detection/host.md +2 -1
- package/vendor-core/docs-content/detection/local.md +1 -1
- package/vendor-core/docs-content/detection/mem.md +5 -2
- package/vendor-core/docs-content/detection/plan.md +2 -1
- package/vendor-core/docs-content/detection/sfail.md +2 -1
- package/vendor-core/docs-content/detection/shape.md +5 -4
- package/vendor-core/docs-content/detection/skew.md +3 -1
- package/vendor-core/docs-content/detection/slow.md +2 -2
- package/vendor-core/docs-content/detection/spec.md +2 -3
- package/vendor-core/docs-content/detection/spill.md +1 -1
- package/vendor-core/docs-site-config.js +1 -1
- package/vendor-core/effective-conf.js +107 -0
- package/vendor-core/efficiency-model.js +8 -6
- package/vendor-core/event-handlers.js +160 -47
- package/vendor-core/event-schemas.js +2 -0
- package/vendor-core/evidence-report.js +18 -10
- package/vendor-core/finding-generic-recommendation.js +20 -1
- package/vendor-core/finding-names.js +7 -0
- package/vendor-core/finding-presentation.js +61 -26
- package/vendor-core/finding-tag-help.js +1 -1
- package/vendor-core/finding-types.js +12 -0
- package/vendor-core/format-utils.js +4 -3
- package/vendor-core/impact-estimator.js +20 -2
- package/vendor-core/impact-format.js +14 -13
- package/vendor-core/impact-model.js +27 -5
- package/vendor-core/ingest.js +4 -2
- package/vendor-core/list-runs.js +5 -2
- package/vendor-core/mcp-tools.js +1 -1
- package/vendor-core/model-assembler.js +23 -1
- package/vendor-core/parser-worker.js +1 -1
- package/vendor-core/proxy.js +3 -1
- package/vendor-core/python-stage.js +25 -0
- package/vendor-core/recommendation-rollup.js +16 -9
- package/vendor-core/redact.js +51 -10
- package/vendor-core/remediation.js +20 -0
- package/vendor-core/run-comparison.js +43 -24
- package/vendor-core/run-interpretation.js +2 -1
- package/vendor-core/run-metrics.js +198 -0
- package/vendor-core/run-totals.js +24 -0
- package/vendor-core/run-verdict.js +3 -4
- package/vendor-core/scorecard-estimates.js +1 -0
- package/vendor-core/session-snapshot.js +7 -0
- package/vendor-core/shs-schemas.js +2 -2
- package/vendor-core/spark-memory.js +17 -0
- package/vendor-core/stage-plan-nodes.js +18 -0
- package/vendor-core/stage-quantiles.js +4 -0
- package/vendor-core/types.js +49 -1
- package/vendor-core/wasted-core-hours.js +10 -7
- package/vendor-core/write-targets.js +312 -0
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
// Explicit .ts extensions: plain Node's ESM resolver (the runtime CLI/MCP path
|
|
2
2
|
// runs under) requires the exact specifier, unlike a bundler.
|
|
3
3
|
import { mergeIntervals } from './intervals.js';
|
|
4
|
-
import { formatWallClockRange, worstImpactBand, IMPACT_BAND_ORDER } from './format-utils.js';
|
|
4
|
+
import { formatRawWaste, formatWallClockRange, readsAsZero, worstImpactBand, IMPACT_BAND_ORDER } from './format-utils.js';
|
|
5
5
|
|
|
6
6
|
|
|
7
7
|
|
|
@@ -124,11 +124,15 @@ export function buildRecommendationRollup(
|
|
|
124
124
|
});
|
|
125
125
|
}
|
|
126
126
|
|
|
127
|
-
/** A group's
|
|
128
|
-
|
|
129
|
-
|
|
130
|
-
|
|
131
|
-
|
|
127
|
+
/** A resource group's summed waste as the board and CLI print it; null when it reads as zero. */
|
|
128
|
+
export function resourceGroupTotal(group ) {
|
|
129
|
+
const text = formatRawWaste({ value: group.total, unit: group.unit });
|
|
130
|
+
return readsAsZero(text) ? null : text;
|
|
131
|
+
}
|
|
132
|
+
|
|
133
|
+
/** A group's trailing figure on the Findings board, and its tooltip. Every kind leads with the
|
|
134
|
+
* "×N" finding count (the "worth expanding" signal). The row stays terse ("×2 · 476ms
|
|
135
|
+
* recoverable") to fit a dense right-aligned column; the title spells the shorthand out. */
|
|
132
136
|
export function rollupGroupStat(group ) {
|
|
133
137
|
if (group.kind === 'time') {
|
|
134
138
|
const recoverable = formatWallClockRange(group.recoverableMsHigh, group.recoverableMsHigh);
|
|
@@ -138,13 +142,16 @@ export function rollupGroupStat(group )
|
|
|
138
142
|
};
|
|
139
143
|
}
|
|
140
144
|
if (group.kind === 'resource') {
|
|
145
|
+
const total = resourceGroupTotal(group);
|
|
141
146
|
return {
|
|
142
|
-
stat: `×${group.findingCount} ·
|
|
143
|
-
statTitle: `${group.findingCount} findings of this type
|
|
147
|
+
stat: total ? `×${group.findingCount} · ${total}` : `×${group.findingCount}`,
|
|
148
|
+
statTitle: `${group.findingCount} findings of this type${total ? `; ${total} in total, a resource cost, not run time` : ''}`,
|
|
144
149
|
};
|
|
145
150
|
}
|
|
151
|
+
// The band heading above the row already names a one-band group's band.
|
|
152
|
+
const bands = Object.entries(group.byImpactBand);
|
|
146
153
|
return {
|
|
147
|
-
stat:
|
|
154
|
+
stat: bands.length === 1 ? `×${group.findingCount}` : `×${group.findingCount} · ${bands.map(([impactBand, count]) => `${count} ${impactBand}`).join(', ')}`,
|
|
148
155
|
statTitle: `${group.findingCount} findings of this type, by impact`,
|
|
149
156
|
};
|
|
150
157
|
}
|
package/vendor-core/redact.js
CHANGED
|
@@ -9,13 +9,12 @@
|
|
|
9
9
|
|
|
10
10
|
import { decodeCollections, encodeCollections, } from './export-data.js';
|
|
11
11
|
|
|
12
|
-
import { redactTaskFailureGroup, } from './task-failure.js';
|
|
12
|
+
import { REDACTED_TEXT, redactTaskFailureGroup, } from './task-failure.js';
|
|
13
13
|
|
|
14
14
|
// Host / IP identifier patterns. Used to enumerate host names that surface only
|
|
15
|
-
// inside free text: recommendation strings,
|
|
16
|
-
//
|
|
17
|
-
//
|
|
18
|
-
// pattern, keeping the scan idempotent.
|
|
15
|
+
// inside free text: recommendation strings, SQL relation/node names, never as a
|
|
16
|
+
// structured `host` field, so redaction reaches those residuals too. Pseudonyms
|
|
17
|
+
// (`host-1`) match neither pattern, keeping the scan idempotent.
|
|
19
18
|
const HOST_PATTERNS = [
|
|
20
19
|
// EC2-style ip-10-1-2-3 with an optional dotted domain (ip-10-1-2-3.ec2.internal).
|
|
21
20
|
// Each domain label must start with an alphanumeric, so a trailing sentence
|
|
@@ -95,6 +94,31 @@ function redactFailureGroups (node ) {
|
|
|
95
94
|
return node;
|
|
96
95
|
}
|
|
97
96
|
|
|
97
|
+
// A stage's failure reason is Spark's free-text message and can carry file paths and data values
|
|
98
|
+
// too, so it is replaced outright, like a failure group's message: a `stageFailed` finding's
|
|
99
|
+
// `valueText`, the `stageFailureReason` of a stage record, a job's `exception` (the run outcome's
|
|
100
|
+
// fallback reason), and the report's `failureReason` copy. Walking by key name, like
|
|
101
|
+
// collectHostFields, needs no path list. Returns a fresh tree.
|
|
102
|
+
const REASON_KEYS = new Set(['stageFailureReason', 'exception', 'failureReason']);
|
|
103
|
+
function redactStageFailureReasons (node ) {
|
|
104
|
+
if (Array.isArray(node)) return node.map((n) => redactStageFailureReasons(n)) ;
|
|
105
|
+
if (node && typeof node === 'object') {
|
|
106
|
+
const isStageFailed = (node ).type === 'stageFailed';
|
|
107
|
+
const out = {};
|
|
108
|
+
for (const [k, v] of Object.entries(node)) {
|
|
109
|
+
const isReason = REASON_KEYS.has(k) || (isStageFailed && k === 'valueText');
|
|
110
|
+
out[k] = isReason && typeof v === 'string' ? REDACTED_TEXT : redactStageFailureReasons(v);
|
|
111
|
+
}
|
|
112
|
+
return out ;
|
|
113
|
+
}
|
|
114
|
+
return node;
|
|
115
|
+
}
|
|
116
|
+
|
|
117
|
+
// Both passes for free text no host/app-id pattern recognizes.
|
|
118
|
+
function redactFailureText (node ) {
|
|
119
|
+
return redactStageFailureReasons(redactFailureGroups(node));
|
|
120
|
+
}
|
|
121
|
+
|
|
98
122
|
// Spark config keys ending in `host`/`hostname` (e.g. spark.driver.host,
|
|
99
123
|
// spark.yarn.am.hostname) carry plain FQDN host names that neither
|
|
100
124
|
// HOST_PATTERNS matches (no IP/EC2 shape) nor collectHostFields's by-key-name
|
|
@@ -122,7 +146,7 @@ function collectConfigHostValues(config , hos
|
|
|
122
146
|
|
|
123
147
|
|
|
124
148
|
|
|
125
|
-
|
|
149
|
+
|
|
126
150
|
|
|
127
151
|
|
|
128
152
|
|
|
@@ -188,10 +212,19 @@ function applyReplacements (node , ids
|
|
|
188
212
|
return deepReplace(node, merged) ;
|
|
189
213
|
}
|
|
190
214
|
|
|
215
|
+
// The app name identifies a job as much as its id, so a redacted name becomes the id's
|
|
216
|
+
// pseudonym, as list_runs does. Only the structured field is replaced, never substrings: a
|
|
217
|
+
// short name ("t", "etl") would otherwise corrupt every string it happens to occur in.
|
|
218
|
+
function withRedactedName (app ) {
|
|
219
|
+
return app.name == null ? app : { ...app, name: app.id ?? null };
|
|
220
|
+
}
|
|
221
|
+
|
|
191
222
|
export function redactReport (input ) {
|
|
192
|
-
const report =
|
|
223
|
+
const report = redactFailureText(input);
|
|
193
224
|
const { appIds, hosts } = collectIds(report);
|
|
194
|
-
|
|
225
|
+
const out = applyReplacements(report, { appIds, hosts });
|
|
226
|
+
const app = out.summary?.app;
|
|
227
|
+
return app ? { ...out, summary: { ...out.summary, app: withRedactedName(app) } } : out;
|
|
195
228
|
}
|
|
196
229
|
|
|
197
230
|
// Run-comparison counterpart: no single app-id *field* to pseudonymize
|
|
@@ -218,7 +251,7 @@ export function redactComparison (comparison ) {
|
|
|
218
251
|
|
|
219
252
|
|
|
220
253
|
function redactRunTree (input ) {
|
|
221
|
-
const data =
|
|
254
|
+
const data = redactFailureText(input);
|
|
222
255
|
const appIds = new Set ();
|
|
223
256
|
const hosts = new Set ();
|
|
224
257
|
const appId = data.app?.id;
|
|
@@ -229,7 +262,12 @@ function redactRunTree (input ) {
|
|
|
229
262
|
collectHostFields(data.configFindings, hosts);
|
|
230
263
|
collectConfigHostValues(data.app?.config, hosts);
|
|
231
264
|
scanTokens(data, [{ patterns: HOST_PATTERNS, out: hosts }, { patterns: APP_ID_PATTERNS, out: appIds }]);
|
|
232
|
-
|
|
265
|
+
const out = applyReplacements(data, { appIds, hosts });
|
|
266
|
+
if (!out.app) return out;
|
|
267
|
+
const app = withRedactedName(out.app);
|
|
268
|
+
// spark.app.name repeats the name in the config the HTML export ships.
|
|
269
|
+
if (app.config?.['spark.app.name'] == null) return { ...out, app };
|
|
270
|
+
return { ...out, app: { ...app, config: { ...app.config, 'spark.app.name': app.id ?? '' } } };
|
|
233
271
|
}
|
|
234
272
|
|
|
235
273
|
/** A run's model and findings with every identifier pseudonymized. Redact this before anything derives
|
|
@@ -264,6 +302,9 @@ export function redactRunModel(
|
|
|
264
302
|
executors: redacted.executors,
|
|
265
303
|
runAggregates: redacted.runAggregates ,
|
|
266
304
|
evidenceAvailability: redacted.evidenceAvailability ,
|
|
305
|
+
// Counts and execution ids, nothing to pseudonymize.
|
|
306
|
+
skippedLines: appModel.skippedLines,
|
|
307
|
+
unreadableSqlExecutions: appModel.unreadableSqlExecutions,
|
|
267
308
|
} ,
|
|
268
309
|
catalog: redacted.catalog,
|
|
269
310
|
configFindings: redacted.configFindings,
|
|
@@ -0,0 +1,20 @@
|
|
|
1
|
+
// Constructors for the structured `remediation` a detector attaches next to its prose
|
|
2
|
+
// `recommendation`: only where the detector already names the Spark property, never an invented one.
|
|
3
|
+
|
|
4
|
+
|
|
5
|
+
|
|
6
|
+
|
|
7
|
+
/** The property should be raised; `suggested` is the value when the detector computed one. */
|
|
8
|
+
export function increaseConf(key , suggested = null) {
|
|
9
|
+
return { kind: 'conf', key, direction: 'increase', suggested };
|
|
10
|
+
}
|
|
11
|
+
|
|
12
|
+
/** The property should be lowered; `suggested` is the value when the detector computed one. */
|
|
13
|
+
export function decreaseConf(key , suggested = null) {
|
|
14
|
+
return { kind: 'conf', key, direction: 'decrease', suggested };
|
|
15
|
+
}
|
|
16
|
+
|
|
17
|
+
/** The property should take a specific value (a switch or a class name), not move up or down. */
|
|
18
|
+
export function setConf(key , suggested = null) {
|
|
19
|
+
return { kind: 'conf', key, direction: 'set', suggested };
|
|
20
|
+
}
|
|
@@ -1,6 +1,9 @@
|
|
|
1
1
|
import { computeWallClock } from './wall-clock.js';
|
|
2
2
|
import { normalizeDetail } from './detectors.js';
|
|
3
3
|
import { cyrb53 } from './string-hash.js';
|
|
4
|
+
import { computeAllocation } from './allocation.js';
|
|
5
|
+
import { totalExecutorCpuMs, withEarlierAttempts } from './run-totals.js';
|
|
6
|
+
import { planNodesOfStage } from './stage-plan-nodes.js';
|
|
4
7
|
import { captureSnapshot } from './session-snapshot.js';
|
|
5
8
|
import { tunedRunNote } from './threshold-overrides.js';
|
|
6
9
|
|
|
@@ -76,23 +79,18 @@ function planTreeIdentity(root ) {
|
|
|
76
79
|
// Falls back to the coarser whole-tree identity when the stage has no
|
|
77
80
|
// attributed nodes (hand-built snapshots without `stageIds`, or unmatched
|
|
78
81
|
// accumulables).
|
|
79
|
-
function sqlNodeIdentity(stage , snapshot
|
|
82
|
+
function sqlNodeIdentity(stage , snapshot ) {
|
|
80
83
|
const execId = stage.sqlExecutionId;
|
|
81
84
|
if (execId == null) return '';
|
|
82
85
|
const root = snapshot.sql.get(execId)?.planTree ?? null;
|
|
83
86
|
if (!root) return '';
|
|
84
|
-
const fingerprints
|
|
85
|
-
|
|
86
|
-
if (node.stageIds?.includes(stage.id)) {
|
|
87
|
-
fingerprints.push(JSON.stringify([normalizeStageName(node.name ?? ''), normalizeDetail(node.detail ?? '')]));
|
|
88
|
-
}
|
|
89
|
-
for (const child of node.children ?? []) collect(child);
|
|
90
|
-
})(root);
|
|
87
|
+
const fingerprints = planNodesOfStage(stage, snapshot.sql)
|
|
88
|
+
.map((node) => JSON.stringify([normalizeStageName(node.name ?? ''), normalizeDetail(node.detail ?? '')]));
|
|
91
89
|
if (fingerprints.length === 0) return planTreeIdentity(root) ?? '';
|
|
92
90
|
return cyrb53(JSON.stringify(fingerprints.sort()));
|
|
93
91
|
}
|
|
94
92
|
|
|
95
|
-
export function stageIdentity(stage , snapshot
|
|
93
|
+
export function stageIdentity(stage , snapshot ) {
|
|
96
94
|
return normalizeStageName(stage.name ?? '') + '§' + sqlNodeIdentity(stage, snapshot);
|
|
97
95
|
}
|
|
98
96
|
|
|
@@ -199,6 +197,7 @@ function skewRatios(stages ) {
|
|
|
199
197
|
export const COMPARISON_METRIC_KEYS = [
|
|
200
198
|
'wallClock', 'shuffleSpill', 'taskSkew', 'failedTaskRate', 'diskSpill', 'gcTime',
|
|
201
199
|
'inputBytes', 'outputBytes', 'executorRunTime', 'taskCount', 'executorsAdded',
|
|
200
|
+
'executorCpuTime', 'allocatedCoreHours',
|
|
202
201
|
];
|
|
203
202
|
|
|
204
203
|
// Volume/count metrics, not cost metrics: more or less input/output data, or
|
|
@@ -228,20 +227,24 @@ function metric(
|
|
|
228
227
|
|
|
229
228
|
export function metricDeltas(baseSnap , candSnap ) {
|
|
230
229
|
const out = [];
|
|
230
|
+
// Run totals count every task attempt, failed and speculative ones too (run-totals.ts), the
|
|
231
|
+
// same sums the CLI metrics block reports; skew stays on each stage's latest attempt.
|
|
232
|
+
const attemptsOf = (snap ) => withEarlierAttempts([...snap.stages.values()]);
|
|
231
233
|
|
|
232
234
|
// Wall-clock: always computable (computeWallClock tolerates a null app).
|
|
233
235
|
out.push(metric('wallClock', 'Wall-clock duration',
|
|
234
236
|
computeWallClock(baseSnap.app, baseSnap.stages).total,
|
|
235
237
|
computeWallClock(candSnap.app, candSnap.stages).total));
|
|
236
238
|
|
|
237
|
-
//
|
|
238
|
-
// matching is unreliable on real logs
|
|
239
|
-
//
|
|
240
|
-
|
|
241
|
-
const
|
|
242
|
-
|
|
239
|
+
// Memory spill (Spark's memoryBytesSpilled) over the whole run: a sum needs no
|
|
240
|
+
// stage matching, and matching is unreliable on real logs, so scope it to all
|
|
241
|
+
// stages exactly like task-skew and failed-rate below. The key stays
|
|
242
|
+
// `shuffleSpill` so existing --regression-metric callers keep working.
|
|
243
|
+
const bSpill = sumField(attemptsOf(baseSnap), 'memoryBytesSpilled');
|
|
244
|
+
const cSpill = sumField(attemptsOf(candSnap), 'memoryBytesSpilled');
|
|
245
|
+
out.push(metric('shuffleSpill', 'Memory spill',
|
|
243
246
|
bSpill.present ? bSpill.sum : null, cSpill.present ? cSpill.sum : null,
|
|
244
|
-
{ unavailableReason: bSpill.present && cSpill.present ? undefined : 'No
|
|
247
|
+
{ unavailableReason: bSpill.present && cSpill.present ? undefined : 'No memory-spill data recorded for a run' }));
|
|
245
248
|
|
|
246
249
|
// Task skew: p95 of per-stage ratio across the whole run.
|
|
247
250
|
const bSkew = p95(skewRatios([...baseSnap.stages.values()]));
|
|
@@ -250,10 +253,10 @@ export function metricDeltas(baseSnap , candSnap
|
|
|
250
253
|
{ unavailableReason: bSkew != null && cSkew != null ? undefined : 'No stage had measurable duration for a run' }));
|
|
251
254
|
|
|
252
255
|
// Failed-task rate: Σ failedTasks / Σ taskCount.
|
|
253
|
-
const bTasks = sumField(
|
|
254
|
-
const cTasks = sumField(
|
|
255
|
-
const bFailed = sumField(
|
|
256
|
-
const cFailed = sumField(
|
|
256
|
+
const bTasks = sumField(attemptsOf(baseSnap), 'taskCount');
|
|
257
|
+
const cTasks = sumField(attemptsOf(candSnap), 'taskCount');
|
|
258
|
+
const bFailed = sumField(attemptsOf(baseSnap), 'failedTasks');
|
|
259
|
+
const cFailed = sumField(attemptsOf(candSnap), 'failedTasks');
|
|
257
260
|
// Guard on BOTH inputs: a missing `failedTasks` field must render Unavailable,
|
|
258
261
|
// not a false 0% rate (dividing an absent-and-therefore-0 numerator).
|
|
259
262
|
const bRate = bTasks.present && bTasks.sum > 0 && bFailed.present ? bFailed.sum / bTasks.sum : null;
|
|
@@ -265,8 +268,8 @@ export function metricDeltas(baseSnap , candSnap
|
|
|
265
268
|
// carries (set in finalizeStage). Correct at any match coverage, like the
|
|
266
269
|
// sums above; no parser or detector change.
|
|
267
270
|
const sumMetric = (key , label , field , reason ) => {
|
|
268
|
-
const b = sumField(
|
|
269
|
-
const c = sumField(
|
|
271
|
+
const b = sumField(attemptsOf(baseSnap), field);
|
|
272
|
+
const c = sumField(attemptsOf(candSnap), field);
|
|
270
273
|
out.push(metric(key, label, b.present ? b.sum : null, c.present ? c.sum : null,
|
|
271
274
|
{ unavailableReason: b.present && c.present ? undefined : reason }));
|
|
272
275
|
};
|
|
@@ -285,6 +288,18 @@ export function metricDeltas(baseSnap , candSnap
|
|
|
285
288
|
out.push(metric('executorsAdded', 'Executors added', bExec, cExec,
|
|
286
289
|
{ unavailableReason: bExec != null && cExec != null ? undefined : 'No executor events recorded for a run' }));
|
|
287
290
|
|
|
291
|
+
// Executor CPU time (ms) and allocated core-hours cost resources, so less is better. CPU time is
|
|
292
|
+
// null, not 0, on a run whose log never recorded it (older Spark).
|
|
293
|
+
const cpuMs = (snap ) => totalExecutorCpuMs(attemptsOf(snap));
|
|
294
|
+
const bCpu = cpuMs(baseSnap), cCpu = cpuMs(candSnap);
|
|
295
|
+
out.push(metric('executorCpuTime', 'Executor CPU time', bCpu, cCpu,
|
|
296
|
+
{ unavailableReason: bCpu != null && cCpu != null ? undefined : 'No executor CPU time recorded for a run' }));
|
|
297
|
+
const coreHours = (snap ) =>
|
|
298
|
+
(snap.executors ? computeAllocation(snap).coreHours : null);
|
|
299
|
+
const bCore = coreHours(baseSnap), cCore = coreHours(candSnap);
|
|
300
|
+
out.push(metric('allocatedCoreHours', 'Allocated core-hours', bCore, cCore,
|
|
301
|
+
{ unavailableReason: bCore != null && cCore != null ? undefined : 'No executor lifecycle or core count recorded for a run' }));
|
|
302
|
+
|
|
288
303
|
return out;
|
|
289
304
|
}
|
|
290
305
|
|
|
@@ -467,8 +482,12 @@ export function renderComparisonMarkdown(
|
|
|
467
482
|
const lines = ['', '## Comparison to baseline', ''];
|
|
468
483
|
if (tuned) lines.push(`- Tuned thresholds (both runs): ${tunedRunNote(tuned)}`, '');
|
|
469
484
|
if (verdict) {
|
|
470
|
-
//
|
|
471
|
-
|
|
485
|
+
// Names each run by its label (MCP's run IDs), unless the labels are just the role names the
|
|
486
|
+
// CLI passes, where the verdict below already says baseline and candidate.
|
|
487
|
+
if (comparison.baselineLabel !== 'baseline' || comparison.candidateLabel !== 'candidate') {
|
|
488
|
+
lines.push(`Baseline: ${comparison.baselineLabel} · Candidate: ${comparison.candidateLabel}`, '');
|
|
489
|
+
}
|
|
490
|
+
lines.push(verdict.title);
|
|
472
491
|
if (verdict.sentences.length > 0) lines.push('', verdict.sentences.join(' '));
|
|
473
492
|
lines.push('');
|
|
474
493
|
}
|
|
@@ -192,6 +192,7 @@ function interpretEfficiency(appModel ) {
|
|
|
192
192
|
app: appModel.app,
|
|
193
193
|
stages: appModel.stages,
|
|
194
194
|
executorsAdded: appModel.executors.added,
|
|
195
|
+
executorsRemoved: appModel.executors.removed,
|
|
195
196
|
runAggregates: appModel.runAggregates,
|
|
196
197
|
});
|
|
197
198
|
}
|
|
@@ -270,7 +271,7 @@ export function interpretRun(appModel , catalog , configFindi
|
|
|
270
271
|
coverage: interpretCoverage(appModel, allFindings),
|
|
271
272
|
runShape: interpretRunShape(appModel),
|
|
272
273
|
wallClock: computeWallClock(appModel.app, appModel.stages),
|
|
273
|
-
wastedCoreHours: computeWastedCoreHours(appModel.app, appModel.executors.added, appModel.runAggregates),
|
|
274
|
+
wastedCoreHours: computeWastedCoreHours(appModel.app, appModel.executors.added, appModel.runAggregates, appModel.executors.removed),
|
|
274
275
|
efficiency: interpretEfficiency(appModel),
|
|
275
276
|
wallClockReliable: checkConcurrentJobGroups(appModel.jobs).wallClockReliable,
|
|
276
277
|
etlPhases: attributeEtlPhases(appModel.stages),
|
|
@@ -0,0 +1,198 @@
|
|
|
1
|
+
// The machine-readable metrics block of the CLI's JSON output: run-level totals and per-stage rows
|
|
2
|
+
// an automated tuning loop can compare across runs without reading a report. Its schemaVersion
|
|
3
|
+
// moves independently of the evidence report's. A figure the log cannot provide is null, never 0.
|
|
4
|
+
import { computeAllocation, } from './allocation.js';
|
|
5
|
+
import { computeSkewRatio, ENTRY_BY_TYPE, } from './detectors.js';
|
|
6
|
+
import { totalExecutorCpuMs, withEarlierAttempts } from './run-totals.js';
|
|
7
|
+
import { isPythonStage } from './python-stage.js';
|
|
8
|
+
import { stageIdentity } from './run-comparison.js';
|
|
9
|
+
import { hasCompleteApplicationInterval } from './scorecard-estimates.js';
|
|
10
|
+
import { effectiveThresholds } from './threshold-overrides.js';
|
|
11
|
+
import { computeWallClock } from './wall-clock.js';
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
export const METRICS_SCHEMA_VERSION = 1;
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
const finite = (v ) => typeof v === 'number' && Number.isFinite(v);
|
|
73
|
+
|
|
74
|
+
// Σ of a field over the stages that carry a finite value; null when none does.
|
|
75
|
+
function sumOf(stages , pick ) {
|
|
76
|
+
let sum = 0, present = false;
|
|
77
|
+
for (const s of stages) {
|
|
78
|
+
const v = pick(s);
|
|
79
|
+
if (finite(v)) { sum += v; present = true; }
|
|
80
|
+
}
|
|
81
|
+
return present ? sum : null;
|
|
82
|
+
}
|
|
83
|
+
|
|
84
|
+
// 0 means either no execution memory used or not recorded; only a positive peak is reported.
|
|
85
|
+
function peakExecutionMemory(stages ) {
|
|
86
|
+
const peaks = stages.map((s) => s.peakExecutionMemoryMax).filter((v) => finite(v) && v > 0);
|
|
87
|
+
return peaks.length > 0 ? Math.max(...peaks) : null;
|
|
88
|
+
}
|
|
89
|
+
|
|
90
|
+
function maxSkew(stages , minTasksForP95 ) {
|
|
91
|
+
const ratios = stages
|
|
92
|
+
.map((s) => computeSkewRatio(s, minTasksForP95)?.ratio)
|
|
93
|
+
.filter((r) => finite(r));
|
|
94
|
+
return ratios.length > 0 ? Math.max(...ratios) : null;
|
|
95
|
+
}
|
|
96
|
+
|
|
97
|
+
// Failed stage attempts and stages submitted more than once, from the parser's per-stage attempt
|
|
98
|
+
// counts; null when any stage record lacks them.
|
|
99
|
+
function stageAttempts(stages ) {
|
|
100
|
+
let failed = 0, retried = 0;
|
|
101
|
+
for (const s of stages) {
|
|
102
|
+
if (!finite(s.stageAttempts) || !finite(s.failedStageAttempts)) return null;
|
|
103
|
+
failed += s.failedStageAttempts;
|
|
104
|
+
if (s.stageAttempts > 1) retried++;
|
|
105
|
+
}
|
|
106
|
+
return { failed, retried };
|
|
107
|
+
}
|
|
108
|
+
|
|
109
|
+
// Metrics of a set of stages, every attempt included; a row folds the stages sharing one
|
|
110
|
+
// fingerprint. Skew describes the latest attempt's task durations.
|
|
111
|
+
function stageMetrics(stages , minTasksForP95 ) {
|
|
112
|
+
const attempts = withEarlierAttempts(stages);
|
|
113
|
+
// Task-derived figures are null for stages that never finished (no task records).
|
|
114
|
+
// Superseded attempts carry work but no task count of their own, so they stay in once any
|
|
115
|
+
// attempt finished a task.
|
|
116
|
+
const finished = attempts.some((s) => (s.taskCount ?? 0) > 0) ? attempts : [];
|
|
117
|
+
const tasks = sumOf(attempts, (s) => s.taskCount);
|
|
118
|
+
const fromTasks = (pick ) => (finished.length > 0 ? sumOf(finished, pick) : null);
|
|
119
|
+
const durations = attempts.filter((s) => finite(s.submittedAt) && finite(s.completedAt) && (s.completedAt ) >= (s.submittedAt ));
|
|
120
|
+
return {
|
|
121
|
+
durationMs: durations.length > 0 ? sumOf(durations, (s) => (s.completedAt ) - (s.submittedAt )) : null,
|
|
122
|
+
executorCpuTimeMs: totalExecutorCpuMs(finished),
|
|
123
|
+
executorRunTimeMs: fromTasks((s) => s.executorRunTime),
|
|
124
|
+
gcTimeMs: fromTasks((s) => s.jvmGCTime),
|
|
125
|
+
memorySpillBytes: fromTasks((s) => s.memoryBytesSpilled),
|
|
126
|
+
diskSpillBytes: fromTasks((s) => s.diskBytesSpilled),
|
|
127
|
+
shuffleReadBytes: fromTasks((s) => s.shuffleReadBytes),
|
|
128
|
+
shuffleWriteBytes: fromTasks((s) => s.shuffleWriteBytes),
|
|
129
|
+
inputBytes: fromTasks((s) => s.inputBytes),
|
|
130
|
+
outputBytes: fromTasks((s) => s.outputBytes),
|
|
131
|
+
outputRows: sumOf(finished, (s) => s.outputRecords),
|
|
132
|
+
peakExecutionMemoryBytes: peakExecutionMemory(finished),
|
|
133
|
+
taskCount: tasks != null && tasks > 0 ? tasks : null,
|
|
134
|
+
failedTasks: fromTasks((s) => s.failedTasks),
|
|
135
|
+
retriedTasks: fromTasks((s) => (s.wastedAttempts ) ?? 0),
|
|
136
|
+
skew: maxSkew(stages.filter((s) => (s.taskCount ?? 0) > 0), minTasksForP95),
|
|
137
|
+
};
|
|
138
|
+
}
|
|
139
|
+
|
|
140
|
+
/** The run's metrics block. `thresholds` only moves the skew ratio's P95 cutoff, as for the
|
|
141
|
+
* --max-skew budget. */
|
|
142
|
+
export function computeRunMetrics(appModel , thresholds ) {
|
|
143
|
+
const stageList = [...appModel.stages.values()];
|
|
144
|
+
const minTasksForP95 = effectiveThresholds(ENTRY_BY_TYPE.get('skew') , thresholds).minTasksForP95 ;
|
|
145
|
+
const all = stageMetrics(stageList, minTasksForP95);
|
|
146
|
+
const python = stageList.filter((s) => isPythonStage(s, appModel.sql));
|
|
147
|
+
const pythonRunTimeMs = sumOf(withEarlierAttempts(python), (s) => s.executorRunTime);
|
|
148
|
+
|
|
149
|
+
const byFingerprint = new Map ();
|
|
150
|
+
for (const s of stageList) {
|
|
151
|
+
const key = stageIdentity(s, appModel);
|
|
152
|
+
const group = byFingerprint.get(key);
|
|
153
|
+
if (group) group.push(s); else byFingerprint.set(key, [s]);
|
|
154
|
+
}
|
|
155
|
+
const stages = {};
|
|
156
|
+
for (const [key, group] of byFingerprint) {
|
|
157
|
+
const attempts = stageAttempts(group);
|
|
158
|
+
stages[key] = {
|
|
159
|
+
stageIds: group.map((s) => s.id).sort((a, b) => a - b),
|
|
160
|
+
failed: attempts ? attempts.failed > 0 : null,
|
|
161
|
+
retried: attempts ? attempts.retried > 0 : null,
|
|
162
|
+
python: group.some((s) => isPythonStage(s, appModel.sql)),
|
|
163
|
+
...stageMetrics(group, minTasksForP95),
|
|
164
|
+
};
|
|
165
|
+
}
|
|
166
|
+
|
|
167
|
+
const attempts = stageAttempts(stageList);
|
|
168
|
+
return {
|
|
169
|
+
schemaVersion: METRICS_SCHEMA_VERSION,
|
|
170
|
+
runComplete: appModel.app?.endTime != null,
|
|
171
|
+
time: {
|
|
172
|
+
wallClockMs: hasCompleteApplicationInterval(appModel.app) ? computeWallClock(appModel.app, appModel.stages).total : null,
|
|
173
|
+
executorCpuTimeMs: all.executorCpuTimeMs,
|
|
174
|
+
executorRunTimeMs: all.executorRunTimeMs,
|
|
175
|
+
gcTimeMs: all.gcTimeMs,
|
|
176
|
+
},
|
|
177
|
+
data: {
|
|
178
|
+
memorySpillBytes: all.memorySpillBytes, diskSpillBytes: all.diskSpillBytes,
|
|
179
|
+
shuffleReadBytes: all.shuffleReadBytes, shuffleWriteBytes: all.shuffleWriteBytes,
|
|
180
|
+
inputBytes: all.inputBytes, outputBytes: all.outputBytes, outputRows: all.outputRows,
|
|
181
|
+
peakExecutionMemoryBytes: all.peakExecutionMemoryBytes,
|
|
182
|
+
},
|
|
183
|
+
shape: {
|
|
184
|
+
taskCount: all.taskCount,
|
|
185
|
+
stageCount: stageList.length,
|
|
186
|
+
failedStageAttempts: attempts?.failed ?? null,
|
|
187
|
+
retriedStages: attempts?.retried ?? null,
|
|
188
|
+
failedTasks: all.failedTasks,
|
|
189
|
+
retriedTasks: all.retriedTasks,
|
|
190
|
+
maxSkew: all.skew,
|
|
191
|
+
},
|
|
192
|
+
allocation: computeAllocation(appModel),
|
|
193
|
+
python: {
|
|
194
|
+
shareOfTaskRunTime: all.executorRunTimeMs != null && all.executorRunTimeMs > 0 ? (pythonRunTimeMs ?? 0) / all.executorRunTimeMs : null,
|
|
195
|
+
},
|
|
196
|
+
stages,
|
|
197
|
+
};
|
|
198
|
+
}
|
|
@@ -0,0 +1,24 @@
|
|
|
1
|
+
// The run-level sums every surface reads: the CLI metrics block and the run comparison (so the
|
|
2
|
+
// dashboard's comparison view and the MCP compare_runs tool) share these, so one name never means
|
|
3
|
+
// two figures.
|
|
4
|
+
import { nsToMs } from './format-utils.js';
|
|
5
|
+
|
|
6
|
+
|
|
7
|
+
/** Each stage followed by the work its own figures leave out, shaped like a stage: the attempts a
|
|
8
|
+
* resubmit replaced, and task attempts the stage record dropped (a failed attempt's late tasks,
|
|
9
|
+
* failed retries, losing speculative copies). Summing the result counts every attempt's work,
|
|
10
|
+
* failed ones too. A stage's own record, which the detectors and per-stage views read, is left as
|
|
11
|
+
* it is. */
|
|
12
|
+
export function withEarlierAttempts(stages ) {
|
|
13
|
+
return stages.flatMap((s) => [s, ...[s.earlierAttempts, s.lateAttemptWork]
|
|
14
|
+
.filter((work) => work != null)
|
|
15
|
+
.map(({ durationMs, ...totals }) => ({ id: s.id, ...totals, submittedAt: 0, completedAt: durationMs ?? undefined }))]);
|
|
16
|
+
}
|
|
17
|
+
|
|
18
|
+
/** Summed executor CPU time in ms. Spark records it in nanoseconds and the parser reads an absent
|
|
19
|
+
* metric as 0, so stages that all read 0 never recorded it (older Spark): null, not 0. */
|
|
20
|
+
export function totalExecutorCpuMs(stages ) {
|
|
21
|
+
let ns = 0;
|
|
22
|
+
for (const s of stages) if (typeof s.executorCpuTime === 'number' && Number.isFinite(s.executorCpuTime) && s.executorCpuTime > 0) ns += s.executorCpuTime;
|
|
23
|
+
return ns > 0 ? nsToMs(ns) : null;
|
|
24
|
+
}
|
|
@@ -236,13 +236,12 @@ export function verdictSummary(eligible , steps , facts
|
|
|
236
236
|
} else {
|
|
237
237
|
sentences.push(`${plural(eligible.length, 'finding')} in ${plural(steps.length, 'place')}.`);
|
|
238
238
|
const wallClock = steps[0].lead.impactEstimate?.wallClock;
|
|
239
|
-
|
|
240
|
-
|
|
241
|
-
} else if (wallClock && facts.runMs != null) {
|
|
239
|
+
// An idle-capacity lead needs no sentence here: the title gives the idle share and step 1 the fix.
|
|
240
|
+
if (!isIdleCapacityStep(steps[0]) && wallClock && facts.runMs != null) {
|
|
242
241
|
sentences.push(`The first fix could save up to ${formatDuration(wallClock.high)} of this ${formatDuration(facts.runMs)} run.`);
|
|
243
242
|
}
|
|
244
243
|
if (steps.some((step) => step.related.length > 0)) {
|
|
245
|
-
sentences.push('Findings
|
|
244
|
+
sentences.push('Findings on one stage are grouped, and their savings overlap.');
|
|
246
245
|
}
|
|
247
246
|
}
|
|
248
247
|
if (facts.incomplete) {
|
|
@@ -24,6 +24,8 @@ import { isSupportedEvidenceAvailability } from './evidence-availability.js';
|
|
|
24
24
|
|
|
25
25
|
|
|
26
26
|
|
|
27
|
+
|
|
28
|
+
|
|
27
29
|
|
|
28
30
|
|
|
29
31
|
|
|
@@ -44,6 +46,8 @@ export function captureSnapshot(
|
|
|
44
46
|
jobs: new Map(appModel.jobs),
|
|
45
47
|
runAggregates: appModel.runAggregates,
|
|
46
48
|
evidenceAvailability: appModel.evidenceAvailability,
|
|
49
|
+
skippedLines: appModel.skippedLines,
|
|
50
|
+
unreadableSqlExecutions: appModel.unreadableSqlExecutions && [...appModel.unreadableSqlExecutions],
|
|
47
51
|
catalog: [...catalog],
|
|
48
52
|
taskData: new Map(taskDataCache),
|
|
49
53
|
};
|
|
@@ -71,6 +75,9 @@ export function applySnapshot(
|
|
|
71
75
|
appModel.evidenceAvailability = isSupportedEvidenceAvailability(snapshot.evidenceAvailability)
|
|
72
76
|
? snapshot.evidenceAvailability
|
|
73
77
|
: null;
|
|
78
|
+
appModel.skippedLines = snapshot.skippedLines;
|
|
79
|
+
if (snapshot.unreadableSqlExecutions) appModel.unreadableSqlExecutions = [...snapshot.unreadableSqlExecutions];
|
|
80
|
+
else delete appModel.unreadableSqlExecutions;
|
|
74
81
|
|
|
75
82
|
taskDataCache.clear();
|
|
76
83
|
for (const [k, v] of snapshot.taskData) taskDataCache.set(k, v);
|
|
@@ -1,8 +1,8 @@
|
|
|
1
1
|
import { z } from 'zod';
|
|
2
2
|
|
|
3
3
|
// Proxy-level error envelope: the JSON body the local server's SHS proxy
|
|
4
|
-
// (./
|
|
5
|
-
// `{ code: '
|
|
4
|
+
// (./proxy.js) sends back on a non-OK upstream response, e.g.
|
|
5
|
+
// `{ code: 'upstream-unreachable' }`. This is NOT a SparkListener* event shape, so
|
|
6
6
|
// it lives here rather than in event-schemas.ts. `.passthrough()` since the
|
|
7
7
|
// proxy may attach extra debugging fields the consumer doesn't care about;
|
|
8
8
|
// only `code` is read.
|
|
@@ -0,0 +1,17 @@
|
|
|
1
|
+
// Parse a Spark memory-size string to MiB. Spark's JVM-memory configs use bytesConf(ByteUnit.MiB),
|
|
2
|
+
// so a bare number means MiB. A k/m/g/t suffix sets the unit (trailing "b" redundant); a lone "b"
|
|
3
|
+
// ("10b") means bytes.
|
|
4
|
+
export function parseSparkMemoryMB(value ) {
|
|
5
|
+
if (value == null) return null;
|
|
6
|
+
const m = String(value).trim().toLowerCase().match(/^([\d.]+)\s*([kmgt]?)(b?)$/);
|
|
7
|
+
if (!m) return null;
|
|
8
|
+
const n = parseFloat(m[1]);
|
|
9
|
+
if (!Number.isFinite(n)) return null;
|
|
10
|
+
switch (m[2]) {
|
|
11
|
+
case 'k': return Math.round(n / 1024);
|
|
12
|
+
case 'g': return Math.round(n * 1024);
|
|
13
|
+
case 't': return Math.round(n * 1024 * 1024);
|
|
14
|
+
case 'm': return Math.round(n);
|
|
15
|
+
default: return m[3] === 'b' ? Math.round(n / (1024 * 1024)) : Math.round(n);
|
|
16
|
+
}
|
|
17
|
+
}
|