sparkforensics-mcp 0.2.4 → 0.4.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -0
- package/README.md +7 -1
- package/bin/sparkforensics-mcp.mjs +41 -10
- package/package.json +1 -1
- package/vendor-core/analyzer.js +156 -48
- package/vendor-core/check-coverage.js +88 -0
- package/vendor-core/cli/budgets.js +31 -18
- package/vendor-core/cli/collect-run.js +76 -31
- package/vendor-core/cli/native-zstd.js +2 -2
- package/vendor-core/cli/threshold-config.js +28 -0
- package/vendor-core/comparison-verdict.js +177 -0
- package/vendor-core/core-source-hash.txt +1 -0
- package/vendor-core/core-usage-locality.js +56 -2
- package/vendor-core/detector-docs.js +58 -0
- package/vendor-core/detectors.js +933 -459
- package/vendor-core/docs-config.js +0 -36
- package/vendor-core/docs-content/chapters/nav-index.json +31 -0
- package/vendor-core/docs-content/detection/cstor.md +9 -0
- package/vendor-core/docs-content/detection/fail.md +6 -2
- package/vendor-core/docs-site-config.js +3 -0
- package/vendor-core/event-handlers.js +191 -6
- package/vendor-core/event-schemas.js +29 -0
- package/vendor-core/evidence-report.js +440 -112
- package/vendor-core/export-data.js +79 -6
- package/vendor-core/finding-action-label.js +9 -88
- package/vendor-core/finding-filter-predicate.js +9 -0
- package/vendor-core/finding-generic-recommendation.js +6 -104
- package/vendor-core/finding-names.js +21 -45
- package/vendor-core/finding-presentation.js +333 -0
- package/vendor-core/finding-tag-help.js +110 -0
- package/vendor-core/finding-types.js +361 -0
- package/vendor-core/findings-of-type.js +11 -0
- package/vendor-core/format-utils.js +92 -27
- package/vendor-core/html-export.js +51 -0
- package/vendor-core/impact-band.js +21 -8
- package/vendor-core/impact-estimator.js +8 -521
- package/vendor-core/impact-format.js +114 -0
- package/vendor-core/impact-model.js +175 -0
- package/vendor-core/ingest.js +2 -0
- package/vendor-core/intervals.js +13 -0
- package/vendor-core/list-runs.js +2 -3
- package/vendor-core/load-vendored.js +70 -5
- package/vendor-core/mcp-server-factory.js +14 -10
- package/vendor-core/mcp-tools.js +105 -45
- package/vendor-core/model-assembler.js +12 -0
- package/vendor-core/occupancy.js +1 -1
- package/vendor-core/parser-worker.js +22 -5
- package/vendor-core/plan-graph-model.js +3 -2
- package/vendor-core/plan-node-detail.js +1 -1
- package/vendor-core/recommendation-rollup.js +63 -3
- package/vendor-core/redact.js +68 -28
- package/vendor-core/run-comparison.js +40 -7
- package/vendor-core/run-interpretation.js +290 -0
- package/vendor-core/run-outcome.js +74 -0
- package/vendor-core/run-payload.js +17 -0
- package/vendor-core/run-shape.js +40 -0
- package/vendor-core/run-verdict.js +353 -0
- package/vendor-core/scaling-sim.js +4 -5
- package/vendor-core/scorecard-estimates.js +62 -0
- package/vendor-core/shs-fetch.js +175 -65
- package/vendor-core/shs-load.js +1 -1
- package/vendor-core/sql-stages.js +11 -0
- package/vendor-core/stage-quantiles.js +14 -0
- package/vendor-core/task-failure.js +151 -0
- package/vendor-core/threshold-overrides.js +160 -0
- package/vendor-core/threshold-summary.js +11 -33
- package/vendor-core/types.js +6 -42
- package/vendor-core/vendor/fflate.js +1 -1
- package/vendor-core/wall-clock.js +1 -12
- package/vendor-core/wasted-core-hours.js +2 -2
- package/vendor-core/zip-archive.js +167 -0
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import {
|
|
1
|
+
import { readdirSync, statSync, openSync, readSync, closeSync } from 'node:fs';
|
|
2
2
|
import { join, basename } from 'node:path';
|
|
3
3
|
import { createState, runParse, runParseFiles, reassembleRollingEntries } from '../parser-worker.js';
|
|
4
4
|
import { nodeParseCodecs } from './native-zstd.js';
|
|
@@ -6,10 +6,13 @@ import { createModelCallbacks } from '../model-assembler.js';
|
|
|
6
6
|
import { routeMessage, } from '../ingest.js';
|
|
7
7
|
|
|
8
8
|
|
|
9
|
+
// No whole-file arrayBuffer(): the parser only ever reads bounded slices, and a
|
|
10
|
+
// whole-file read is what capped local logs at 2 GiB.
|
|
9
11
|
|
|
10
12
|
|
|
11
13
|
|
|
12
|
-
|
|
14
|
+
|
|
15
|
+
|
|
13
16
|
|
|
14
17
|
|
|
15
18
|
export function emptyAppModel() {
|
|
@@ -24,18 +27,48 @@ export function emptyAppModel() {
|
|
|
24
27
|
};
|
|
25
28
|
}
|
|
26
29
|
|
|
27
|
-
// File-like shape runParse/runParseFiles need
|
|
30
|
+
// File-like shape runParse/runParseFiles need (name, size, slice().arrayBuffer()), read with
|
|
31
|
+
// positioned readSync so only the requested slice is ever in memory, whatever the file size.
|
|
32
|
+
// The descriptor opens on the first read and closes once a read reaches the end of the file,
|
|
33
|
+
// so a rolling directory's parts hold at most one open descriptor at a time while streaming.
|
|
34
|
+
// A later read (a zip archive reads its tail first) reopens it; callers must still call
|
|
35
|
+
// close() once parsing settles, to cover reads that stopped early on an error.
|
|
28
36
|
export function nodeFileFromPath(path ) {
|
|
29
|
-
const
|
|
30
|
-
const
|
|
37
|
+
const name = basename(path);
|
|
38
|
+
const size = statSync(path).size;
|
|
39
|
+
let fd = null;
|
|
40
|
+
const close = () => {
|
|
41
|
+
if (fd === null) return;
|
|
42
|
+
const open = fd;
|
|
43
|
+
fd = null;
|
|
44
|
+
closeSync(open);
|
|
45
|
+
};
|
|
31
46
|
return {
|
|
32
|
-
name
|
|
33
|
-
size
|
|
47
|
+
name,
|
|
48
|
+
size,
|
|
34
49
|
slice(start , end ) {
|
|
35
|
-
|
|
36
|
-
|
|
50
|
+
return {
|
|
51
|
+
async arrayBuffer() {
|
|
52
|
+
const from = Math.max(0, start);
|
|
53
|
+
const to = Math.min(end, size);
|
|
54
|
+
if (to <= from) return new ArrayBuffer(0);
|
|
55
|
+
const buf = new Uint8Array(to - from);
|
|
56
|
+
fd ??= openSync(path, 'r');
|
|
57
|
+
// readSync may return fewer bytes than asked; loop until the slice is full.
|
|
58
|
+
for (let filled = 0; filled < buf.length;) {
|
|
59
|
+
const n = readSync(fd, buf, filled, buf.length - filled, from + filled);
|
|
60
|
+
if (n === 0) {
|
|
61
|
+
close();
|
|
62
|
+
throw new Error(`"${name}" shrank while it was being read`);
|
|
63
|
+
}
|
|
64
|
+
filled += n;
|
|
65
|
+
}
|
|
66
|
+
if (to === size) close();
|
|
67
|
+
return buf.buffer;
|
|
68
|
+
},
|
|
69
|
+
};
|
|
37
70
|
},
|
|
38
|
-
|
|
71
|
+
close,
|
|
39
72
|
};
|
|
40
73
|
}
|
|
41
74
|
|
|
@@ -107,26 +140,38 @@ export function collectViaDispatch(
|
|
|
107
140
|
|
|
108
141
|
export async function collectRun(inputPath ) {
|
|
109
142
|
const stat = statSync(inputPath);
|
|
110
|
-
|
|
111
|
-
|
|
112
|
-
|
|
113
|
-
|
|
114
|
-
|
|
115
|
-
|
|
116
|
-
|
|
117
|
-
|
|
118
|
-
|
|
119
|
-
|
|
120
|
-
|
|
121
|
-
|
|
122
|
-
|
|
143
|
+
// Every file opened for this run, closed once parsing settles on any path (done, parse
|
|
144
|
+
// error, decode error, or a throw past the parser's guards).
|
|
145
|
+
const opened = [];
|
|
146
|
+
const open = (path ) => {
|
|
147
|
+
const file = nodeFileFromPath(path);
|
|
148
|
+
opened.push(file);
|
|
149
|
+
return file;
|
|
150
|
+
};
|
|
151
|
+
try {
|
|
152
|
+
return await collectViaDispatch((state, emit, reject) => {
|
|
153
|
+
if (stat.isDirectory()) {
|
|
154
|
+
if (!isRollingLogDirectory(inputPath)) {
|
|
155
|
+
reject(new Error("This isn't a Spark rolling event-log directory. Pass a single event-log file instead."));
|
|
156
|
+
return;
|
|
157
|
+
}
|
|
158
|
+
const names = readdirSync(inputPath);
|
|
159
|
+
let ordered;
|
|
160
|
+
try {
|
|
161
|
+
ordered = reassembleRollingEntries(names);
|
|
162
|
+
} catch (e) {
|
|
163
|
+
reject(e);
|
|
164
|
+
return;
|
|
165
|
+
}
|
|
166
|
+
const files = ordered.map((name) => open(join(inputPath, name)));
|
|
167
|
+
// .catch(reject), not void: a throw past the parser's guards would otherwise leave
|
|
168
|
+
// this Promise pending forever, surfacing only as an unhandled rejection.
|
|
169
|
+
runParseFiles(files, state, { emit, ...nodeParseCodecs }).catch(reject);
|
|
170
|
+
} else {
|
|
171
|
+
runParse(open(inputPath), state, { emit, ...nodeParseCodecs }).catch(reject);
|
|
123
172
|
}
|
|
124
|
-
|
|
125
|
-
|
|
126
|
-
|
|
127
|
-
|
|
128
|
-
} else {
|
|
129
|
-
runParse(nodeFileFromPath(inputPath), state, { emit, ...nodeParseCodecs }).catch(reject);
|
|
130
|
-
}
|
|
131
|
-
}, (msg) => new Error((msg ).message));
|
|
173
|
+
}, (msg) => new Error((msg ).message));
|
|
174
|
+
} finally {
|
|
175
|
+
for (const file of opened) file.close();
|
|
176
|
+
}
|
|
132
177
|
}
|
|
@@ -343,8 +343,8 @@ export function createThreadedZstdDecoder(
|
|
|
343
343
|
}
|
|
344
344
|
|
|
345
345
|
// Node's native zstd where this Node has it (22.15+/23.8+); older Nodes keep the vendored fzstd.
|
|
346
|
-
// runParse/runParseFiles (collectRun) take the threaded decoder
|
|
347
|
-
//
|
|
346
|
+
// runParse/runParseFiles (collectRun) take the threaded decoder; the SHS archive loader
|
|
347
|
+
// (decodeShsArchive over an in-memory download) takes the inline one.
|
|
348
348
|
export const nodeParseCodecs =
|
|
349
349
|
nativeZstdAvailable ? { zstdDecoder: createThreadedZstdDecoder } : {};
|
|
350
350
|
export const nodeArchiveCodecs =
|
|
@@ -0,0 +1,28 @@
|
|
|
1
|
+
// Reads a --thresholds config file for the CLI and the MCP server. Every failure throws an Error
|
|
2
|
+
// naming the file and the problem, so the caller refuses to run instead of quietly falling back
|
|
3
|
+
// to the defaults the user meant to change.
|
|
4
|
+
import { readFileSync } from 'node:fs';
|
|
5
|
+
import { resolve } from 'node:path';
|
|
6
|
+
import { parseThresholdOverrides } from '../threshold-overrides.js';
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
export function loadThresholdOverrides(path ) {
|
|
10
|
+
const fullPath = resolve(path);
|
|
11
|
+
let text ;
|
|
12
|
+
try {
|
|
13
|
+
text = readFileSync(fullPath, 'utf8');
|
|
14
|
+
} catch (e) {
|
|
15
|
+
throw new Error(`Cannot read thresholds file ${fullPath}: ${(e ).message}`, { cause: e });
|
|
16
|
+
}
|
|
17
|
+
let raw ;
|
|
18
|
+
try {
|
|
19
|
+
raw = JSON.parse(text);
|
|
20
|
+
} catch (e) {
|
|
21
|
+
throw new Error(`Thresholds file ${fullPath} is not valid JSON: ${(e ).message}`, { cause: e });
|
|
22
|
+
}
|
|
23
|
+
try {
|
|
24
|
+
return parseThresholdOverrides(raw);
|
|
25
|
+
} catch (e) {
|
|
26
|
+
throw new Error(`Thresholds file ${fullPath}: ${(e ).message}`, { cause: e });
|
|
27
|
+
}
|
|
28
|
+
}
|
|
@@ -0,0 +1,177 @@
|
|
|
1
|
+
import { formatDuration, typeTag } from './format-utils.js';
|
|
2
|
+
import { TAG_HELP } from './finding-tag-help.js';
|
|
3
|
+
import { NEUTRAL_METRIC_KEYS, } from './run-comparison.js';
|
|
4
|
+
|
|
5
|
+
/** The slice of a comparison metric row the verdict reads. */
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
/** The slice of a finding-category delta the verdict reads. */
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
/** How many of a run's ended jobs failed, counted as `summarizeRunOutcome`
|
|
23
|
+
* counts them. */
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
/** A change under this share of run A's value reads as "about the same", for
|
|
41
|
+
* run time and cost metrics alike: run-to-run noise on a shared cluster easily
|
|
42
|
+
* moves a job a percent or two. */
|
|
43
|
+
export const SAME_CHANGE_SHARE = 0.02;
|
|
44
|
+
|
|
45
|
+
function measured(metric ) {
|
|
46
|
+
return metric.baseline != null && metric.candidate != null;
|
|
47
|
+
}
|
|
48
|
+
|
|
49
|
+
function movedPastNoise(metric ) {
|
|
50
|
+
if (metric.baseline === 0) return metric.candidate !== 0;
|
|
51
|
+
return Math.abs(metric.candidate - metric.baseline) / Math.abs(metric.baseline) >= SAME_CHANGE_SHARE;
|
|
52
|
+
}
|
|
53
|
+
|
|
54
|
+
/** Plain name for a finding type ("Memory and disk spill"), falling back to
|
|
55
|
+
* its tag when the tag has no help entry. */
|
|
56
|
+
function categoryName(type ) {
|
|
57
|
+
const tag = typeTag(type);
|
|
58
|
+
return TAG_HELP[tag]?.expansion ?? tag;
|
|
59
|
+
}
|
|
60
|
+
|
|
61
|
+
/** Net count change per displayed category name. Categories are tallied per
|
|
62
|
+
* (rule, impact band), and several rules share one name (every Plan Advisor
|
|
63
|
+
* type reads "Plan advisor", every stage-shape sub-rule "Stage shape"), so a
|
|
64
|
+
* rule whose findings moved from critical to warning, or two rules under one
|
|
65
|
+
* name moving opposite ways, would otherwise
|
|
66
|
+
* read as both "introduced" and "resolved". Order follows first appearance,
|
|
67
|
+
* introduced before resolved. */
|
|
68
|
+
function netByCategory(findings ) {
|
|
69
|
+
const net = new Map ();
|
|
70
|
+
for (const item of [...findings.introduced, ...findings.resolved]) {
|
|
71
|
+
const name = categoryName(item.type);
|
|
72
|
+
net.set(name, (net.get(name) ?? 0) + item.candCount - item.baseCount);
|
|
73
|
+
}
|
|
74
|
+
return net;
|
|
75
|
+
}
|
|
76
|
+
|
|
77
|
+
function namesWhere(net , keep ) {
|
|
78
|
+
return [...net].filter(([, change]) => keep(change)).map(([name]) => name);
|
|
79
|
+
}
|
|
80
|
+
|
|
81
|
+
/** "Run B had 2 of 5 jobs fail", or, when every job failed, "Run B's only
|
|
82
|
+
* job failed" / "All 3 of run B's jobs failed" rather than "1 of 1 jobs". */
|
|
83
|
+
function jobsFailed(run , { failedJobs, totalJobs } ) {
|
|
84
|
+
if (failedJobs < totalJobs) return `Run ${run} had ${failedJobs} of ${totalJobs} jobs fail`;
|
|
85
|
+
return totalJobs === 1 ? `Run ${run}'s only job failed` : `All ${totalJobs} of run ${run}'s jobs failed`;
|
|
86
|
+
}
|
|
87
|
+
|
|
88
|
+
/** Run A's failures in the parenthesis after run B's. */
|
|
89
|
+
function baselineFailures({ failedJobs, totalJobs } ) {
|
|
90
|
+
if (failedJobs === 0) return 'none';
|
|
91
|
+
if (failedJobs < totalJobs) return `${failedJobs} of ${totalJobs}`;
|
|
92
|
+
return totalJobs === 1 ? 'its only job failed' : `all ${totalJobs} failed`;
|
|
93
|
+
}
|
|
94
|
+
|
|
95
|
+
/** The headline when either run had failed jobs, as the run verdict leads
|
|
96
|
+
* with a failure: a faster run B that dropped work is not an improvement,
|
|
97
|
+
* and with equal failure counts the tone stays neutral since a failing run
|
|
98
|
+
* that ends sooner may just have failed earlier. Null when both runs
|
|
99
|
+
* completed. */
|
|
100
|
+
function failureHeadline(base , cand ) {
|
|
101
|
+
if (base.failedJobs === 0 && cand.failedJobs === 0) return null;
|
|
102
|
+
const tone = cand.failedJobs > base.failedJobs ? 'worse' : cand.failedJobs < base.failedJobs ? 'better' : 'same';
|
|
103
|
+
if (cand.failedJobs === 0) return { title: `${jobsFailed('A', base)}; run B completed`, tone };
|
|
104
|
+
return { title: `${jobsFailed('B', cand)} (run A: ${baselineFailures(base)})`, tone };
|
|
105
|
+
}
|
|
106
|
+
|
|
107
|
+
/** One plain answer to "did run B get better or worse than run A", from the
|
|
108
|
+
* comparison's own whole-run metrics and finding-category tallies: run time
|
|
109
|
+
* first, then which cost metrics moved each way past run-to-run noise, then which finding
|
|
110
|
+
* categories appeared or went away. When either run had failed jobs, that
|
|
111
|
+
* leads instead and run time becomes the first sentence. When either log is
|
|
112
|
+
* incomplete, run time is stated as what each log covers, never as faster or
|
|
113
|
+
* slower, and the tone stays neutral. Volume and count metrics (input, output,
|
|
114
|
+
* tasks, executors) are left out because more or less of them is not
|
|
115
|
+
* inherently better or worse. */
|
|
116
|
+
export function summarizeComparison(
|
|
117
|
+
metrics ,
|
|
118
|
+
findings ,
|
|
119
|
+
jobs ,
|
|
120
|
+
) {
|
|
121
|
+
const wall = metrics.find((metric) => metric.key === 'wallClock');
|
|
122
|
+
let title = 'Run time could not be compared between run A and run B';
|
|
123
|
+
let tone = 'unknown';
|
|
124
|
+
// A log with no end-of-run record stops where the run was cut off, so its
|
|
125
|
+
// shorter time is not a speed-up: say how much each log covers, neutrally.
|
|
126
|
+
const incompleteRuns = jobs ? (['A', 'B'] ).filter((run) => (run === 'A' ? jobs.baseline : jobs.candidate).incomplete) : [];
|
|
127
|
+
if (wall && wall.baseline != null && wall.baseline > 0 && wall.candidate != null && incompleteRuns.length > 0) {
|
|
128
|
+
const change = wall.candidate - wall.baseline;
|
|
129
|
+
title = Math.abs(change / wall.baseline) < SAME_CHANGE_SHARE
|
|
130
|
+
? "Run B's log covers about as much run time as run A's"
|
|
131
|
+
: `Run B's log covers ${formatDuration(Math.abs(change))} ${change < 0 ? 'less' : 'more'} run time than run A's`;
|
|
132
|
+
} else if (wall && wall.baseline != null && wall.baseline > 0 && wall.candidate != null) {
|
|
133
|
+
const change = wall.candidate - wall.baseline;
|
|
134
|
+
const share = change / wall.baseline;
|
|
135
|
+
if (Math.abs(share) < SAME_CHANGE_SHARE) {
|
|
136
|
+
title = 'Run B took about as long as run A';
|
|
137
|
+
tone = 'same';
|
|
138
|
+
} else {
|
|
139
|
+
const faster = change < 0;
|
|
140
|
+
title = `Run B finished ${formatDuration(Math.abs(change))} ${faster ? 'faster' : 'slower'} than run A (${Math.round(Math.abs(share) * 100)}%)`;
|
|
141
|
+
tone = faster ? 'better' : 'worse';
|
|
142
|
+
}
|
|
143
|
+
}
|
|
144
|
+
|
|
145
|
+
const cost = metrics.filter((metric) => metric.key !== 'wallClock' && !NEUTRAL_METRIC_KEYS.has(metric.key)).filter(measured);
|
|
146
|
+
const moved = cost.filter(movedPastNoise);
|
|
147
|
+
const worse = moved.filter((metric) => metric.direction === 'regression').map((metric) => metric.label);
|
|
148
|
+
const better = moved.filter((metric) => metric.direction === 'improvement').map((metric) => metric.label);
|
|
149
|
+
const sentences = [];
|
|
150
|
+
const failure = jobs ? failureHeadline(jobs.baseline, jobs.candidate) : null;
|
|
151
|
+
if (failure) {
|
|
152
|
+
sentences.push(`${title}.`);
|
|
153
|
+
title = failure.title;
|
|
154
|
+
tone = failure.tone;
|
|
155
|
+
}
|
|
156
|
+
if (incompleteRuns.length === 2) {
|
|
157
|
+
sentences.push('Neither log has an end-of-run record, so their times cover only what each log captured, not how long the runs took.');
|
|
158
|
+
} else if (incompleteRuns.length === 1) {
|
|
159
|
+
sentences.push(`Run ${incompleteRuns[0]}'s log has no end-of-run record, so its time covers only what the log captured, not how long the run took.`);
|
|
160
|
+
}
|
|
161
|
+
if (worse.length > 0) sentences.push(`Worse in run B: ${worse.join(', ')}.`);
|
|
162
|
+
if (better.length > 0) sentences.push(`Better in run B: ${better.join(', ')}.`);
|
|
163
|
+
if (cost.length > 0 && worse.length === 0 && better.length === 0) sentences.push('Other measured cost metrics look about the same.');
|
|
164
|
+
|
|
165
|
+
const net = netByCategory(findings);
|
|
166
|
+
const introduced = namesWhere(net, (change) => change > 0);
|
|
167
|
+
const resolved = namesWhere(net, (change) => change < 0);
|
|
168
|
+
if (introduced.length > 0) sentences.push(`New or more frequent in run B: ${introduced.join(', ')}.`);
|
|
169
|
+
if (resolved.length > 0) sentences.push(`Less frequent in run B: ${resolved.join(', ')}.`);
|
|
170
|
+
return { title, tone, sentences };
|
|
171
|
+
}
|
|
172
|
+
|
|
173
|
+
/** The comparison verdict for a core comparison result: the one the dashboard's comparison page
|
|
174
|
+
* leads with, and the one the CLI's --baseline report and MCP compare_runs carry. */
|
|
175
|
+
export function comparisonVerdict(result ) {
|
|
176
|
+
return summarizeComparison(result.metrics, result.findings, result.jobOutcomes);
|
|
177
|
+
}
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
b5b0e0b6178129de4de482d4b59fbf881822655d8911b374c4a60a33b38f63fb
|
|
@@ -18,6 +18,8 @@ export const LOCALITY_TIERS = ['PROCESS_LOCAL', 'NODE_LOCAL', 'RACK_LO
|
|
|
18
18
|
|
|
19
19
|
|
|
20
20
|
|
|
21
|
+
|
|
22
|
+
|
|
21
23
|
|
|
22
24
|
|
|
23
25
|
export function computeLocalityAreaSeries(
|
|
@@ -25,7 +27,7 @@ export function computeLocalityAreaSeries(
|
|
|
25
27
|
{ bucketWidthMs = 60_000, tiers = LOCALITY_TIERS } = {},
|
|
26
28
|
) {
|
|
27
29
|
const valid = stages.filter(s => (s.completedAt ?? 0) > (s.submittedAt ?? 0) && (s.executorRunTime ?? 0) > 0);
|
|
28
|
-
if (valid.length === 0) return { labels: [], series: {} };
|
|
30
|
+
if (valid.length === 0) return { labels: [], series: {}, endTime: 0 };
|
|
29
31
|
const startTime = valid.reduce((m, s) => Math.min(m, s.submittedAt ?? m), Infinity);
|
|
30
32
|
const endTime = valid.reduce((m, s) => Math.max(m, s.completedAt ?? m), -Infinity);
|
|
31
33
|
const nBuckets = Math.max(1, Math.ceil((endTime - startTime) / bucketWidthMs));
|
|
@@ -65,5 +67,57 @@ export function computeLocalityAreaSeries(
|
|
|
65
67
|
|
|
66
68
|
const labels = [];
|
|
67
69
|
for (let b = 0; b < nBuckets; b++) labels.push(startTime + b * bucketWidthMs);
|
|
68
|
-
return { labels, series };
|
|
70
|
+
return { labels, series, endTime };
|
|
69
71
|
}
|
|
72
|
+
|
|
73
|
+
/** How many time buckets the Core Usage by Locality chart aims for across a run. */
|
|
74
|
+
export const LOCALITY_CHART_TARGET_BUCKETS = 60;
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
|
|
83
|
+
|
|
84
|
+
|
|
85
|
+
|
|
86
|
+
/** The Core Usage by Locality chart's points and its "busy at the peak" figure, shared by the
|
|
87
|
+
* dashboard widget and the CLI/MCP run summary. Buckets are at least a minute wide; a last bucket
|
|
88
|
+
* the stages only partly cover is rescaled to the covered part, and each bucket's idle cores are
|
|
89
|
+
* the peak minus its busy cores. */
|
|
90
|
+
export function buildLocalityChart(
|
|
91
|
+
stages ,
|
|
92
|
+
app ,
|
|
93
|
+
targetBuckets = LOCALITY_CHART_TARGET_BUCKETS,
|
|
94
|
+
) {
|
|
95
|
+
const hasActivity = stages.some((s) => (s.executorRunTime ?? 0) > 0 && (s.completedAt ?? 0) > (s.submittedAt ?? 0));
|
|
96
|
+
if (!hasActivity) return { hasActivity: false };
|
|
97
|
+
|
|
98
|
+
const start = app?.startTime ?? 0;
|
|
99
|
+
const end = app?.endTime ?? start;
|
|
100
|
+
const bucketWidthMs = Math.max(60_000, Math.ceil(Math.max(1, end - start) / targetBuckets));
|
|
101
|
+
const { labels, series, endTime: seriesEnd } = computeLocalityAreaSeries(stages, { bucketWidthMs });
|
|
102
|
+
const order = [...LOCALITY_TIERS.filter((t) => series[t]), ...(series.OTHER ? ['OTHER'] : []), 'idle'];
|
|
103
|
+
|
|
104
|
+
const points = labels.map((t, i) => {
|
|
105
|
+
const point = { t: Math.round((t - start) / 1000) };
|
|
106
|
+
// The series averages each bucket over its full width, so a last bucket
|
|
107
|
+
// that runs past the series' own end (or a run shorter than one bucket)
|
|
108
|
+
// reads diluted: a 10s stage in a 60s bucket showed well under its busy
|
|
109
|
+
// cores. Rescale to the part of the bucket the series actually covers.
|
|
110
|
+
const coveredMs = Math.min(bucketWidthMs, seriesEnd - t);
|
|
111
|
+
const scale = bucketWidthMs / coveredMs;
|
|
112
|
+
for (const tier of order) if (tier !== 'idle') point[tier] = (series[tier]?.[i] ?? 0) * scale;
|
|
113
|
+
return point;
|
|
114
|
+
});
|
|
115
|
+
const busyTiers = order.filter((t) => t !== 'idle');
|
|
116
|
+
const busyTotal = (p ) => busyTiers.reduce((sum, t) => sum + p[t], 0);
|
|
117
|
+
const peakCores = points.reduce((max, p) => Math.max(max, busyTotal(p)), 0);
|
|
118
|
+
// Same rule as the core series (peak busy minus each bucket's busy), redone
|
|
119
|
+
// on the rescaled values so the stack still tops out at the peak.
|
|
120
|
+
for (const p of points) p.idle = Math.max(0, peakCores - busyTotal(p));
|
|
121
|
+
return { hasActivity: true, order, points, peakCores };
|
|
122
|
+
}
|
|
123
|
+
|
|
@@ -0,0 +1,58 @@
|
|
|
1
|
+
// Type-level doc anchors, read off the detector catalog. Kept out of docs-config.ts so the docs
|
|
2
|
+
// URL helpers (used by every renderer) don't pull in the detectors themselves.
|
|
3
|
+
import { DETECTORS, ENTRY_BY_TYPE } from './detectors.js';
|
|
4
|
+
import { isKnownDocAnchor } from './docs-config.js';
|
|
5
|
+
import { getThresholdSummary } from './threshold-summary.js';
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
// DETECTORS is static, so this grouping is built once (lazily) instead of re-scanning per
|
|
9
|
+
// docAnchorForType call (called once per TagBadge per render). Keyed by emitted finding type, so
|
|
10
|
+
// broadcastSizing's anchor lands on underBroadcast/overBroadcast.
|
|
11
|
+
let anchorsByTypeCache ;
|
|
12
|
+
|
|
13
|
+
function anchorsByType() {
|
|
14
|
+
if (!anchorsByTypeCache) {
|
|
15
|
+
anchorsByTypeCache = new Map();
|
|
16
|
+
for (const entry of DETECTORS ) {
|
|
17
|
+
for (const type of entry.emits) {
|
|
18
|
+
const anchors = anchorsByTypeCache.get(type) ?? new Set();
|
|
19
|
+
anchors.add(entry.docAnchor);
|
|
20
|
+
anchorsByTypeCache.set(type, anchors);
|
|
21
|
+
}
|
|
22
|
+
}
|
|
23
|
+
}
|
|
24
|
+
return anchorsByTypeCache;
|
|
25
|
+
}
|
|
26
|
+
|
|
27
|
+
/** Resolves a finding `type` to its documented anchor from the DETECTORS entries that emit it.
|
|
28
|
+
* Returns undefined when entries sharing the type disagree on docAnchor (only configAudit today),
|
|
29
|
+
* or when the resolved anchor isn't in the allowlist (isKnownDocAnchor, the same gate DocsLink uses). */
|
|
30
|
+
export function docAnchorForType(type ) {
|
|
31
|
+
const anchors = anchorsByType().get(type) ?? new Set();
|
|
32
|
+
if (anchors.size !== 1) return undefined;
|
|
33
|
+
const [anchor] = anchors;
|
|
34
|
+
return anchor && isKnownDocAnchor(anchor) ? anchor : undefined;
|
|
35
|
+
}
|
|
36
|
+
|
|
37
|
+
/** What a renderer shows about one finding type without running its detector. */
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
/** Every emitted finding type's `DetectorInfo`, in `DETECTORS` declaration order of each entry's
|
|
50
|
+
* `emits` list (a type repeated across entries takes its first position, as `ENTRY_BY_TYPE` does).
|
|
51
|
+
* The key order is the widget order's tie-break, so it is part of the result. */
|
|
52
|
+
export function detectorInfoByType() {
|
|
53
|
+
const info = {};
|
|
54
|
+
for (const [type, { order, scope }] of ENTRY_BY_TYPE) {
|
|
55
|
+
info[type] = { order, docAnchor: docAnchorForType(type) ?? null, thresholdSummary: getThresholdSummary(type), scope };
|
|
56
|
+
}
|
|
57
|
+
return info;
|
|
58
|
+
}
|