sparkforensics-mcp 0.2.4 → 0.4.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -0
- package/README.md +7 -1
- package/bin/sparkforensics-mcp.mjs +41 -10
- package/package.json +1 -1
- package/vendor-core/analyzer.js +156 -48
- package/vendor-core/check-coverage.js +88 -0
- package/vendor-core/cli/budgets.js +31 -18
- package/vendor-core/cli/collect-run.js +76 -31
- package/vendor-core/cli/native-zstd.js +2 -2
- package/vendor-core/cli/threshold-config.js +28 -0
- package/vendor-core/comparison-verdict.js +177 -0
- package/vendor-core/core-source-hash.txt +1 -0
- package/vendor-core/core-usage-locality.js +56 -2
- package/vendor-core/detector-docs.js +58 -0
- package/vendor-core/detectors.js +933 -459
- package/vendor-core/docs-config.js +0 -36
- package/vendor-core/docs-content/chapters/nav-index.json +31 -0
- package/vendor-core/docs-content/detection/cstor.md +9 -0
- package/vendor-core/docs-content/detection/fail.md +6 -2
- package/vendor-core/docs-site-config.js +3 -0
- package/vendor-core/event-handlers.js +191 -6
- package/vendor-core/event-schemas.js +29 -0
- package/vendor-core/evidence-report.js +440 -112
- package/vendor-core/export-data.js +79 -6
- package/vendor-core/finding-action-label.js +9 -88
- package/vendor-core/finding-filter-predicate.js +9 -0
- package/vendor-core/finding-generic-recommendation.js +6 -104
- package/vendor-core/finding-names.js +21 -45
- package/vendor-core/finding-presentation.js +333 -0
- package/vendor-core/finding-tag-help.js +110 -0
- package/vendor-core/finding-types.js +361 -0
- package/vendor-core/findings-of-type.js +11 -0
- package/vendor-core/format-utils.js +92 -27
- package/vendor-core/html-export.js +51 -0
- package/vendor-core/impact-band.js +21 -8
- package/vendor-core/impact-estimator.js +8 -521
- package/vendor-core/impact-format.js +114 -0
- package/vendor-core/impact-model.js +175 -0
- package/vendor-core/ingest.js +2 -0
- package/vendor-core/intervals.js +13 -0
- package/vendor-core/list-runs.js +2 -3
- package/vendor-core/load-vendored.js +70 -5
- package/vendor-core/mcp-server-factory.js +14 -10
- package/vendor-core/mcp-tools.js +105 -45
- package/vendor-core/model-assembler.js +12 -0
- package/vendor-core/occupancy.js +1 -1
- package/vendor-core/parser-worker.js +22 -5
- package/vendor-core/plan-graph-model.js +3 -2
- package/vendor-core/plan-node-detail.js +1 -1
- package/vendor-core/recommendation-rollup.js +63 -3
- package/vendor-core/redact.js +68 -28
- package/vendor-core/run-comparison.js +40 -7
- package/vendor-core/run-interpretation.js +290 -0
- package/vendor-core/run-outcome.js +74 -0
- package/vendor-core/run-payload.js +17 -0
- package/vendor-core/run-shape.js +40 -0
- package/vendor-core/run-verdict.js +353 -0
- package/vendor-core/scaling-sim.js +4 -5
- package/vendor-core/scorecard-estimates.js +62 -0
- package/vendor-core/shs-fetch.js +175 -65
- package/vendor-core/shs-load.js +1 -1
- package/vendor-core/sql-stages.js +11 -0
- package/vendor-core/stage-quantiles.js +14 -0
- package/vendor-core/task-failure.js +151 -0
- package/vendor-core/threshold-overrides.js +160 -0
- package/vendor-core/threshold-summary.js +11 -33
- package/vendor-core/types.js +6 -42
- package/vendor-core/vendor/fflate.js +1 -1
- package/vendor-core/wall-clock.js +1 -12
- package/vendor-core/wasted-core-hours.js +2 -2
- package/vendor-core/zip-archive.js +167 -0
package/LICENSE
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Agustin Recoba
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
package/README.md
CHANGED
|
@@ -2,5 +2,11 @@
|
|
|
2
2
|
|
|
3
3
|
MCP server exposing Apache Spark event-log diagnostics over stdio.
|
|
4
4
|
|
|
5
|
+
```bash
|
|
6
|
+
npx sparkforensics-mcp
|
|
7
|
+
```
|
|
8
|
+
|
|
9
|
+
Point your MCP client at that command.
|
|
10
|
+
|
|
5
11
|
See the [main README](https://github.com/shuffle-works/sparkforensics#readme)
|
|
6
|
-
for
|
|
12
|
+
for the full tool reference.
|
|
@@ -1,5 +1,6 @@
|
|
|
1
1
|
#!/usr/bin/env node
|
|
2
2
|
import { existsSync } from 'node:fs';
|
|
3
|
+
import { parseArgs } from 'node:util';
|
|
3
4
|
import { join, dirname } from 'node:path';
|
|
4
5
|
import { fileURLToPath, pathToFileURL } from 'node:url';
|
|
5
6
|
import { StdioServerTransport } from '@modelcontextprotocol/sdk/server/stdio.js';
|
|
@@ -7,16 +8,20 @@ import { StdioServerTransport } from '@modelcontextprotocol/sdk/server/stdio.js'
|
|
|
7
8
|
const binDir = dirname(fileURLToPath(import.meta.url));
|
|
8
9
|
const pkgDir = dirname(binDir);
|
|
9
10
|
|
|
10
|
-
// vendor-core/
|
|
11
|
-
//
|
|
11
|
+
// vendor-core/ is populated by vendor-core.mjs at pack time. In the monorepo the
|
|
12
|
+
// packages/core/src/ sibling's load-vendored.js is used, which prefers a leftover
|
|
13
|
+
// vendor-core/ only while it still matches core/src and warns when it does not.
|
|
12
14
|
// load-vendored.js is the one module located by hand; the rest load through its
|
|
13
15
|
// exported loadVendored().
|
|
14
|
-
async function
|
|
15
|
-
const
|
|
16
|
-
const helperPath = existsSync(
|
|
16
|
+
async function loadCoreModule(moduleName) {
|
|
17
|
+
const srcHelper = join(pkgDir, '..', 'core', 'src', 'load-vendored.js');
|
|
18
|
+
const helperPath = existsSync(srcHelper) ? srcHelper : join(pkgDir, 'vendor-core', 'load-vendored.js');
|
|
17
19
|
const { loadVendored } = await import(pathToFileURL(helperPath).href);
|
|
18
|
-
|
|
19
|
-
|
|
20
|
+
return loadVendored(pkgDir, moduleName);
|
|
21
|
+
}
|
|
22
|
+
|
|
23
|
+
async function loadCreateMcpServer() {
|
|
24
|
+
return (await loadCoreModule('mcp-server-factory')).createMcpServer;
|
|
20
25
|
}
|
|
21
26
|
|
|
22
27
|
// Tool names come from the server the bin actually builds, so --help can't drift
|
|
@@ -24,25 +29,51 @@ async function loadCreateMcpServer() {
|
|
|
24
29
|
// no I/O. _registeredTools is the SDK's registry; the MCP package test checks that
|
|
25
30
|
// this list matches what listTools reports.
|
|
26
31
|
function usage(toolNames) {
|
|
27
|
-
return `Usage: sparkforensics-mcp
|
|
32
|
+
return `Usage: sparkforensics-mcp [--thresholds <file>]
|
|
28
33
|
|
|
29
34
|
Starts the SparkForensics MCP server, speaking the MCP protocol over
|
|
30
35
|
stdio. Point an MCP client (Claude Desktop, Claude Code, etc.) at this
|
|
31
36
|
command; it exposes ${toolNames.length} tools for diagnosing Apache Spark event logs:
|
|
32
37
|
${toolNames.join(', ')}.
|
|
33
38
|
|
|
39
|
+
--thresholds <file> Run every tool's detectors with the threshold overrides in
|
|
40
|
+
this JSON file ({"<detector>": {"<threshold>": value}}).
|
|
41
|
+
Findings a tuned detector produces carry tunedThresholds.
|
|
42
|
+
An unreadable or invalid file stops the server from starting.
|
|
43
|
+
|
|
34
44
|
See https://github.com/shuffle-works/sparkforensics#readme for details.
|
|
35
45
|
`;
|
|
36
46
|
}
|
|
37
47
|
|
|
38
48
|
async function main() {
|
|
39
|
-
|
|
49
|
+
// Not strict: arguments this server never read before stay ignored, as they always were.
|
|
50
|
+
const { values } = parseArgs({
|
|
51
|
+
args: process.argv.slice(2), strict: false,
|
|
52
|
+
options: { thresholds: { type: 'string' }, help: { type: 'boolean', short: 'h' } },
|
|
53
|
+
});
|
|
54
|
+
if (values.help) {
|
|
40
55
|
const createMcpServer = await loadCreateMcpServer();
|
|
41
56
|
process.stderr.write(usage(Object.keys(createMcpServer()._registeredTools)));
|
|
42
57
|
return;
|
|
43
58
|
}
|
|
59
|
+
let thresholds;
|
|
60
|
+
if (values.thresholds !== undefined) {
|
|
61
|
+
if (typeof values.thresholds !== 'string') {
|
|
62
|
+
process.stderr.write('--thresholds requires a file path.\n');
|
|
63
|
+
process.exitCode = 2;
|
|
64
|
+
return;
|
|
65
|
+
}
|
|
66
|
+
// Refuse to start rather than serve default-threshold results the user meant to change.
|
|
67
|
+
try {
|
|
68
|
+
thresholds = (await loadCoreModule('cli/threshold-config')).loadThresholdOverrides(values.thresholds);
|
|
69
|
+
} catch (e) {
|
|
70
|
+
process.stderr.write(`--thresholds: ${e.message}\n`);
|
|
71
|
+
process.exitCode = 2;
|
|
72
|
+
return;
|
|
73
|
+
}
|
|
74
|
+
}
|
|
44
75
|
const createMcpServer = await loadCreateMcpServer();
|
|
45
|
-
const server = createMcpServer();
|
|
76
|
+
const server = createMcpServer({ thresholds });
|
|
46
77
|
const transport = new StdioServerTransport();
|
|
47
78
|
await server.connect(transport);
|
|
48
79
|
}
|
package/package.json
CHANGED
package/vendor-core/analyzer.js
CHANGED
|
@@ -1,14 +1,28 @@
|
|
|
1
|
-
import { DETECTORS,
|
|
1
|
+
import { DETECTORS, ENTRY_BY_TYPE, } from './detectors.js';
|
|
2
|
+
import { effectiveThresholds, findingTunedThresholds, overridesFor, tunedThresholdsNote } from './threshold-overrides.js';
|
|
2
3
|
import { computePeakConcurrentCores } from './core-count.js';
|
|
3
4
|
import { assertNever } from './assert-never.js';
|
|
4
|
-
import { estimateImpact
|
|
5
|
+
import { estimateImpact, } from './impact-estimator.js';
|
|
5
6
|
import { computeOccupancy, } from './occupancy.js';
|
|
6
|
-
import { deriveImpactBand } from './impact-band.js';
|
|
7
|
+
import { deriveImpactBand, IMPACT_FLOOR_PCT_CRIT, IMPACT_FLOOR_PCT_WARN, } from './impact-band.js';
|
|
7
8
|
import { IMPACT_BAND_ORDER } from './format-utils.js';
|
|
8
9
|
|
|
9
|
-
|
|
10
|
+
|
|
10
11
|
|
|
11
12
|
|
|
13
|
+
// The runner reads each entry through the Detector contract, not its own precise `as const` shape:
|
|
14
|
+
// the scope switch below calls each entry's bound detect with that scope's target.
|
|
15
|
+
const detectors = DETECTORS;
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
|
|
12
26
|
// FNV-1a 32-bit stable string hash: deterministic finding id across runs, no timestamps/randomness.
|
|
13
27
|
function fnv1a(str ) {
|
|
14
28
|
let h = 0x811c9dc5;
|
|
@@ -19,28 +33,59 @@ function fnv1a(str ) {
|
|
|
19
33
|
return (h >>> 0).toString(16).padStart(8, '0');
|
|
20
34
|
}
|
|
21
35
|
|
|
36
|
+
// The id hash's discriminator slots, in the order it has always joined them: reordering or renaming
|
|
37
|
+
// one changes every finding id, which needs an EVIDENCE_SCHEMA_VERSION bump.
|
|
38
|
+
const DISCRIMINATOR_SLOTS = [
|
|
39
|
+
'host', 'executorId', 'rule', 'variant', 'dimension',
|
|
40
|
+
'direction', 'nodeName', 'rootName', 'subtreeSize', 'groupIndex', 'largerSideBytes',
|
|
41
|
+
'rddId', 'relation', 'format', 'operator', 'executionIds',
|
|
42
|
+
] ;
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
// Per type, the fields that tell apart sibling findings sharing one location and metric; without
|
|
46
|
+
// them those siblings hash to one id:
|
|
47
|
+
// slowHost (host), memoryUtilization (executorId), partitionSizing (rule);
|
|
48
|
+
// cacheUtilization (rddId+variant), cachingOpportunity (relation/format for leaf,
|
|
49
|
+
// operator+relation for composite, executionIds as last resort);
|
|
50
|
+
// smallFiles (direction/nodeName), duplicatePlanSubtree (groupIndex is the real
|
|
51
|
+
// uniqueness guarantee: rootName+subtreeSize can collide across groups),
|
|
52
|
+
// underBroadcast (value+largerSideBytes per node/side).
|
|
53
|
+
// memoryUtilization leaves out `rule`: one heap band per executor, so executorId is already
|
|
54
|
+
// unique and folding rule in risks id churn if band logic changes. partitionSizing keeps
|
|
55
|
+
// `rule`: a stage can emit several rules at once sharing stageId+metric.
|
|
56
|
+
const ID_DISCRIMINATORS = {
|
|
57
|
+
skew: [], stageShape: ['rule'], shuffle: [], partitionSizing: ['rule'], spill: [], gc: ['direction'],
|
|
58
|
+
slowHost: ['host', 'executorId', 'variant', 'dimension'], stageSlowness: [], stageFailed: ['variant'],
|
|
59
|
+
failures: [], straggler: [], speculationWaste: [], retryWaste: [], tinyTask: [],
|
|
60
|
+
incompleteRun: [], coldStart: [], utilization: [], memoryUtilization: ['executorId', 'variant'],
|
|
61
|
+
cacheUtilization: ['variant', 'rddId'], coreLocality: [], autoscalingChurn: [],
|
|
62
|
+
cachingOpportunity: ['variant', 'relation', 'format', 'operator', 'executionIds'],
|
|
63
|
+
jobFailureRate: [], configAudit: [],
|
|
64
|
+
duplicatePlanSubtree: ['rootName', 'subtreeSize', 'groupIndex'], smallFiles: ['direction', 'nodeName'],
|
|
65
|
+
underBroadcast: ['largerSideBytes'], overBroadcast: [],
|
|
66
|
+
};
|
|
67
|
+
|
|
68
|
+
// ID_DISCRIMINATORS as Sets, built once: findingId runs once per finding.
|
|
69
|
+
const ID_DISCRIMINATOR_SETS = new Map (
|
|
70
|
+
Object.entries(ID_DISCRIMINATORS).map(([type, slots]) => [type, new Set (slots)]),
|
|
71
|
+
);
|
|
72
|
+
|
|
73
|
+
// Location key: stage, else SQL execution, else audited config property.
|
|
74
|
+
function locationKey(f ) {
|
|
75
|
+
if (f.stageId != null) return f.stageId;
|
|
76
|
+
if ('executionId' in f) return f.executionId;
|
|
77
|
+
if (f.type === 'configAudit') return f.property;
|
|
78
|
+
return '';
|
|
79
|
+
}
|
|
80
|
+
|
|
22
81
|
export function findingId(f ) {
|
|
23
|
-
|
|
24
|
-
const
|
|
25
|
-
|
|
26
|
-
|
|
27
|
-
|
|
28
|
-
|
|
29
|
-
|
|
30
|
-
// smallFiles (direction/nodeName), duplicatePlanSubtree (groupIndex is the real
|
|
31
|
-
// uniqueness guarantee: rootName+subtreeSize can collide across groups),
|
|
32
|
-
// broadcastSizing (value+largerSideBytes per node/side).
|
|
33
|
-
// memoryUtilization excludes `rule`: one heap band per executor, so executorId is already
|
|
34
|
-
// unique and folding rule in risks id churn if band logic changes. partitionSizing keeps
|
|
35
|
-
// `rule`: a stage can emit several rules at once sharing stageId+metric.
|
|
36
|
-
const rule = f.type === 'memoryUtilization' ? undefined : f.rule;
|
|
37
|
-
const disc = [
|
|
38
|
-
f.host, f.executorId, rule, f.variant, f.dimension,
|
|
39
|
-
f.direction, f.nodeName, f.rootName, f.subtreeSize, f.groupIndex, f.largerSideBytes,
|
|
40
|
-
f.rddId, f.relation, f.format, f.operator,
|
|
41
|
-
f.executionIds ? f.executionIds.join(',') : '',
|
|
42
|
-
].map((v) => v ?? '').join('|');
|
|
43
|
-
return fnv1a(`${f.type}|${locKey}|${f.metric ?? ''}|${f.value ?? ''}|${disc}`);
|
|
82
|
+
const fields = f ;
|
|
83
|
+
const listed = ID_DISCRIMINATOR_SETS.get(f.type);
|
|
84
|
+
const disc = DISCRIMINATOR_SLOTS.map((slot) => {
|
|
85
|
+
const v = listed?.has(slot) ? fields[slot] : undefined;
|
|
86
|
+
return Array.isArray(v) ? v.join(',') : v ?? '';
|
|
87
|
+
}).join('|');
|
|
88
|
+
return fnv1a(`${f.type}|${locationKey(f)}|${f.metric ?? ''}|${f.value ?? f.valueText ?? ''}|${disc}`);
|
|
44
89
|
}
|
|
45
90
|
|
|
46
91
|
// skew's max/median branch (stage.taskCount below minTasksForP95) and straggler are both driven
|
|
@@ -72,13 +117,13 @@ function flagSkewStragglerOverlap(findings ) {
|
|
|
72
117
|
}
|
|
73
118
|
}
|
|
74
119
|
|
|
75
|
-
function push(out , entry , result ) {
|
|
120
|
+
function push(out , entry , result , tuned = null) {
|
|
76
121
|
if (!result) return;
|
|
77
122
|
const detectorVersion = entry.version ?? 1;
|
|
78
123
|
for (const f of (Array.isArray(result) ? result : [result])) {
|
|
79
124
|
if (!f) continue;
|
|
80
|
-
if (entry.suppressWhen && entry.suppressWhen(f, out)) continue;
|
|
81
125
|
const stamped = { ...f, docAnchor: f.docAnchor ?? entry.docAnchor, detectorVersion };
|
|
126
|
+
if (tuned) stamped.tunedThresholds = tuned;
|
|
82
127
|
const id = findingId(stamped);
|
|
83
128
|
// Dedup guard: same id => same finding, keep first. Correctness depends on
|
|
84
129
|
// findingId's discriminators being unique per distinct finding, not on this line.
|
|
@@ -87,6 +132,50 @@ function push(out , entry , result
|
|
|
87
132
|
}
|
|
88
133
|
}
|
|
89
134
|
|
|
135
|
+
// A tuned finding's caveat, once its estimate is known: only a finding with an estimate figure
|
|
136
|
+
// (wall-clock or raw waste) says that figure is unvalidated.
|
|
137
|
+
function noteTunedThresholds(findings ) {
|
|
138
|
+
for (const f of findings) {
|
|
139
|
+
if (!f.tunedThresholds) continue;
|
|
140
|
+
const hasFigure = f.impactEstimate != null && f.impactEstimate.basis !== 'informational';
|
|
141
|
+
const note = tunedThresholdsNote(f.tunedThresholds, hasFigure);
|
|
142
|
+
f.validationRequired = f.validationRequired ? `${f.validationRequired} ${note}` : note;
|
|
143
|
+
}
|
|
144
|
+
}
|
|
145
|
+
|
|
146
|
+
// skew and straggler gate on floorPctWarn/floorPctCrit thresholds that default to the band's own
|
|
147
|
+
// floors, so a finding they admit at their warn floor grades at least warning. A tuned floor grades
|
|
148
|
+
// that entry's findings too; a type whose entry has no such threshold keeps the defaults.
|
|
149
|
+
function bandFloors(type , overrides ) {
|
|
150
|
+
const entry = ENTRY_BY_TYPE.get(type);
|
|
151
|
+
if (!entry) return null;
|
|
152
|
+
const { floorPctWarn, floorPctCrit } = effectiveThresholds(entry, overrides);
|
|
153
|
+
return {
|
|
154
|
+
warnPct: typeof floorPctWarn === 'number' ? floorPctWarn : IMPACT_FLOOR_PCT_WARN,
|
|
155
|
+
critPct: typeof floorPctCrit === 'number' ? floorPctCrit : IMPACT_FLOOR_PCT_CRIT,
|
|
156
|
+
};
|
|
157
|
+
}
|
|
158
|
+
|
|
159
|
+
// An entry's `suppressedBy` names another entry: drop its findings on every stage that entry
|
|
160
|
+
// flagged. Runs once every detector has, so neither declaration order matters, and reads the
|
|
161
|
+
// unsuppressed findings, so the result doesn't depend on which suppression is applied first.
|
|
162
|
+
function applySuppression(out ) {
|
|
163
|
+
const dropped = new Map ();
|
|
164
|
+
for (const entry of detectors) {
|
|
165
|
+
if (!entry.suppressedBy) continue;
|
|
166
|
+
const suppressorTypes = new Set (
|
|
167
|
+
detectors.filter((d) => d.type === entry.suppressedBy).flatMap((d) => d.emits),
|
|
168
|
+
);
|
|
169
|
+
for (const type of entry.emits) {
|
|
170
|
+
const stagesToDrop = dropped.get(type) ?? new Set ();
|
|
171
|
+
for (const f of out) if (suppressorTypes.has(f.type) && f.stageId != null) stagesToDrop.add(f.stageId);
|
|
172
|
+
dropped.set(type, stagesToDrop);
|
|
173
|
+
}
|
|
174
|
+
}
|
|
175
|
+
if (dropped.size === 0) return out;
|
|
176
|
+
return out.filter((f) => f.stageId == null || !dropped.get(f.type)?.has(f.stageId));
|
|
177
|
+
}
|
|
178
|
+
|
|
90
179
|
// `app` widened to `SparkAppInfo | null` to match real callers (AppModel.app is
|
|
91
180
|
// nullable at the type level); every detector below tolerates a null app.
|
|
92
181
|
export function analyze(
|
|
@@ -97,6 +186,7 @@ export function analyze(
|
|
|
97
186
|
jobs ,
|
|
98
187
|
sql = new Map(),
|
|
99
188
|
runAggregates = null,
|
|
189
|
+
{ thresholds } = {},
|
|
100
190
|
) {
|
|
101
191
|
// `app ?? {}`: detectors tolerate a null app (malformed logs), so this must too.
|
|
102
192
|
// computePeakConcurrentCores (not computeTotalCores): the occupancy ceiling needs a
|
|
@@ -108,37 +198,55 @@ export function analyze(
|
|
|
108
198
|
executorsAdded ,
|
|
109
199
|
executorsRemoved ,
|
|
110
200
|
);
|
|
111
|
-
//
|
|
112
|
-
//
|
|
113
|
-
const
|
|
114
|
-
|
|
115
|
-
|
|
201
|
+
// One occupancy sweep per analysis, shared by the detectors' runtime floors and every entry's
|
|
202
|
+
// estimate(), so a floor gates on the same occupancy-clipped figure displayed as savings.
|
|
203
|
+
const impact = {
|
|
204
|
+
stages, totalCores,
|
|
205
|
+
occupancy: computeOccupancy(stages , totalCores),
|
|
206
|
+
};
|
|
207
|
+
// The one cast from the posted-model types to the detector-side shapes: types.ts's Stage and
|
|
208
|
+
// SqlExecution carry a catch-all index signature, while every field DetectorStage/DetectorSqlExec
|
|
209
|
+
// declare is one finalizeStage and event-handlers.ts always set (see detectors.ts's header).
|
|
210
|
+
const ctx = {
|
|
211
|
+
app, jobs, executorsAdded, executorsRemoved, runAggregates, impact,
|
|
212
|
+
stages: stages ,
|
|
213
|
+
sql: sql ,
|
|
116
214
|
};
|
|
117
215
|
const out = [];
|
|
118
|
-
for (const d of
|
|
216
|
+
for (const d of detectors) {
|
|
119
217
|
if (d.inScorecard === false) continue;
|
|
218
|
+
const overrides = overridesFor(d, thresholds);
|
|
219
|
+
const tuned = findingTunedThresholds(d, thresholds);
|
|
120
220
|
switch (d.scope) {
|
|
121
|
-
case 'stage':
|
|
122
|
-
|
|
221
|
+
case 'stage': {
|
|
222
|
+
const detect = d.withThresholds(overrides);
|
|
223
|
+
for (const s of ctx.stages.values()) push(out, d, detect(s, ctx), tuned);
|
|
123
224
|
break;
|
|
124
|
-
|
|
125
|
-
|
|
225
|
+
}
|
|
226
|
+
case 'sql': {
|
|
227
|
+
const detect = d.withThresholds(overrides);
|
|
228
|
+
for (const e of ctx.sql.values()) push(out, d, detect(e, ctx), tuned);
|
|
126
229
|
break;
|
|
230
|
+
}
|
|
127
231
|
case 'app':
|
|
232
|
+
push(out, d, d.withThresholds(overrides)(ctx), tuned);
|
|
233
|
+
break;
|
|
128
234
|
case 'config':
|
|
129
|
-
push(out, d, d.
|
|
235
|
+
push(out, d, d.withThresholds(overrides)(ctx ), tuned);
|
|
130
236
|
break;
|
|
131
237
|
default:
|
|
132
|
-
assertNever(d
|
|
238
|
+
assertNever(d);
|
|
133
239
|
}
|
|
134
240
|
}
|
|
135
|
-
|
|
136
|
-
|
|
137
|
-
|
|
241
|
+
const findings = applySuppression(out);
|
|
242
|
+
estimateImpact(findings, impact);
|
|
243
|
+
noteTunedThresholds(findings);
|
|
244
|
+
deriveImpactBand(findings, app, thresholds ? (type) => bandFloors(type, thresholds) : undefined);
|
|
245
|
+
flagSkewStragglerOverlap(findings);
|
|
138
246
|
// Ascending IMPACT_BAND_ORDER (critical 0 -> info 2) puts the worst band first;
|
|
139
247
|
// stable sort keeps DETECTORS declaration order within a band.
|
|
140
|
-
|
|
141
|
-
return
|
|
248
|
+
findings.sort((a, b) => IMPACT_BAND_ORDER[a.impactBand] - IMPACT_BAND_ORDER[b.impactBand]);
|
|
249
|
+
return findings;
|
|
142
250
|
}
|
|
143
251
|
|
|
144
252
|
// Memoizes auditConfig by `app` identity (like evidence-report.ts's jsonCache) so config detectors
|
|
@@ -149,10 +257,10 @@ const auditConfigCache = new WeakMap ();
|
|
|
149
257
|
|
|
150
258
|
function computeAuditConfig(app ) {
|
|
151
259
|
const out = [];
|
|
152
|
-
for (const d of
|
|
153
|
-
// configAudit's
|
|
154
|
-
//
|
|
155
|
-
estimateImpact(out, new Map());
|
|
260
|
+
for (const d of detectors) if (d.scope === 'config') push(out, d, d.withThresholds()({ app }));
|
|
261
|
+
// configAudit's estimate is unconditionally costOnly('none'): needs no stages/totalCores, so an
|
|
262
|
+
// empty context gives parity with analyze().
|
|
263
|
+
estimateImpact(out, { stages: new Map(), occupancy: new Map(), totalCores: 0 });
|
|
156
264
|
deriveImpactBand(out, app);
|
|
157
265
|
return out.map((f) => ({ ...f, stageId: f.stageId ?? null }));
|
|
158
266
|
}
|
|
@@ -0,0 +1,88 @@
|
|
|
1
|
+
// Which checks a log could actually run. Shared by the dashboard (verdict, top bar, Clean checks)
|
|
2
|
+
// and the CLI/MCP evidence report, so a check the log lacked the data for never reads as passed on
|
|
3
|
+
// one path and "not checked" on another.
|
|
4
|
+
import { detectorCatalog } from './detectors.js';
|
|
5
|
+
import { isRealFinding } from './recommendation-rollup.js';
|
|
6
|
+
import { summarizeRunOutcome } from './run-outcome.js';
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
export const NO_FINISHED_STAGE_GAP = 'No stage in this log recorded an end, so the stage checks had nothing to measure.';
|
|
10
|
+
export const INCOMPLETE_RUN_GAP =
|
|
11
|
+
'The log has no end-of-run record, so the core usage, memory and executor churn checks had no run length to measure.';
|
|
12
|
+
|
|
13
|
+
/** True when at least one stage recorded both a start and an end, so the
|
|
14
|
+
* stage checks had something to measure. */
|
|
15
|
+
export function hasFinishedStage(stages ) {
|
|
16
|
+
return [...stages.values()].some((stage) => stage.submittedAt != null && stage.completedAt != null);
|
|
17
|
+
}
|
|
18
|
+
|
|
19
|
+
/** A finding that reports a check could not run for lack of evidence (cache
|
|
20
|
+
* storage without block updates, memory without executor metrics), rather
|
|
21
|
+
* than a problem found. Its recommendation names the setting to turn on. */
|
|
22
|
+
export function isEvidenceCaveat(finding ) {
|
|
23
|
+
return ('dataUnavailable' in finding && finding.dataUnavailable === true) || !isRealFinding(finding);
|
|
24
|
+
}
|
|
25
|
+
|
|
26
|
+
/** App-level checks measured over the run's full span, which a log with no
|
|
27
|
+
* ApplicationEnd (an `incompleteRun` finding) cannot give them. */
|
|
28
|
+
export const RUN_SPAN_CHECK_TYPES = new Set(['utilization', 'memoryUtilization', 'autoscalingChurn']);
|
|
29
|
+
|
|
30
|
+
/** Detector types that measure each stage, read from the detector catalog. */
|
|
31
|
+
export const PER_STAGE_CHECK_TYPES = new Set(
|
|
32
|
+
detectorCatalog().filter((entry) => entry.scope === 'stage').map((entry) => entry.type),
|
|
33
|
+
);
|
|
34
|
+
|
|
35
|
+
export function isIncompleteRun(allFindings ) {
|
|
36
|
+
return allFindings.some((finding) => finding.type === 'incompleteRun');
|
|
37
|
+
}
|
|
38
|
+
|
|
39
|
+
/** What this log could not check, in plain sentences, each saying what to
|
|
40
|
+
* turn on for the next run where the detector names it. */
|
|
41
|
+
export function verdictGaps(allFindings , noFinishedStages ) {
|
|
42
|
+
const gaps = new Set ();
|
|
43
|
+
if (noFinishedStages) gaps.add(NO_FINISHED_STAGE_GAP);
|
|
44
|
+
if (isIncompleteRun(allFindings)) gaps.add(INCOMPLETE_RUN_GAP);
|
|
45
|
+
for (const finding of allFindings) {
|
|
46
|
+
if (isEvidenceCaveat(finding) && finding.recommendation) gaps.add(finding.recommendation);
|
|
47
|
+
}
|
|
48
|
+
return [...gaps];
|
|
49
|
+
}
|
|
50
|
+
|
|
51
|
+
/** The one rule for calling a run clean, shared by the verdict, the top bar
|
|
52
|
+
* and the evidence report: no finding at all, no failed job, and nothing the
|
|
53
|
+
* log lacked to run a check. */
|
|
54
|
+
export function isCleanRun(appModel , allFindings ) {
|
|
55
|
+
if (summarizeRunOutcome(appModel.jobs, allFindings).failedJobs > 0) return false;
|
|
56
|
+
if (allFindings.some(isRealFinding)) return false;
|
|
57
|
+
return verdictGaps(allFindings, !hasFinishedStage(appModel.stages)).length === 0;
|
|
58
|
+
}
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
/** The not-run rule for one run: an evidence caveat of that type, a per-stage
|
|
69
|
+
* check on a log where no stage finished, or a run-span check on a log with
|
|
70
|
+
* no ApplicationEnd. */
|
|
71
|
+
export function checkCoverage(stages , allFindings ) {
|
|
72
|
+
const noFinishedStages = !hasFinishedStage(stages);
|
|
73
|
+
const incomplete = isIncompleteRun(allFindings);
|
|
74
|
+
const caveatReasons = new Map ();
|
|
75
|
+
for (const finding of allFindings) {
|
|
76
|
+
if (!isEvidenceCaveat(finding)) continue;
|
|
77
|
+
if (!caveatReasons.get(finding.type)) caveatReasons.set(finding.type, finding.recommendation ?? null);
|
|
78
|
+
}
|
|
79
|
+
const notRunReason = (type ) => {
|
|
80
|
+
const caveatReason = caveatReasons.get(type);
|
|
81
|
+
if (caveatReason) return caveatReason;
|
|
82
|
+
if (noFinishedStages && PER_STAGE_CHECK_TYPES.has(type)) return NO_FINISHED_STAGE_GAP;
|
|
83
|
+
if (incomplete && RUN_SPAN_CHECK_TYPES.has(type)) return INCOMPLETE_RUN_GAP;
|
|
84
|
+
// A caveat with no recommendation still means the check did not run.
|
|
85
|
+
return caveatReasons.has(type) ? 'The log lacked the data this check needs.' : null;
|
|
86
|
+
};
|
|
87
|
+
return { isNotRun: (type) => notRunReason(type) != null, notRunReason };
|
|
88
|
+
}
|
|
@@ -1,5 +1,6 @@
|
|
|
1
1
|
import { computeEfficiencyModel } from '../efficiency-model.js';
|
|
2
|
-
import { computeSkewRatio,
|
|
2
|
+
import { computeSkewRatio, ENTRY_BY_TYPE, } from '../detectors.js';
|
|
3
|
+
import { effectiveThresholds } from '../threshold-overrides.js';
|
|
3
4
|
import { IMPACT_BAND_ORDER } from '../format-utils.js';
|
|
4
5
|
|
|
5
6
|
|
|
@@ -13,26 +14,26 @@ const IMPACT_BANDS = Object.keys(IMPACT_BAND_ORDER) ;
|
|
|
13
14
|
|
|
14
15
|
|
|
15
16
|
|
|
16
|
-
|
|
17
|
+
|
|
17
18
|
|
|
18
19
|
|
|
19
20
|
|
|
20
21
|
|
|
21
|
-
|
|
22
|
-
|
|
22
|
+
// The skew entry's minTasksForP95 under the run's overrides, so the budget measures the same
|
|
23
|
+
// ratio (P95/median or max/median) the skew finding reports.
|
|
24
|
+
function skewMinTasksForP95(thresholds ) {
|
|
25
|
+
return effectiveThresholds(ENTRY_BY_TYPE.get('skew') , thresholds).minTasksForP95 ;
|
|
26
|
+
}
|
|
23
27
|
|
|
24
28
|
function taskDataTrusted(appModel ) {
|
|
25
29
|
const entry = appModel.evidenceAvailability?.entries?.find((e) => e.key === 'taskCoreTime');
|
|
26
30
|
return entry?.state === 'present';
|
|
27
31
|
}
|
|
28
32
|
|
|
29
|
-
// Finding.value is number|string (some detectors put text there); spill findings are
|
|
30
|
-
// always numeric, so the typeof guard narrows without changing behavior for real input.
|
|
31
33
|
function maxFindingValue(catalog , type ) {
|
|
32
34
|
const values = catalog
|
|
33
35
|
.filter((f) => f.type === type)
|
|
34
|
-
.map((f) => f.value ?? 0)
|
|
35
|
-
.filter((v) => typeof v === 'number');
|
|
36
|
+
.map((f) => f.value ?? 0);
|
|
36
37
|
return values.length > 0 ? Math.max(...values) : null;
|
|
37
38
|
}
|
|
38
39
|
|
|
@@ -58,7 +59,7 @@ function checkSpill(appModel , catalog , maxSpillGb )
|
|
|
58
59
|
: { name: 'max-spill', status: 'pass', detail: `Peak stage spill ${maxBytes} bytes within budget ${budgetBytes} bytes.` };
|
|
59
60
|
}
|
|
60
61
|
|
|
61
|
-
function checkSkew(appModel , maxSkewRatio ) {
|
|
62
|
+
function checkSkew(appModel , maxSkewRatio , minTasksForP95 ) {
|
|
62
63
|
if (!taskDataTrusted(appModel)) {
|
|
63
64
|
return { name: 'max-skew', status: 'inconclusive', detail: 'No trustworthy task-level evidence to measure skew.' };
|
|
64
65
|
}
|
|
@@ -69,7 +70,7 @@ function checkSkew(appModel , maxSkewRatio ) {
|
|
|
69
70
|
// Recompute the true ratio per stage: the skew detector floors findings at
|
|
70
71
|
// thresholds.ratioWarn (3), so a stricter budget can't be enforced from catalog alone.
|
|
71
72
|
const ratios = stages
|
|
72
|
-
.map((stage) => computeSkewRatio(stage,
|
|
73
|
+
.map((stage) => computeSkewRatio(stage, minTasksForP95))
|
|
73
74
|
.filter((r) => r !== null)
|
|
74
75
|
.map((r) => r.ratio);
|
|
75
76
|
if (ratios.length === 0) {
|
|
@@ -97,21 +98,24 @@ function checkFailedTaskRate(appModel , catalog , maxPct
|
|
|
97
98
|
return { name: 'max-failed-task-rate', status: 'pass', detail: `No job-failure-rate finding: task failure rate is below the detector's reporting floor.` };
|
|
98
99
|
}
|
|
99
100
|
|
|
101
|
+
// Busy core time: the share of available executor core time that ran tasks, 100 minus the
|
|
102
|
+
// dashboard's "Unused core time". Not the dashboard's Efficiency tile (the share of wall-clock with
|
|
103
|
+
// a stage running), so the detail never calls it "Efficiency".
|
|
100
104
|
function checkEfficiency(appModel , minPct ) {
|
|
101
105
|
if (!taskDataTrusted(appModel)) {
|
|
102
|
-
return { name: 'min-efficiency', status: 'inconclusive', detail: 'No trustworthy task-level evidence to measure
|
|
106
|
+
return { name: 'min-efficiency', status: 'inconclusive', detail: 'No trustworthy task-level evidence to measure busy core time.' };
|
|
103
107
|
}
|
|
104
108
|
const model = computeEfficiencyModel({
|
|
105
109
|
app: appModel.app, stages: appModel.stages,
|
|
106
110
|
executorsAdded: appModel.executors.added, runAggregates: appModel.runAggregates,
|
|
107
111
|
});
|
|
108
112
|
if (model.wastagePct == null) {
|
|
109
|
-
return { name: 'min-efficiency', status: 'inconclusive', detail: '
|
|
113
|
+
return { name: 'min-efficiency', status: 'inconclusive', detail: 'Busy core time could not be computed (no available compute hours).' };
|
|
110
114
|
}
|
|
111
|
-
const
|
|
112
|
-
return
|
|
113
|
-
? { name: 'min-efficiency', status: 'violation', detail: `
|
|
114
|
-
: { name: 'min-efficiency', status: 'pass', detail: `
|
|
115
|
+
const busyCorePct = 100 - model.wastagePct;
|
|
116
|
+
return busyCorePct < minPct
|
|
117
|
+
? { name: 'min-efficiency', status: 'violation', detail: `Busy core time ${busyCorePct}% below budget ${minPct}%.` }
|
|
118
|
+
: { name: 'min-efficiency', status: 'pass', detail: `Busy core time ${busyCorePct}% meets budget ${minPct}%.` };
|
|
115
119
|
}
|
|
116
120
|
|
|
117
121
|
function checkRegression(comparison , maxRegressionPct , regressionMetric ) {
|
|
@@ -158,13 +162,16 @@ function pushComparisonBudget(
|
|
|
158
162
|
results.push(comparison ? check(comparison) : { name, status: 'inconclusive', detail: 'No baseline comparison available to evaluate this budget.' });
|
|
159
163
|
}
|
|
160
164
|
|
|
161
|
-
|
|
165
|
+
/** `thresholds`: the overrides the catalog was analyzed with, so a budget that recomputes a
|
|
166
|
+
* detector's figure (--max-skew) uses the same thresholds. */
|
|
167
|
+
export function evaluateBudgets({ appModel, catalog, budgets, comparison, thresholds }
|
|
162
168
|
|
|
169
|
+
|
|
163
170
|
) {
|
|
164
171
|
const results = [];
|
|
165
172
|
if (Number.isFinite(budgets.maxRuntimeMs)) results.push(checkRuntime(appModel, budgets.maxRuntimeMs ));
|
|
166
173
|
if (Number.isFinite(budgets.maxSpillGb)) results.push(checkSpill(appModel, catalog, budgets.maxSpillGb ));
|
|
167
|
-
if (Number.isFinite(budgets.maxSkewRatio)) results.push(checkSkew(appModel, budgets.maxSkewRatio ));
|
|
174
|
+
if (Number.isFinite(budgets.maxSkewRatio)) results.push(checkSkew(appModel, budgets.maxSkewRatio , skewMinTasksForP95(thresholds)));
|
|
168
175
|
if (Number.isFinite(budgets.maxFailedTaskRatePct)) results.push(checkFailedTaskRate(appModel, catalog, budgets.maxFailedTaskRatePct ));
|
|
169
176
|
if (Number.isFinite(budgets.minEfficiencyPct)) results.push(checkEfficiency(appModel, budgets.minEfficiencyPct ));
|
|
170
177
|
// Guarded here so any evaluateBudgets caller benefits: regressionMetric without
|
|
@@ -181,6 +188,12 @@ export function evaluateBudgets({ appModel, catalog, budgets, comparison }
|
|
|
181
188
|
pushComparisonBudget(results, comparison, 'fail-on-introduced',
|
|
182
189
|
(c) => checkFailOnIntroduced(c, budgets.failOnIntroduced ));
|
|
183
190
|
}
|
|
191
|
+
// Always checked, unlike the opt-in budgets above: a run with no ApplicationEnd is
|
|
192
|
+
// inconclusive by default, so a passing budget can't hide a truncated log.
|
|
193
|
+
const incompleteRunFinding = catalog.find((f) => f.type === 'incompleteRun');
|
|
194
|
+
if (incompleteRunFinding) {
|
|
195
|
+
results.push({ name: 'run-complete', status: 'inconclusive', detail: incompleteRunFinding.recommendation ?? 'Event log has no ApplicationEnd event.' });
|
|
196
|
+
}
|
|
184
197
|
return {
|
|
185
198
|
results,
|
|
186
199
|
violated: results.some((r) => r.status === 'violation'),
|