sparkforensics-mcp 0.3.0 → 0.4.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/bin/sparkforensics-mcp.mjs +41 -10
- package/package.json +1 -1
- package/vendor-core/analyzer.js +156 -48
- package/vendor-core/check-coverage.js +88 -0
- package/vendor-core/cli/budgets.js +31 -18
- package/vendor-core/cli/collect-run.js +76 -31
- package/vendor-core/cli/threshold-config.js +28 -0
- package/vendor-core/comparison-verdict.js +177 -0
- package/vendor-core/core-source-hash.txt +1 -0
- package/vendor-core/core-usage-locality.js +56 -2
- package/vendor-core/detector-docs.js +58 -0
- package/vendor-core/detectors.js +918 -458
- package/vendor-core/docs-config.js +0 -36
- package/vendor-core/docs-content/chapters/nav-index.json +31 -0
- package/vendor-core/docs-content/detection/cstor.md +9 -0
- package/vendor-core/docs-site-config.js +3 -0
- package/vendor-core/event-handlers.js +170 -6
- package/vendor-core/event-schemas.js +21 -0
- package/vendor-core/evidence-report.js +421 -112
- package/vendor-core/export-data.js +79 -6
- package/vendor-core/finding-action-label.js +9 -88
- package/vendor-core/finding-filter-predicate.js +9 -0
- package/vendor-core/finding-generic-recommendation.js +6 -104
- package/vendor-core/finding-names.js +21 -45
- package/vendor-core/finding-presentation.js +333 -0
- package/vendor-core/finding-tag-help.js +110 -0
- package/vendor-core/finding-types.js +361 -0
- package/vendor-core/findings-of-type.js +11 -0
- package/vendor-core/format-utils.js +92 -27
- package/vendor-core/html-export.js +51 -0
- package/vendor-core/impact-band.js +21 -8
- package/vendor-core/impact-estimator.js +8 -521
- package/vendor-core/impact-format.js +114 -0
- package/vendor-core/impact-model.js +175 -0
- package/vendor-core/ingest.js +2 -0
- package/vendor-core/intervals.js +13 -0
- package/vendor-core/list-runs.js +2 -3
- package/vendor-core/load-vendored.js +70 -5
- package/vendor-core/mcp-server-factory.js +14 -10
- package/vendor-core/mcp-tools.js +105 -45
- package/vendor-core/model-assembler.js +12 -0
- package/vendor-core/occupancy.js +1 -1
- package/vendor-core/parser-worker.js +1 -1
- package/vendor-core/plan-graph-model.js +3 -2
- package/vendor-core/plan-node-detail.js +1 -1
- package/vendor-core/recommendation-rollup.js +63 -3
- package/vendor-core/redact.js +45 -27
- package/vendor-core/run-comparison.js +32 -7
- package/vendor-core/run-interpretation.js +290 -0
- package/vendor-core/run-outcome.js +74 -0
- package/vendor-core/run-payload.js +17 -0
- package/vendor-core/run-shape.js +40 -0
- package/vendor-core/run-verdict.js +353 -0
- package/vendor-core/scaling-sim.js +4 -5
- package/vendor-core/scorecard-estimates.js +62 -0
- package/vendor-core/sql-stages.js +11 -0
- package/vendor-core/stage-quantiles.js +2 -0
- package/vendor-core/threshold-overrides.js +160 -0
- package/vendor-core/threshold-summary.js +11 -33
- package/vendor-core/types.js +6 -42
- package/vendor-core/wall-clock.js +1 -12
- package/vendor-core/wasted-core-hours.js +2 -2
|
@@ -1,527 +1,14 @@
|
|
|
1
|
-
|
|
2
|
-
import {
|
|
3
|
-
|
|
4
|
-
computeOccupancy, estimateSingleStage, estimateMultiStage, tailRecoveryMs, tailRemovedWorkMs, stragglerFixLongestTaskMs,
|
|
5
|
-
|
|
6
|
-
} from './occupancy.js';
|
|
1
|
+
|
|
2
|
+
import { ENTRY_BY_TYPE } from './detectors.js';
|
|
3
|
+
|
|
7
4
|
|
|
8
|
-
|
|
9
|
-
const SHUFFLE_THROUGHPUT_BPS = 125_000_000;
|
|
10
|
-
// Assumed disk I/O throughput per executor for spilled data, ~200 MB/s (conservative HDD/SSD blend).
|
|
11
|
-
const SPILL_IO_THROUGHPUT_BPS = 200_000_000;
|
|
5
|
+
|
|
12
6
|
|
|
13
|
-
|
|
14
|
-
|
|
15
|
-
|
|
16
|
-
// concurrency (task time / stage duration). Null when the stage has no such data or no overhead.
|
|
17
|
-
function measuredTaskOverhead(stage ) {
|
|
18
|
-
const executorStats = Array.isArray(stage.executorStats)
|
|
19
|
-
? (stage.executorStats ) : [];
|
|
20
|
-
const taskTimeMs = executorStats.reduce((sum, e) => sum + (e.totalDuration ?? 0), 0);
|
|
21
|
-
const taskCount = stage.taskCount ?? 0;
|
|
22
|
-
const durationMs = (stage.completedAt ?? 0) - (stage.submittedAt ?? 0);
|
|
23
|
-
const overheadMs = taskTimeMs - (stage.executorRunTime ?? 0);
|
|
24
|
-
if (taskTimeMs <= 0 || taskCount <= 0 || durationMs <= 0 || overheadMs <= 0) return null;
|
|
25
|
-
return { perTaskMs: overheadMs / taskCount, concurrency: taskTimeMs / durationMs };
|
|
26
|
-
}
|
|
27
|
-
|
|
28
|
-
// Both constants above are single-device figures (one NIC, one local disk). A stage's shuffle
|
|
29
|
-
// reads and spills are spread over every executor that ran its tasks, each moving its own share
|
|
30
|
-
// in parallel, so the stage's aggregate bandwidth scales with that executor count. Dividing a
|
|
31
|
-
// stage-wide byte total by one device's bandwidth modeled the whole cluster as one link: on 14
|
|
32
|
-
// real logs that claimed up to 1,870s of shuffle time on stages whose tasks measured ~0s of
|
|
33
|
-
// shuffle fetch wait (222 of 284 shuffle findings). No executor data falls back to one device.
|
|
34
|
-
function stageIoParallelism(stage ) {
|
|
35
|
-
const executors = Array.isArray(stage.executorStats) ? stage.executorStats.length : 0;
|
|
36
|
-
return Math.max(1, executors);
|
|
37
|
-
}
|
|
38
|
-
|
|
39
|
-
// Wall-clock the stage's tasks spent blocked fetching shuffle blocks: fetchWaitTime is a
|
|
40
|
-
// cross-task sum like executorRunTime, so dividing by the stage's average concurrency converts it
|
|
41
|
-
// (the gc estimate's conversion). Null when the stage has no run time or duration to convert with.
|
|
42
|
-
// On 14 real logs the link model claimed 2780s over 284 shuffle findings, 256s once capped at
|
|
43
|
-
// this, with about zero fetch wait on 5 of the 10 non-info ones: reads that overlapped compute
|
|
44
|
-
// stalled nothing.
|
|
45
|
-
function fetchWaitWallClockMs(stage ) {
|
|
46
|
-
const fetchWaitMs = stage.fetchWaitTime;
|
|
47
|
-
const runTimeMs = stage.executorRunTime ?? 0;
|
|
48
|
-
const durationMs = (stage.completedAt ?? 0) - (stage.submittedAt ?? 0);
|
|
49
|
-
if (typeof fetchWaitMs !== 'number' || runTimeMs <= 0 || durationMs <= 0) return null;
|
|
50
|
-
return fetchWaitMs / (runTimeMs / durationMs);
|
|
51
|
-
}
|
|
52
|
-
|
|
53
|
-
// Wall-clock a stage's wasted (retried) attempts cost it. Each one delayed only its own task, and
|
|
54
|
-
// attempts of different tasks ran side by side: one lost executor fails every task it was running
|
|
55
|
-
// at once (4 wasted attempts of 36.6s each, all first attempts, on a real stage that ran 41 tasks
|
|
56
|
-
// at once, claimed as 146.6s). So the stage lost at most its longest retry chain, the most attempts
|
|
57
|
-
// one task wasted (its highest attempt number + 1) times the mean wasted attempt, or their summed
|
|
58
|
-
// time spread over its slots, whichever is larger. Without a sample of every wasted attempt
|
|
59
|
-
// (retryTaskSamples is capped), the chain isn't known: the summed time.
|
|
60
|
-
function retryWallClockMs(stage ) {
|
|
61
|
-
const totalMs = (stage.retryWasteMs ) ?? 0;
|
|
62
|
-
const attempts = (stage.wastedAttempts ) ?? 0;
|
|
63
|
-
const samples = Array.isArray(stage.retryTaskSamples)
|
|
64
|
-
? (stage.retryTaskSamples ) : [];
|
|
65
|
-
if (totalMs <= 0 || attempts <= 0 || samples.length < attempts) return totalMs;
|
|
66
|
-
const chain = samples.reduce((longest, s) => Math.max(longest, (s.attemptNumber ?? 0) + 1), 1);
|
|
67
|
-
const slots = Math.max(1, stage.peakConcurrentTasks ?? 1);
|
|
68
|
-
return Math.min(totalMs, Math.max((chain * totalMs) / attempts, totalMs / slots));
|
|
69
|
-
}
|
|
70
|
-
// Spark's classic recommended shuffle partition size.
|
|
71
|
-
const IDEAL_BYTES_PER_PARTITION_TASK = 128 * 1024 * 1024;
|
|
72
|
-
// Assumed per-task scheduling/launch overhead: the fallback when a stage lacks the per-executor
|
|
73
|
-
// task-time sums measuredTaskOverhead needs.
|
|
74
|
-
const TASK_SCHEDULING_OVERHEAD_MS = 50;
|
|
75
|
-
// Assumed per-file open latency (small-file overhead).
|
|
76
|
-
const FILE_OPEN_OVERHEAD_MS = 10;
|
|
77
|
-
// Assumed broadcast-transfer bandwidth, shared with overBroadcast/underBroadcast.
|
|
78
|
-
const BROADCAST_BANDWIDTH_BPS = 125_000_000;
|
|
79
|
-
// Assumed per-non-local-task network-fetch penalty, reported as extra core-time.
|
|
80
|
-
const NETWORK_FETCH_PENALTY_MS = 20;
|
|
81
|
-
// Assumed executor JVM+container startup overhead.
|
|
82
|
-
const EXECUTOR_STARTUP_OVERHEAD_MS = 15000;
|
|
83
|
-
// Assumed re-read throughput, shared by cachingOpportunity and cacheUtilization.
|
|
84
|
-
const RE_READ_THROUGHPUT_BPS = 125_000_000;
|
|
85
|
-
|
|
86
|
-
// Below this share of executorRunTime spent on CPU, a stage's tasks were idle, waiting on something
|
|
87
|
-
// outside Spark: on 14 real logs (2026-09-23) every non-Python stage under 1% was a JDBC read, a
|
|
88
|
-
// file listing or a Delta log read, while file writes, which more partitions do parallelize,
|
|
89
|
-
// start at 2%.
|
|
90
|
-
const IDLE_CPU_SHARE_MAX = 0.01;
|
|
91
|
-
|
|
92
|
-
// True when the stage's tasks spent under IDLE_CPU_SHARE_MAX of their run time on CPU. False when
|
|
93
|
-
// the share can't be trusted: no CPU time recorded (older Spark logs omit the metric), or Python
|
|
94
|
-
// code run through PythonRDD, whose worker-process CPU executorCpuTime (the JVM task thread's)
|
|
95
|
-
// never counts (such stages read 0.1% on the same logs while computing).
|
|
96
|
-
function tasksMostlyIdle(stage ) {
|
|
97
|
-
const runMs = stage.executorRunTime ?? 0;
|
|
98
|
-
const cpuMs = nsToMs(stage.executorCpuTime ?? 0);
|
|
99
|
-
if (runMs <= 0 || cpuMs <= 0) return false;
|
|
100
|
-
if (/PythonRDD/.test(stage.name ?? '') || /org\.apache\.spark\.api\.python\./.test(stage.details ?? '')) return false;
|
|
101
|
-
return cpuMs / runMs < IDLE_CPU_SHARE_MAX;
|
|
102
|
-
}
|
|
103
|
-
|
|
104
|
-
// skew and straggler claim time off the stage's longest task itself, so the occupancy clip must
|
|
105
|
-
// not floor them at that same task (see estimateSingleStage). detectors.ts's clippedWasteMs gates
|
|
106
|
-
// both detectors on the same option so the firing floor and the displayed estimate agree.
|
|
107
|
-
const TAIL_CLAIM = { shortensLongestTask: true };
|
|
108
|
-
|
|
109
|
-
// No quantifiable magnitude -> 'informational'; a rawWaste figure with no stage window ->
|
|
110
|
-
// 'resourceOnly'. Never a fake {low:0, high:0}: wallClock is null in both cases.
|
|
111
|
-
function costOnly(estimateMethod , rawWaste ) {
|
|
112
|
-
return rawWaste
|
|
113
|
-
? { basis: 'resourceOnly', wallClock: null, estimateMethod, rawWaste }
|
|
114
|
-
: { basis: 'informational', wallClock: null, estimateMethod };
|
|
115
|
-
}
|
|
116
|
-
|
|
117
|
-
function singleStageImpact(
|
|
118
|
-
wasteMs ,
|
|
119
|
-
stageId ,
|
|
120
|
-
stages ,
|
|
121
|
-
occupancy ,
|
|
122
|
-
estimateMethod ,
|
|
123
|
-
rawWaste ,
|
|
124
|
-
opts ,
|
|
125
|
-
) {
|
|
126
|
-
const est = estimateSingleStage(wasteMs, stageId, stages , occupancy, opts);
|
|
127
|
-
if (est) return { basis: est.basis, wallClock: est.wallClock, estimateMethod, rawWaste };
|
|
128
|
-
return costOnly(estimateMethod, rawWaste); // stage excluded from the sweep (duration <= 0)
|
|
129
|
-
}
|
|
130
|
-
|
|
131
|
-
function stageMappableWasteOrCostOnly(
|
|
132
|
-
wasteMs ,
|
|
133
|
-
stageIds ,
|
|
134
|
-
stages ,
|
|
135
|
-
occupancy ,
|
|
136
|
-
) {
|
|
137
|
-
const rawWaste = wasteMs > 0 ? { value: wasteMs, unit: 'ms' } : undefined;
|
|
138
|
-
if (!stageIds || stageIds.length === 0) {
|
|
139
|
-
return costOnly('modeled', rawWaste);
|
|
140
|
-
}
|
|
141
|
-
// One waste event spread over a span of stages, not N independent wastes: apportion evenly so
|
|
142
|
-
// estimateMultiStage's union cap doesn't absorb the same amount claimed once per stage.
|
|
143
|
-
const perStageWasteMs = wasteMs / stageIds.length;
|
|
144
|
-
const wasteMsByStage = new Map(stageIds.map((id) => [id, perStageWasteMs]));
|
|
145
|
-
const est = estimateMultiStage(stageIds, wasteMsByStage, stages , occupancy);
|
|
146
|
-
if (!est) return costOnly('modeled', rawWaste); // every stage excluded from the sweep
|
|
147
|
-
return { basis: est.basis, wallClock: est.wallClock, estimateMethod: 'modeled', rawWaste };
|
|
148
|
-
}
|
|
149
|
-
|
|
150
|
-
/** Per-finding-type dispatch. A type with no case stays uncovered (no impactEstimate attached);
|
|
151
|
-
* docs/architecture.md's Impact estimation table is the authoritative completeness check, not this switch. */
|
|
152
|
-
function computeEstimateForFinding(
|
|
153
|
-
finding ,
|
|
154
|
-
stages ,
|
|
155
|
-
occupancy ,
|
|
156
|
-
totalCores ,
|
|
157
|
-
) {
|
|
158
|
-
switch (finding.type) {
|
|
159
|
-
case 'retryWaste': {
|
|
160
|
-
// The waste figure lives on the Stage, not the Finding: the detector only re-publishes it as metric/value.
|
|
161
|
-
if (finding.stageId == null) return null;
|
|
162
|
-
const stage = stages.get(finding.stageId);
|
|
163
|
-
if (!stage) return null;
|
|
164
|
-
const wasteMs = (stage.retryWasteMs ) ?? 0;
|
|
165
|
-
const wallClockMs = retryWallClockMs(stage);
|
|
166
|
-
return singleStageImpact(wallClockMs, finding.stageId, stages, occupancy,
|
|
167
|
-
wallClockMs === wasteMs ? 'measured' : 'modeled', { value: wasteMs, unit: 'ms' });
|
|
168
|
-
}
|
|
169
|
-
case 'speculationWaste': {
|
|
170
|
-
if (finding.stageId == null) return null;
|
|
171
|
-
const stage = stages.get(finding.stageId);
|
|
172
|
-
if (!stage) return null;
|
|
173
|
-
const wasteMs = (stage.speculationWasteMs ) ?? 0;
|
|
174
|
-
return singleStageImpact(wasteMs, finding.stageId, stages, occupancy, 'measured', { value: wasteMs, unit: 'ms' });
|
|
175
|
-
}
|
|
176
|
-
case 'coldStart': {
|
|
177
|
-
// The detector reports the gap as `metric: 'startupGapSeconds', value: <seconds>`.
|
|
178
|
-
if (typeof finding.value !== 'number') return null;
|
|
179
|
-
const wasteMs = finding.value * 1000;
|
|
180
|
-
// Time before any task starts can never overlap any stage; a genuine unclipped point estimate,
|
|
181
|
-
// not tied to any stage's gate (coldStart is app-scoped, stageId: null).
|
|
182
|
-
return { basis: 'serial', wallClock: { low: wasteMs, high: wasteMs }, estimateMethod: 'measured' };
|
|
183
|
-
}
|
|
184
|
-
case 'gc': {
|
|
185
|
-
// The low-GC branch is an over-provisioning signal whose fix (less executor memory) raises
|
|
186
|
-
// GC rather than recovering it: the stage's GC time is no saving there, so no waste model.
|
|
187
|
-
if (finding.direction === 'low') return costOnly('none');
|
|
188
|
-
if (finding.stageId == null) return null;
|
|
189
|
-
const stage = stages.get(finding.stageId);
|
|
190
|
-
if (!stage) return null;
|
|
191
|
-
const stageDurationMs = (stage.completedAt ?? 0) - (stage.submittedAt ?? 0);
|
|
192
|
-
const executorRunTime = stage.executorRunTime ?? 0;
|
|
193
|
-
const jvmGCTime = stage.jvmGCTime ?? 0;
|
|
194
|
-
// The raw cross-task core-time sum, before any conversion: the one figure here
|
|
195
|
-
// that is straight from the log rather than modeled.
|
|
196
|
-
const rawWaste = { value: jvmGCTime, unit: 'coreMs' } ;
|
|
197
|
-
if (executorRunTime <= 0 || stageDurationMs <= 0) {
|
|
198
|
-
return costOnly('modeled', rawWaste);
|
|
199
|
-
}
|
|
200
|
-
const avgConcurrency = executorRunTime / stageDurationMs;
|
|
201
|
-
// jvmGCTime is a cross-task core-time sum (same shape as executorRunTime); dividing by the
|
|
202
|
-
// stage's average concurrency converts it to an approximate wall-clock figure. Modeled, not exact.
|
|
203
|
-
const wasteMs = jvmGCTime / avgConcurrency;
|
|
204
|
-
return singleStageImpact(wasteMs, finding.stageId, stages, occupancy, 'modeled', rawWaste);
|
|
205
|
-
}
|
|
206
|
-
case 'skew': {
|
|
207
|
-
if (finding.stageId == null) return null;
|
|
208
|
-
const stage = stages.get(finding.stageId);
|
|
209
|
-
if (!stage) return null;
|
|
210
|
-
const p50 = stage.taskDurationP50 ?? 0;
|
|
211
|
-
// computeSkewRatio's own metric labels (src/detectors.ts): 'P95/median' or 'max/median'.
|
|
212
|
-
const usesP95Branch = finding.metric === 'P95/median';
|
|
213
|
-
const singleDelta = Math.max(0, usesP95Branch ? (stage.taskDurationP95 ?? 0) - p50 : (stage.taskDurationMax ?? 0) - p50);
|
|
214
|
-
const wasteMs = tailRecoveryMs(stage, singleDelta);
|
|
215
|
-
// Fixing the skew still waits on the longest task it leaves, as for straggler.
|
|
216
|
-
return singleStageImpact(wasteMs, finding.stageId, stages, occupancy, 'measured', { value: wasteMs, unit: 'ms' },
|
|
217
|
-
{ ...TAIL_CLAIM, removedCoreWorkMs: tailRemovedWorkMs(stage, singleDelta), longestTaskAfterFixMs: stragglerFixLongestTaskMs(stage) });
|
|
218
|
-
}
|
|
219
|
-
case 'straggler':
|
|
220
|
-
case 'stageShape': {
|
|
221
|
-
// Shared case for two finding types. straggler (no `rule` field) falls through to the max-P50
|
|
222
|
-
// computation below; every stageShape rule returns early (not `break`, which would fall off
|
|
223
|
-
// the switch and return undefined instead of null since the switch is the function's last statement).
|
|
224
|
-
if (finding.type === 'stageShape') {
|
|
225
|
-
if (finding.rule === 'lowParallelism') {
|
|
226
|
-
const stage = stages.get(finding.stageId );
|
|
227
|
-
if (!stage) return null;
|
|
228
|
-
const stageDurationMs = (stage.completedAt ?? 0) - (stage.submittedAt ?? 0);
|
|
229
|
-
const idleCoreMs =
|
|
230
|
-
Math.max(0, ((finding.totalCores ) ?? 0) - (stage.taskCount ?? 0)) * stageDurationMs;
|
|
231
|
-
// Real per-stage data (cores, task count, duration), no assumed constant.
|
|
232
|
-
return costOnly('measured', { value: idleCoreMs, unit: 'coreMs' });
|
|
233
|
-
}
|
|
234
|
-
if (finding.rule === 'dataExplosion') {
|
|
235
|
-
const stage = stages.get(finding.stageId );
|
|
236
|
-
if (!stage) return null;
|
|
237
|
-
const excessBytes = Math.max(0, (stage.outputBytes ?? 0) - (stage.inputBytes ?? 0));
|
|
238
|
-
// Measured input/output byte counts, no assumed constant.
|
|
239
|
-
return costOnly('measured', { value: excessBytes, unit: 'bytes' });
|
|
240
|
-
}
|
|
241
|
-
if (finding.rule === 'taskStageSkew') {
|
|
242
|
-
const stage = stages.get(finding.stageId );
|
|
243
|
-
if (!stage) return null;
|
|
244
|
-
const totalCores = (finding.totalCores ) ?? 0;
|
|
245
|
-
const taskCount = stage.taskCount ?? 0;
|
|
246
|
-
// Cores idle during the straggler's tail, at achieved concurrency (not full cluster
|
|
247
|
-
// capacity, which is lowParallelism's territory): this rule's trigger forces the
|
|
248
|
-
// occupancy-clipped estimate to zero on every firing, so it's resourceOnly, not a wall-clock claim.
|
|
249
|
-
const idleCoreMs =
|
|
250
|
-
Math.max(0, Math.min(totalCores, taskCount) - 1) *
|
|
251
|
-
Math.max(0, (stage.taskDurationMax ?? 0) - (stage.taskDurationP50 ?? 0));
|
|
252
|
-
return costOnly('measured', { value: idleCoreMs, unit: 'coreMs' });
|
|
253
|
-
}
|
|
254
|
-
return null;
|
|
255
|
-
}
|
|
256
|
-
if (finding.stageId == null) return null;
|
|
257
|
-
const stage = stages.get(finding.stageId);
|
|
258
|
-
if (!stage) return null;
|
|
259
|
-
const longestTaskAfterFixMs = stragglerFixLongestTaskMs(stage);
|
|
260
|
-
const singleDelta = Math.max(0, (stage.taskDurationMax ?? 0) - longestTaskAfterFixMs);
|
|
261
|
-
const wasteMs = tailRecoveryMs(stage, singleDelta);
|
|
262
|
-
return singleStageImpact(wasteMs, finding.stageId, stages, occupancy, 'measured', { value: wasteMs, unit: 'ms' },
|
|
263
|
-
{ ...TAIL_CLAIM, removedCoreWorkMs: tailRemovedWorkMs(stage, singleDelta), longestTaskAfterFixMs });
|
|
264
|
-
}
|
|
265
|
-
case 'slowHost': {
|
|
266
|
-
// Three duration-based shapes, each carrying its absolute-ms figure under a different field
|
|
267
|
-
// (`value` is always a ratio/share, never ms): the per-host mean branch (discriminated by
|
|
268
|
-
// `metric`), the duration-share branch (`variant`), and the multiDim taskTime dimension. Every
|
|
269
|
-
// byte-based multiDim dimension has no absolute figure today, so it stays informational.
|
|
270
|
-
const absoluteMs =
|
|
271
|
-
finding.metric === 'hostMeanRatio' || finding.variant === 'durationShare'
|
|
272
|
-
? (finding.hostMeanMs )
|
|
273
|
-
: finding.variant === 'multiDim' && finding.dimension === 'taskTime'
|
|
274
|
-
? (finding.execMaxValue )
|
|
275
|
-
: null;
|
|
276
|
-
if (absoluteMs == null) {
|
|
277
|
-
return costOnly('none'); // byte-based multiDim dims: no absolute figure today, no model applied
|
|
278
|
-
}
|
|
279
|
-
if (finding.stageId == null) return null;
|
|
280
|
-
const stage = stages.get(finding.stageId);
|
|
281
|
-
if (!stage) return null;
|
|
282
|
-
const wasteMs = Math.max(0, absoluteMs - (stage.taskDurationP50 ?? 0));
|
|
283
|
-
return singleStageImpact(wasteMs, finding.stageId, stages, occupancy, 'measured', { value: wasteMs, unit: 'ms' });
|
|
284
|
-
}
|
|
285
|
-
case 'duplicatePlanSubtree': {
|
|
286
|
-
const stageIds = finding.stageIds ;
|
|
287
|
-
if (!stageIds || stageIds.length === 0) return null;
|
|
288
|
-
// The detector reports subtreeOccurrences >= 2. Only repeats past the first are redundant:
|
|
289
|
-
// computing the subtree once is real work, so waste is (occurrences-1)/occurrences of the stages' time.
|
|
290
|
-
const occurrences = typeof finding.value === 'number' ? finding.value : 0;
|
|
291
|
-
// Defensive only: minOccurrences guarantees occurrences >= 2 on real data; a malformed-value fallback.
|
|
292
|
-
if (occurrences < 2) return costOnly('none');
|
|
293
|
-
// Same-shaped repeats whose details differ compute different data: nothing is known to be
|
|
294
|
-
// recomputed, so there is no time to claim.
|
|
295
|
-
if (finding.occurrencesIdentical === false) return costOnly('none');
|
|
296
|
-
// Each stage contributes the share of its operators inside the repeated subtree: a stage it
|
|
297
|
-
// shares with other operators (the consuming join, the join's other side) isn't all its
|
|
298
|
-
// time, and claiming whole stages let sibling groups claim the same stage twice. Findings
|
|
299
|
-
// built without the field (hand-made fixtures) count every linked stage whole.
|
|
300
|
-
const shares = (finding.stageShares ?? null) ;
|
|
301
|
-
const redundantFraction = (occurrences - 1) / occurrences;
|
|
302
|
-
const wasteMsByStage = new Map ();
|
|
303
|
-
for (const id of stageIds) {
|
|
304
|
-
const s = stages.get(id);
|
|
305
|
-
const share = shares ? (shares[id] ?? 0) : 1;
|
|
306
|
-
if (s && share > 0) {
|
|
307
|
-
// Time with tasks running, not submit-to-complete: a stage left waiting for cores
|
|
308
|
-
// (2491s open, 60s of tasks on a real log) isn't recomputing anything while it waits.
|
|
309
|
-
const activeMs = s.taskActiveMs ?? Math.max(0, (s.completedAt ?? 0) - (s.submittedAt ?? 0));
|
|
310
|
-
wasteMsByStage.set(id, activeMs * share * redundantFraction);
|
|
311
|
-
}
|
|
312
|
-
}
|
|
313
|
-
// No operator of the subtree ran in a known stage: no time to attribute.
|
|
314
|
-
if (wasteMsByStage.size === 0) return costOnly('none');
|
|
315
|
-
const totalWasteMs = [...wasteMsByStage.values()].reduce((sum, ms) => sum + ms, 0);
|
|
316
|
-
const rawWaste = { value: totalWasteMs, unit: 'ms' };
|
|
317
|
-
const est = estimateMultiStage([...wasteMsByStage.keys()], wasteMsByStage, stages , occupancy);
|
|
318
|
-
if (!est) return costOnly('measured', rawWaste);
|
|
319
|
-
return { basis: est.basis, wallClock: est.wallClock, estimateMethod: 'measured', rawWaste };
|
|
320
|
-
}
|
|
321
|
-
case 'shuffle': {
|
|
322
|
-
if (finding.stageId == null) return null;
|
|
323
|
-
const stage = stages.get(finding.stageId);
|
|
324
|
-
if (!stage) return null;
|
|
325
|
-
const shuffleReadBytes = stage.shuffleReadBytes ?? 0;
|
|
326
|
-
const modeledMs = (shuffleReadBytes / (SHUFFLE_THROUGHPUT_BPS * stageIoParallelism(stage))) * 1000;
|
|
327
|
-
// The link model can't see whether the reads stalled the tasks: capped at the fetch wait the
|
|
328
|
-
// tasks measured, the claim never exceeds what the stage spent blocked on the network.
|
|
329
|
-
const measuredMs = fetchWaitWallClockMs(stage);
|
|
330
|
-
const wasteMs = measuredMs == null ? modeledMs : Math.min(modeledMs, measuredMs);
|
|
331
|
-
// rawWaste: the measured byte volume behind the modeled figure.
|
|
332
|
-
return singleStageImpact(wasteMs, finding.stageId, stages, occupancy,
|
|
333
|
-
measuredMs != null && measuredMs < modeledMs ? 'measured' : 'modeled', { value: shuffleReadBytes, unit: 'bytes' });
|
|
334
|
-
}
|
|
335
|
-
case 'spill': {
|
|
336
|
-
if (finding.stageId == null) return null;
|
|
337
|
-
const stage = stages.get(finding.stageId);
|
|
338
|
-
if (!stage) return null;
|
|
339
|
-
const diskBytesSpilled = stage.diskBytesSpilled ?? 0;
|
|
340
|
-
const wasteMs = (diskBytesSpilled / (SPILL_IO_THROUGHPUT_BPS * stageIoParallelism(stage))) * 1000;
|
|
341
|
-
// Surfaces the number the formula uses: the displayed metric is memoryBytesSpilled, but disk spill costs the I/O time.
|
|
342
|
-
return singleStageImpact(wasteMs, finding.stageId, stages, occupancy, 'modeled', { value: diskBytesSpilled, unit: 'bytes' });
|
|
343
|
-
}
|
|
344
|
-
case 'stageSlowness': {
|
|
345
|
-
if (finding.stageId == null) return null;
|
|
346
|
-
const stage = stages.get(finding.stageId);
|
|
347
|
-
if (!stage) return null;
|
|
348
|
-
// The recommended fix is more partitions, which only helps a stage that ran fewer tasks than
|
|
349
|
-
// the cluster has cores: the time its tasks were running could then spread over up to
|
|
350
|
-
// totalCores (lowShuffleParallelism's shape). Time the stage sat open with no task running
|
|
351
|
-
// is queueing no partition count recovers. Splitting partitions splits the longest task
|
|
352
|
-
// too, hence TAIL_CLAIM's post-fix floor. Unknown cluster size: no defensible figure.
|
|
353
|
-
if (totalCores <= 0) return costOnly('modeled');
|
|
354
|
-
// A stage that read no input and no shuffle, its tasks idle waiting on an external system,
|
|
355
|
-
// gains nothing from more partitions (a 1-task JDBC count stage open 27 minutes on 5s of CPU
|
|
356
|
-
// was claimed 99% recoverable): claim 0. Stages that read bytes keep their claim.
|
|
357
|
-
const readBytes = (stage.inputBytes ?? 0) + (stage.shuffleReadBytes ?? 0);
|
|
358
|
-
const activeMs = typeof stage.taskActiveMs === 'number'
|
|
359
|
-
? stage.taskActiveMs
|
|
360
|
-
: Math.max(0, (stage.completedAt ?? 0) - (stage.submittedAt ?? 0));
|
|
361
|
-
const taskCount = stage.taskCount ?? 0;
|
|
362
|
-
const wasteMs = readBytes <= 0 && tasksMostlyIdle(stage) ? 0 : activeMs * Math.max(0, 1 - taskCount / totalCores);
|
|
363
|
-
return singleStageImpact(wasteMs, finding.stageId, stages, occupancy, 'modeled', { value: wasteMs, unit: 'ms' }, TAIL_CLAIM);
|
|
364
|
-
}
|
|
365
|
-
case 'partitionSizing': {
|
|
366
|
-
if (finding.stageId == null) return null;
|
|
367
|
-
const stage = stages.get(finding.stageId);
|
|
368
|
-
if (!stage) return null;
|
|
369
|
-
let wasteMs = 0;
|
|
370
|
-
if (finding.rule === 'maxPartitionTooBig') {
|
|
371
|
-
wasteMs = ((stage.shuffleReadMax ?? 0) / SHUFFLE_THROUGHPUT_BPS) * 1000;
|
|
372
|
-
} else if (finding.rule === 'shufflePartitionSkew') {
|
|
373
|
-
const delta = Math.max(0, (stage.shuffleReadMax ?? 0) - (stage.shuffleReadP50 ?? 0));
|
|
374
|
-
wasteMs = (delta / SHUFFLE_THROUGHPUT_BPS) * 1000;
|
|
375
|
-
} else if (finding.rule === 'lowShuffleParallelism') {
|
|
376
|
-
const targetTaskCount = Math.ceil((stage.shuffleReadBytes ?? 0) / IDEAL_BYTES_PER_PARTITION_TASK);
|
|
377
|
-
const taskCount = stage.taskCount ?? 0;
|
|
378
|
-
if (targetTaskCount > taskCount && taskCount > 0) {
|
|
379
|
-
const stageDurationMs = Math.max(0, (stage.completedAt ?? 0) - (stage.submittedAt ?? 0));
|
|
380
|
-
// Too few shuffle partitions means each task processes more than the ideal bytes,
|
|
381
|
-
// serializing work more partitions would run concurrently: the waste is that serialized
|
|
382
|
-
// work, not the scheduling cost of tasks you'd add (adding tasks incurs overhead, recovers
|
|
383
|
-
// nothing). Model the achievable duration at target parallelism by scaling down proportionally.
|
|
384
|
-
wasteMs = stageDurationMs * (1 - taskCount / targetTaskCount);
|
|
385
|
-
}
|
|
386
|
-
} else {
|
|
387
|
-
return null;
|
|
388
|
-
}
|
|
389
|
-
return singleStageImpact(wasteMs, finding.stageId, stages, occupancy, 'modeled', { value: wasteMs, unit: 'ms' });
|
|
390
|
-
}
|
|
391
|
-
case 'tinyTask': {
|
|
392
|
-
if (finding.stageId == null) return null;
|
|
393
|
-
const stage = stages.get(finding.stageId);
|
|
394
|
-
if (!stage) return null;
|
|
395
|
-
const taskCount = stage.taskCount ?? 0;
|
|
396
|
-
const excessTaskCount = Math.max(0, taskCount - Math.round(taskCount / 10));
|
|
397
|
-
const measured = measuredTaskOverhead(stage);
|
|
398
|
-
if (measured) {
|
|
399
|
-
// Coalescing to a tenth of the tasks removes the excess tasks' per-task overhead: core
|
|
400
|
-
// time spent in parallel, so wall-clock at the stage's achieved concurrency (floored at
|
|
401
|
-
// 1: a mostly-idle stage can't save more wall-clock than the task time it removes).
|
|
402
|
-
const wasteMs = (excessTaskCount * measured.perTaskMs) / Math.max(1, measured.concurrency);
|
|
403
|
-
return singleStageImpact(wasteMs, finding.stageId, stages, occupancy, 'measured', { value: wasteMs, unit: 'ms' });
|
|
404
|
-
}
|
|
405
|
-
const wasteMs = excessTaskCount * TASK_SCHEDULING_OVERHEAD_MS;
|
|
406
|
-
return singleStageImpact(wasteMs, finding.stageId, stages, occupancy, 'modeled', { value: wasteMs, unit: 'ms' });
|
|
407
|
-
}
|
|
408
|
-
case 'smallFiles': {
|
|
409
|
-
const fileMs = ((finding.fileCount ) ?? 0) * FILE_OPEN_OVERHEAD_MS;
|
|
410
|
-
// A read's files are opened by the scan's tasks, in parallel: spread the per-file cost over
|
|
411
|
-
// the most tasks its stages ran at once (91344 files x 10ms is 913s, claimed against a 117s
|
|
412
|
-
// stage that ran 314 tasks at once). A write keeps the serial sum: the job commit moves each
|
|
413
|
-
// output file on the driver, one after another.
|
|
414
|
-
const stageIds = finding.stageIds ;
|
|
415
|
-
let slots = 1;
|
|
416
|
-
if (finding.direction === 'read') {
|
|
417
|
-
for (const id of stageIds ?? []) slots = Math.max(slots, stages.get(id)?.peakConcurrentTasks ?? 1);
|
|
418
|
-
}
|
|
419
|
-
return stageMappableWasteOrCostOnly(fileMs / slots, stageIds, stages, occupancy);
|
|
420
|
-
}
|
|
421
|
-
case 'overBroadcast': {
|
|
422
|
-
// metric: 'broadcastBytes', value: <bytes>.
|
|
423
|
-
const wasteMs = (((finding.value ) ?? 0) / BROADCAST_BANDWIDTH_BPS) * 1000;
|
|
424
|
-
return stageMappableWasteOrCostOnly(wasteMs, finding.stageIds , stages, occupancy);
|
|
425
|
-
}
|
|
426
|
-
case 'underBroadcast': {
|
|
427
|
-
// metric: 'smallerSideBytes', value: <bytes of the smaller join side>.
|
|
428
|
-
const wasteMs = (((finding.value ) ?? 0) / BROADCAST_BANDWIDTH_BPS) * 1000;
|
|
429
|
-
return stageMappableWasteOrCostOnly(wasteMs, finding.stageIds , stages, occupancy);
|
|
430
|
-
}
|
|
431
|
-
case 'memoryUtilization': {
|
|
432
|
-
// The wasteModel variant reports metric: 'wastedMBSeconds', value: <MB-seconds>.
|
|
433
|
-
if (finding.variant === 'wasteModel' && typeof finding.value === 'number') {
|
|
434
|
-
return costOnly('measured', { value: finding.value, unit: 'mbSeconds' });
|
|
435
|
-
}
|
|
436
|
-
if (finding.variant === 'idleCores') {
|
|
437
|
-
// Idle core-time priced as memory held but unused: the same MB-seconds unit as wasteModel, so comparable.
|
|
438
|
-
const idleRateFraction = finding.idleRateFraction ;
|
|
439
|
-
const allocatedMB = finding.allocatedMB ;
|
|
440
|
-
const peakExecutors = finding.peakExecutors ;
|
|
441
|
-
const appDurationMs = finding.appDurationMs ;
|
|
442
|
-
if (idleRateFraction != null && allocatedMB != null && peakExecutors != null && appDurationMs != null) {
|
|
443
|
-
const wastedMBSeconds = idleRateFraction * allocatedMB * peakExecutors * (appDurationMs / 1000);
|
|
444
|
-
return costOnly('modeled', { value: wastedMBSeconds, unit: 'mbSeconds' });
|
|
445
|
-
}
|
|
446
|
-
return costOnly('modeled');
|
|
447
|
-
}
|
|
448
|
-
// Only the over-provisioned band is a waste; the near-capacity band is an OOM-risk signal with
|
|
449
|
-
// no magnitude, and the dataUnavailable shape has no inputs: both stay informational.
|
|
450
|
-
if (finding.variant === 'memoryBand' && finding.rule === 'heapOverProvisioned') {
|
|
451
|
-
const allocatedBytes = finding.allocatedBytes ;
|
|
452
|
-
const heap = finding.heap ;
|
|
453
|
-
const appDurationMs = finding.appDurationMs ;
|
|
454
|
-
if (allocatedBytes != null && heap != null && appDurationMs != null) {
|
|
455
|
-
const unusedMB = (allocatedBytes - heap) / (1024 * 1024);
|
|
456
|
-
const wastedMBSeconds = unusedMB * (appDurationMs / 1000);
|
|
457
|
-
return costOnly('modeled', { value: wastedMBSeconds, unit: 'mbSeconds' });
|
|
458
|
-
}
|
|
459
|
-
}
|
|
460
|
-
return costOnly('modeled');
|
|
461
|
-
}
|
|
462
|
-
case 'utilization': {
|
|
463
|
-
const fraction = finding.utilizationFraction ;
|
|
464
|
-
const appDurationMs = finding.appDurationMs ;
|
|
465
|
-
const totalCores = finding.totalCores ;
|
|
466
|
-
if (fraction == null || appDurationMs == null || totalCores == null) {
|
|
467
|
-
return costOnly('measured');
|
|
468
|
-
}
|
|
469
|
-
const idleCoreHours = (1 - fraction) * appDurationMs * totalCores / 3.6e6;
|
|
470
|
-
return costOnly('measured', { value: idleCoreHours, unit: 'coreHours' });
|
|
471
|
-
}
|
|
472
|
-
case 'coreLocality': {
|
|
473
|
-
const nonLocal = (finding.nonLocalTaskCount ) ?? 0;
|
|
474
|
-
const coreMs = nonLocal * NETWORK_FETCH_PENALTY_MS;
|
|
475
|
-
return costOnly('modeled', { value: coreMs, unit: 'coreMs' });
|
|
476
|
-
}
|
|
477
|
-
case 'autoscalingChurn': {
|
|
478
|
-
const shortLived = (finding.shortLivedExecutorCount ) ?? 0;
|
|
479
|
-
const executorHours = (shortLived * EXECUTOR_STARTUP_OVERHEAD_MS) / 3.6e6;
|
|
480
|
-
return costOnly('modeled', { value: executorHours, unit: 'coreHours' });
|
|
481
|
-
}
|
|
482
|
-
case 'configAudit': {
|
|
483
|
-
return costOnly('none'); // purely informational: no waste model applied
|
|
484
|
-
}
|
|
485
|
-
case 'jobFailureRate': {
|
|
486
|
-
const failedJobs = (finding.failedJobs ) ?? 0;
|
|
487
|
-
const avgJobDurationMs = (finding.avgJobDurationMs ) ?? 0;
|
|
488
|
-
const coreHoursIsh = (failedJobs * avgJobDurationMs) / 3.6e6;
|
|
489
|
-
return costOnly('modeled', { value: coreHoursIsh, unit: 'coreHours' });
|
|
490
|
-
}
|
|
491
|
-
case 'cachingOpportunity': {
|
|
492
|
-
const totalReadBytes = (finding.totalReadBytes ) ?? 0;
|
|
493
|
-
const wasteMs = (totalReadBytes / RE_READ_THROUGHPUT_BPS) * 1000;
|
|
494
|
-
return costOnly('modeled', { value: wasteMs, unit: 'ms' });
|
|
495
|
-
}
|
|
496
|
-
case 'cacheUtilization': {
|
|
497
|
-
const memorySize = (finding.memorySize ) ?? 0;
|
|
498
|
-
const diskSize = (finding.diskSize ) ?? 0;
|
|
499
|
-
const numCachedPartitions = (finding.numCachedPartitions ) ?? 0;
|
|
500
|
-
const numPartitions = (finding.numPartitions ) ?? 0;
|
|
501
|
-
const numUncachedPartitions = Math.max(0, numPartitions - numCachedPartitions);
|
|
502
|
-
const cachedBytes = memorySize + diskSize;
|
|
503
|
-
// Extrapolate never-cached partitions' size from the CACHED partitions' average (uncached/
|
|
504
|
-
// cached, not uncached/total: numCachedPartitions produced cachedBytes). diskSize is added
|
|
505
|
-
// once more: those bytes are cached but on disk, so re-reading them still costs I/O like an uncached partition.
|
|
506
|
-
const uncachedBytes = numCachedPartitions > 0 ? (cachedBytes / numCachedPartitions) * numUncachedPartitions : 0;
|
|
507
|
-
const uncachedOrSpilledBytes = uncachedBytes + diskSize;
|
|
508
|
-
const wasteMs = (uncachedOrSpilledBytes / RE_READ_THROUGHPUT_BPS) * 1000;
|
|
509
|
-
return costOnly('modeled', { value: wasteMs, unit: 'ms' });
|
|
510
|
-
}
|
|
511
|
-
case 'stageFailed':
|
|
512
|
-
case 'failures':
|
|
513
|
-
case 'incompleteRun': {
|
|
514
|
-
return costOnly('none'); // purely informational: no waste model applied
|
|
515
|
-
}
|
|
516
|
-
default:
|
|
517
|
-
return null;
|
|
518
|
-
}
|
|
519
|
-
}
|
|
520
|
-
|
|
521
|
-
export function estimateImpact(findings , stages , totalCores = 0) {
|
|
522
|
-
const occupancy = computeOccupancy(stages , totalCores);
|
|
7
|
+
/** Attaches each finding's `impactEstimate` from its entry's estimate(), in place. A null estimate
|
|
8
|
+
* leaves the finding uncovered (no impactEstimate). `ctx` is analyze()'s one occupancy sweep. */
|
|
9
|
+
export function estimateImpact(findings , ctx ) {
|
|
523
10
|
for (const f of findings) {
|
|
524
|
-
const estimate =
|
|
11
|
+
const estimate = ENTRY_BY_TYPE.get(f.type)?.estimate(f, ctx);
|
|
525
12
|
if (estimate) f.impactEstimate = estimate;
|
|
526
13
|
}
|
|
527
14
|
return findings;
|
|
@@ -0,0 +1,114 @@
|
|
|
1
|
+
// Savings figures as every surface prints them: the dashboard's widgets and verdict, and the
|
|
2
|
+
// CLI/MCP evidence report. One formatter set, so units and rounding never drift between paths.
|
|
3
|
+
import { fmtMs, formatRawWaste, formatWallClockRange, readsAsZero } from './format-utils.js';
|
|
4
|
+
|
|
5
|
+
|
|
6
|
+
/** A finding's one-line savings figure: the wall-clock range for a time-based
|
|
7
|
+
* finding, the raw resource figure for a `resourceOnly` one, or nothing for a
|
|
8
|
+
* purely informational estimate or a raw figure that rounds to zero ("0.0
|
|
9
|
+
* core-h" reads as a measured nothing). */
|
|
10
|
+
export function impactFigure(finding ) {
|
|
11
|
+
const estimate = finding.impactEstimate;
|
|
12
|
+
if (!estimate) return null;
|
|
13
|
+
if (estimate.wallClock) return formatWallClockRange(estimate.wallClock.low, estimate.wallClock.high);
|
|
14
|
+
if (estimate.rawWaste) {
|
|
15
|
+
const text = formatRawWaste(estimate.rawWaste);
|
|
16
|
+
return readsAsZero(text) ? null : text;
|
|
17
|
+
}
|
|
18
|
+
return null;
|
|
19
|
+
}
|
|
20
|
+
|
|
21
|
+
/** What a raw-waste figure counts, by its unit, as the words that follow it. */
|
|
22
|
+
export function rawWasteMeaning(unit ) {
|
|
23
|
+
switch (unit) {
|
|
24
|
+
case 'mbSeconds': return 'of unused executor memory';
|
|
25
|
+
case 'coreHours':
|
|
26
|
+
case 'coreMs': return 'of core time';
|
|
27
|
+
case 'bytes': return 'of extra data written';
|
|
28
|
+
case 'ms': return 'of task time';
|
|
29
|
+
default: return null;
|
|
30
|
+
}
|
|
31
|
+
}
|
|
32
|
+
|
|
33
|
+
/** What a savings figure counts, as the words that follow it: run time for a
|
|
34
|
+
* wall-clock claim, or the resource a cost-only (`resourceOnly`) figure
|
|
35
|
+
* measures. A time figure and a capacity figure look alike ("58.6s",
|
|
36
|
+
* "0.7 core-h") but only the first shortens the run. Null when the finding
|
|
37
|
+
* shows no figure. */
|
|
38
|
+
export function savingsMeaning(finding ) {
|
|
39
|
+
const estimate = finding.impactEstimate;
|
|
40
|
+
if (!estimate) return null;
|
|
41
|
+
if (estimate.wallClock) return 'of run time';
|
|
42
|
+
return rawWasteMeaning(estimate.rawWaste?.unit);
|
|
43
|
+
}
|
|
44
|
+
|
|
45
|
+
/** A finding's "Potential savings" figure as the widget board shows it: the
|
|
46
|
+
* wall-clock range, or the raw waste only when there is no wall-clock claim,
|
|
47
|
+
* never both (they would read as two competing numbers). A zero-value
|
|
48
|
+
* estimate is suppressed like an informational one: "0s" reads as a measured
|
|
49
|
+
* figure. `meaning` says what the shown figure counts. */
|
|
50
|
+
export function impactEstimateFigure(estimate ) {
|
|
51
|
+
if (!estimate) return null;
|
|
52
|
+
const highText = estimate.wallClock && estimate.wallClock.high > 0 ? fmtMs(estimate.wallClock.high) : null;
|
|
53
|
+
if (highText && !readsAsZero(highText)) {
|
|
54
|
+
return { text: formatWallClockRange(estimate.wallClock .low, estimate.wallClock .high), meaning: 'of run time' };
|
|
55
|
+
}
|
|
56
|
+
const rawWasteText = estimate.rawWaste && estimate.rawWaste.value > 0 ? formatRawWaste(estimate.rawWaste) : null;
|
|
57
|
+
if (rawWasteText && !readsAsZero(rawWasteText)) return { text: rawWasteText, meaning: rawWasteMeaning(estimate.rawWaste .unit) };
|
|
58
|
+
return null;
|
|
59
|
+
}
|
|
60
|
+
|
|
61
|
+
/** Compact single-value form for dense lists (the stage table's finding chips,
|
|
62
|
+
* the stage dialog): the high-end wall-clock figure, or the raw-waste figure
|
|
63
|
+
* when there's no wall-clock claim, or `null` for a purely informational or a
|
|
64
|
+
* zero-value estimate (a `0s` in the spot a real estimate goes would read as a
|
|
65
|
+
* measured nothing). */
|
|
66
|
+
export function impactEstimateCompact(estimate ) {
|
|
67
|
+
if (!estimate) return null;
|
|
68
|
+
if (estimate.wallClock) {
|
|
69
|
+
if (estimate.wallClock.high <= 0) return null;
|
|
70
|
+
const text = fmtMs(estimate.wallClock.high);
|
|
71
|
+
return readsAsZero(text) ? null : text;
|
|
72
|
+
}
|
|
73
|
+
if (estimate.rawWaste) {
|
|
74
|
+
if (estimate.rawWaste.value <= 0) return null;
|
|
75
|
+
const text = formatRawWaste(estimate.rawWaste);
|
|
76
|
+
return readsAsZero(text) ? null : text;
|
|
77
|
+
}
|
|
78
|
+
return null;
|
|
79
|
+
}
|
|
80
|
+
|
|
81
|
+
/** How a step's savings figure was derived, in one plain sentence for
|
|
82
|
+
* Advanced view: the estimate method, whether the stage ran alone (a
|
|
83
|
+
* near-point figure) or shared the cluster (a floor and an optimistic high),
|
|
84
|
+
* and the raw waste behind it. Null when the finding carries no estimate
|
|
85
|
+
* model (`estimateMethod: 'none'`), no estimate at all, or a figure that
|
|
86
|
+
* reads as zero (the step shows no savings then either). Uses the same
|
|
87
|
+
* formatting and zero rules as the step's own savings figure. */
|
|
88
|
+
export function estimateProvenance(finding ) {
|
|
89
|
+
const estimate = finding.impactEstimate;
|
|
90
|
+
if (!estimate || estimate.estimateMethod === 'none') return null;
|
|
91
|
+
const method = estimate.estimateMethod;
|
|
92
|
+
const rawWaste = estimate.rawWaste && estimate.rawWaste.value > 0 ? estimate.rawWaste : null;
|
|
93
|
+
const raw = rawWaste && !readsAsZero(formatRawWaste(rawWaste)) ? formatRawWaste(rawWaste) : null;
|
|
94
|
+
const wallClock = estimate.wallClock;
|
|
95
|
+
if (estimate.basis === 'resourceOnly') {
|
|
96
|
+
return raw ? `No run-time claim, ${method}. ${raw} was wasted, but it may not shorten the run.` : null;
|
|
97
|
+
}
|
|
98
|
+
if (!wallClock || wallClock.high <= 0) return null;
|
|
99
|
+
const highText = formatWallClockRange(wallClock.high, wallClock.high);
|
|
100
|
+
if (readsAsZero(highText)) return null;
|
|
101
|
+
let rawNote = '';
|
|
102
|
+
if (raw && rawWaste .unit !== 'ms') rawNote = ` Resource waste measured: ${raw}.`;
|
|
103
|
+
else if (raw && rawWaste .value > wallClock.high && raw !== highText) rawNote = ` Raw waste before the floor clipped it: ${raw}.`;
|
|
104
|
+
if (estimate.basis === 'serial') {
|
|
105
|
+
return `${highText}, ${method}. The stage ran effectively alone, so this is close to a point estimate.${rawNote}`;
|
|
106
|
+
}
|
|
107
|
+
if (estimate.basis === 'contended') {
|
|
108
|
+
const lowText = formatWallClockRange(wallClock.low, wallClock.low);
|
|
109
|
+
const range = formatWallClockRange(wallClock.low, wallClock.high);
|
|
110
|
+
const spread = lowText === highText ? 'its floor and optimistic high agree' : `${lowText} is the floor, ${highText} assumes the fix fully lands`;
|
|
111
|
+
return `${range}, ${method}. The stage shared the cluster with others: ${spread}.${rawNote}`;
|
|
112
|
+
}
|
|
113
|
+
return null;
|
|
114
|
+
}
|