sparkforensics-mcp 0.4.0 → 0.5.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (64) hide show
  1. package/package.json +1 -1
  2. package/vendor-core/allocation.js +106 -0
  3. package/vendor-core/analyzer.js +13 -13
  4. package/vendor-core/cli/budgets.js +23 -9
  5. package/vendor-core/cli/collect-run.js +11 -4
  6. package/vendor-core/cli/regression-budgets.js +83 -0
  7. package/vendor-core/comparison-verdict.js +22 -22
  8. package/vendor-core/core-source-hash.txt +1 -1
  9. package/vendor-core/detectors.js +202 -68
  10. package/vendor-core/docs-content/detection/cache.md +3 -2
  11. package/vendor-core/docs-content/detection/cfg.md +9 -8
  12. package/vendor-core/docs-content/detection/chrn.md +1 -2
  13. package/vendor-core/docs-content/detection/cold.md +4 -2
  14. package/vendor-core/docs-content/detection/fail.md +3 -2
  15. package/vendor-core/docs-content/detection/gc.md +3 -2
  16. package/vendor-core/docs-content/detection/host.md +2 -1
  17. package/vendor-core/docs-content/detection/local.md +1 -1
  18. package/vendor-core/docs-content/detection/mem.md +5 -2
  19. package/vendor-core/docs-content/detection/plan.md +2 -1
  20. package/vendor-core/docs-content/detection/sfail.md +2 -1
  21. package/vendor-core/docs-content/detection/shape.md +5 -4
  22. package/vendor-core/docs-content/detection/skew.md +3 -1
  23. package/vendor-core/docs-content/detection/slow.md +2 -2
  24. package/vendor-core/docs-content/detection/spec.md +2 -3
  25. package/vendor-core/docs-content/detection/spill.md +1 -1
  26. package/vendor-core/docs-site-config.js +1 -1
  27. package/vendor-core/effective-conf.js +107 -0
  28. package/vendor-core/efficiency-model.js +8 -6
  29. package/vendor-core/event-handlers.js +160 -47
  30. package/vendor-core/event-schemas.js +2 -0
  31. package/vendor-core/evidence-report.js +18 -10
  32. package/vendor-core/finding-generic-recommendation.js +20 -1
  33. package/vendor-core/finding-names.js +7 -0
  34. package/vendor-core/finding-presentation.js +61 -26
  35. package/vendor-core/finding-tag-help.js +1 -1
  36. package/vendor-core/finding-types.js +12 -0
  37. package/vendor-core/format-utils.js +4 -3
  38. package/vendor-core/impact-estimator.js +20 -2
  39. package/vendor-core/impact-format.js +14 -13
  40. package/vendor-core/impact-model.js +27 -5
  41. package/vendor-core/ingest.js +4 -2
  42. package/vendor-core/list-runs.js +5 -2
  43. package/vendor-core/mcp-tools.js +1 -1
  44. package/vendor-core/model-assembler.js +23 -1
  45. package/vendor-core/parser-worker.js +1 -1
  46. package/vendor-core/proxy.js +3 -1
  47. package/vendor-core/python-stage.js +25 -0
  48. package/vendor-core/recommendation-rollup.js +16 -9
  49. package/vendor-core/redact.js +51 -10
  50. package/vendor-core/remediation.js +20 -0
  51. package/vendor-core/run-comparison.js +43 -24
  52. package/vendor-core/run-interpretation.js +2 -1
  53. package/vendor-core/run-metrics.js +198 -0
  54. package/vendor-core/run-totals.js +24 -0
  55. package/vendor-core/run-verdict.js +3 -4
  56. package/vendor-core/scorecard-estimates.js +1 -0
  57. package/vendor-core/session-snapshot.js +7 -0
  58. package/vendor-core/shs-schemas.js +2 -2
  59. package/vendor-core/spark-memory.js +17 -0
  60. package/vendor-core/stage-plan-nodes.js +18 -0
  61. package/vendor-core/stage-quantiles.js +4 -0
  62. package/vendor-core/types.js +49 -1
  63. package/vendor-core/wasted-core-hours.js +10 -7
  64. package/vendor-core/write-targets.js +312 -0
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "sparkforensics-mcp",
3
- "version": "0.4.0",
3
+ "version": "0.5.0",
4
4
  "mcpName": "io.github.shuffle-works/sparkforensics-mcp",
5
5
  "type": "module",
6
6
  "license": "MIT",
@@ -0,0 +1,106 @@
1
+ import { parseSparkMemoryMB } from './spark-memory.js';
2
+
3
+
4
+ const MS_PER_HOUR = 3_600_000;
5
+ const MIB_PER_GIB = 1024;
6
+ // Spark's documented floor and factor for the default executor memory overhead
7
+ // (spark.executor.memoryOverhead = max(factor * executor memory, 384 MiB)).
8
+ const MIN_OVERHEAD_MIB = 384;
9
+ const DEFAULT_OVERHEAD_FACTOR = 0.1;
10
+
11
+
12
+  
13
+
14
+  
15
+
16
+
17
+
18
+
19
+
20
+
21
+
22
+
23
+
24
+
25
+ /** The latest timestamp the log records: the application start and end and every stage and
26
+ * executor event. Where a log with no ApplicationEnd stops. */
27
+ export function lastObservedTimestamp(input ) {
28
+ let last = null;
29
+ const take = (t ) => {
30
+ if (typeof t === 'number' && Number.isFinite(t) && t > 0 && (last == null || t > last)) last = t;
31
+ };
32
+ take(input.app?.startTime);
33
+ take(input.app?.endTime);
34
+ for (const s of input.stages.values()) { take(s.submittedAt); take(s.completedAt); }
35
+ for (const e of [...input.executors.added, ...input.executors.removed]) take(e.timestamp);
36
+ return last;
37
+ }
38
+
39
+ // Spark's default executor memory when spark.executor.memory is unset.
40
+ const DEFAULT_EXECUTOR_MEMORY_MIB = 1024;
41
+
42
+ // One container's memory in MiB, as Spark requests it from the cluster manager: executor heap,
43
+ // plus overhead, plus off-heap and PySpark worker memory when configured. Null when the log
44
+ // records no Spark properties (nothing to tell a default from a missing config) or a memory key it
45
+ // does record cannot be read.
46
+ function executorMemoryMiB(app ) {
47
+ const config = app?.config;
48
+ if (config == null) return null;
49
+ // undefined: key absent; null: present but unreadable.
50
+ const mib = (key ) => (config[key] == null ? undefined : parseSparkMemoryMB(config[key]));
51
+ const heap = mib('spark.executor.memory') ?? (config['spark.executor.memory'] == null ? DEFAULT_EXECUTOR_MEMORY_MIB : null);
52
+ if (heap == null) return null;
53
+
54
+ let overhead = mib('spark.executor.memoryOverhead');
55
+ if (overhead === undefined) overhead = mib('spark.yarn.executor.memoryOverhead'); // legacy key
56
+ if (overhead === undefined) {
57
+ const factor = Number.parseFloat(config['spark.executor.memoryOverheadFactor'] ?? '');
58
+ overhead = Math.max(MIN_OVERHEAD_MIB, Math.round(heap * (Number.isFinite(factor) && factor > 0 ? factor : DEFAULT_OVERHEAD_FACTOR)));
59
+ }
60
+ const offHeap = String(config['spark.memory.offHeap.enabled']).toLowerCase() === 'true' ? mib('spark.memory.offHeap.size') ?? 0 : 0;
61
+ const pyspark = mib('spark.executor.pyspark.memory') ?? 0;
62
+ if (overhead === null || offHeap === null || pyspark === null) return null;
63
+ return heap + overhead + offHeap + pyspark;
64
+ }
65
+
66
+ /** Allocated core-hours and memory GiB-hours from the executor lifecycle: each executor counts
67
+ * from its ExecutorAdded timestamp to its first later ExecutorRemoved timestamp. One with no
68
+ * removal closes at the application end when the log has one, else at the last timestamp the log
69
+ * records (a cut-off log). Cores are the ExecutorAdded event's Total Cores, else
70
+ * spark.executor.cores; memory per executor is spark.executor.memory (default 1g) plus the
71
+ * overhead (spark.executor.memoryOverhead, else the legacy spark.yarn.executor.memoryOverhead,
72
+ * else the larger of 384 MiB and spark.executor.memoryOverheadFactor, default 0.1, times the
73
+ * memory), plus spark.memory.offHeap.size when spark.memory.offHeap.enabled is true, plus
74
+ * spark.executor.pyspark.memory.
75
+ * Null, never 0, for a figure whose inputs the log lacks. */
76
+ export function computeAllocation(input ) {
77
+ const added = input.executors.added.filter((e) => e.kind === 'added');
78
+ if (added.length === 0) return { coreHours: null, memoryGbHours: null };
79
+ const removedAt = new Map ();
80
+ for (const e of input.executors.removed) {
81
+ if (e.kind !== 'removed') continue;
82
+ const times = removedAt.get(e.executorId);
83
+ if (times) times.push(e.timestamp); else removedAt.set(e.executorId, [e.timestamp]);
84
+ }
85
+ const closeAt = input.app?.endTime ?? lastObservedTimestamp(input);
86
+ const configuredCores = Number.parseInt(input.app?.config?.['spark.executor.cores'] ?? '', 10);
87
+ const memoryMiB = executorMemoryMiB(input.app);
88
+
89
+ let coreMs = 0;
90
+ let memoryMiBMs = 0;
91
+ let coresKnown = true;
92
+ const seen = new Set ();
93
+ for (const e of added) {
94
+ if (seen.has(e.executorId)) continue; // a replayed ExecutorAdded is the same executor
95
+ seen.add(e.executorId);
96
+ const removal = (removedAt.get(e.executorId) ?? []).filter((t) => t >= e.timestamp).sort((a, b) => a - b)[0];
97
+ const aliveMs = Math.max(0, (removal ?? closeAt ?? e.timestamp) - e.timestamp);
98
+ const cores = e.totalCores > 0 ? e.totalCores : Number.isFinite(configuredCores) ? configuredCores : null;
99
+ if (cores == null) coresKnown = false; else coreMs += cores * aliveMs;
100
+ memoryMiBMs += (memoryMiB ?? 0) * aliveMs;
101
+ }
102
+ return {
103
+ coreHours: coresKnown ? coreMs / MS_PER_HOUR : null,
104
+ memoryGbHours: memoryMiB != null ? memoryMiBMs / MIB_PER_GIB / MS_PER_HOUR : null,
105
+ };
106
+ }
@@ -88,32 +88,32 @@ export function findingId(f ) {
88
88
  return fnv1a(`${f.type}|${locationKey(f)}|${f.metric ?? ''}|${f.value ?? f.valueText ?? ''}|${disc}`);
89
89
  }
90
90
 
91
- // skew's max/median branch (stage.taskCount below minTasksForP95) and straggler are both driven
92
- // by the identical (taskDurationMax - taskDurationP50) delta on the same stage: the same
93
- // dominant outlier task reported by two detectors, each independently clipped (see "Overlap
94
- // caveat: skew / straggler" in impact-estimation.md). skew's P95/median branch samples a
95
- // different task and stays independent. Flags both sides via validationRequired (rather than
96
- // suppressing either) so neither finding's own diagnostic value is lost; the flag rides the same
91
+ // skew (either branch) and straggler both claim the stage's replayed tail recovery
92
+ // (tailReplayRecoveryMs via tailRecoveryMs): the same slow-task tail reported by two detectors
93
+ // (see "Overlap caveat: skew / straggler" in impact-estimation.md). skew's branch only changes
94
+ // the fallback single-task delta on a stage without the replay, so every skew + straggler pair
95
+ // on a stage is flagged. Flags both sides via validationRequired (rather than suppressing
96
+ // either) so neither finding's own diagnostic value is lost; the flag rides the same
97
97
  // confidence-caveat UI a reader already sees before trusting either finding's magnitude.
98
98
  function overlapNote(otherType ) {
99
- return `This overlaps with the ${otherType} finding on this stage: both are driven by the same dominant outlier task, so don't add their recoverable-time figures together.`;
99
+ return `This overlaps with the ${otherType} finding on this stage: both measure the same slow-task tail, so don't add their recoverable-time figures together.`;
100
100
  }
101
101
 
102
102
  function flagSkewStragglerOverlap(findings ) {
103
- const maxMedianSkewStages = new Set(
104
- findings.filter((f) => f.type === 'skew' && f.metric === 'max/median' && f.stageId != null).map((f) => f.stageId),
103
+ const skewStages = new Set(
104
+ findings.filter((f) => f.type === 'skew' && f.stageId != null).map((f) => f.stageId),
105
105
  );
106
- if (maxMedianSkewStages.size === 0) return;
106
+ if (skewStages.size === 0) return;
107
107
  const stragglerStages = new Set(
108
108
  findings.filter((f) => f.type === 'straggler' && f.stageId != null).map((f) => f.stageId),
109
109
  );
110
- const overlapStages = new Set([...maxMedianSkewStages].filter((id) => stragglerStages.has(id)));
110
+ const overlapStages = new Set([...skewStages].filter((id) => stragglerStages.has(id)));
111
111
  if (overlapStages.size === 0) return;
112
112
  for (const f of findings) {
113
113
  if (f.stageId == null || !overlapStages.has(f.stageId)) continue;
114
114
  const note = f.type === 'skew' ? overlapNote('straggler') : f.type === 'straggler' ? overlapNote('skew') : null;
115
115
  if (!note) continue;
116
- f.validationRequired = f.validationRequired ? `${f.validationRequired} ${note}` : note;
116
+ f.validationRequired = [f.validationRequired, note].filter(Boolean).join(' ');
117
117
  }
118
118
  }
119
119
 
@@ -201,7 +201,7 @@ export function analyze(
201
201
  // One occupancy sweep per analysis, shared by the detectors' runtime floors and every entry's
202
202
  // estimate(), so a floor gates on the same occupancy-clipped figure displayed as savings.
203
203
  const impact = {
204
- stages, totalCores,
204
+ stages, totalCores, sql,
205
205
  occupancy: computeOccupancy(stages , totalCores),
206
206
  };
207
207
  // The one cast from the posted-model types to the detector-side shapes: types.ts's Stage and
@@ -4,6 +4,7 @@ import { effectiveThresholds } from '../threshold-overrides.js';
4
4
  import { IMPACT_BAND_ORDER } from '../format-utils.js';
5
5
 
6
6
 
7
+
7
8
 
8
9
  const IMPACT_BANDS = Object.keys(IMPACT_BAND_ORDER) ;
9
10
 
@@ -11,12 +12,16 @@ const IMPACT_BANDS = Object.keys(IMPACT_BAND_ORDER) ;
11
12
 
12
13
 
13
14
 
15
+
16
+
14
17
 
15
18
 
16
19
 
17
20
 
18
21
 
19
22
 
23
+
24
+
20
25
 
21
26
 
22
27
  // The skew entry's minTasksForP95 under the run's overrides, so the budget measures the same
@@ -107,7 +112,8 @@ function checkEfficiency(appModel , minPct ) {
107
112
  }
108
113
  const model = computeEfficiencyModel({
109
114
  app: appModel.app, stages: appModel.stages,
110
- executorsAdded: appModel.executors.added, runAggregates: appModel.runAggregates,
115
+ executorsAdded: appModel.executors.added, executorsRemoved: appModel.executors.removed,
116
+ runAggregates: appModel.runAggregates,
111
117
  });
112
118
  if (model.wastagePct == null) {
113
119
  return { name: 'min-efficiency', status: 'inconclusive', detail: 'Busy core time could not be computed (no available compute hours).' };
@@ -121,23 +127,23 @@ function checkEfficiency(appModel , minPct ) {
121
127
  function checkRegression(comparison , maxRegressionPct , regressionMetric ) {
122
128
  const row = comparison.metrics.find((m) => m.key === regressionMetric);
123
129
  if (!row || row.direction === 'unavailable' || row.baseline == null || row.delta == null) {
124
- return { name: 'max-regression', status: 'inconclusive', detail: `Metric "${regressionMetric}" is unavailable for this comparison.` };
130
+ return { name: 'max-regression', metric: regressionMetric, status: 'inconclusive', detail: `Metric "${regressionMetric}" is unavailable for this comparison.` };
125
131
  }
126
132
  // A neutral-direction metric (inputBytes/outputBytes/taskCount/executorsAdded) measures
127
133
  // workload volume, not performance: an increase isn't a regression, so no direction to check.
128
134
  if (row.direction === 'neutral') {
129
- return { name: 'max-regression', status: 'inconclusive', detail: `Metric "${regressionMetric}" measures workload volume, not performance: it has no regression direction to check.` };
135
+ return { name: 'max-regression', metric: regressionMetric, status: 'inconclusive', detail: `Metric "${regressionMetric}" measures workload volume, not performance: it has no regression direction to check.` };
130
136
  }
131
137
  if (row.direction !== 'regression') {
132
- return { name: 'max-regression', status: 'pass', detail: `Metric "${regressionMetric}" did not regress (${row.direction}).` };
138
+ return { name: 'max-regression', metric: regressionMetric, status: 'pass', detail: `Metric "${regressionMetric}" did not regress (${row.direction}).` };
133
139
  }
134
140
  const pct = row.baseline === 0 ? Infinity : Math.abs(row.delta / row.baseline) * 100;
135
141
  const pctLabel = row.baseline === 0
136
142
  ? `regressed from 0 to ${row.delta} (was absent/zero in baseline)`
137
143
  : `regressed ${pct.toFixed(1)}%`;
138
144
  return pct > maxRegressionPct
139
- ? { name: 'max-regression', status: 'violation', detail: `Metric "${regressionMetric}" ${pctLabel}, exceeding budget ${maxRegressionPct}%.` }
140
- : { name: 'max-regression', status: 'pass', detail: `Metric "${regressionMetric}" ${pctLabel}, within budget ${maxRegressionPct}%.` };
145
+ ? { name: 'max-regression', metric: regressionMetric, status: 'violation', detail: `Metric "${regressionMetric}" ${pctLabel}, exceeding budget ${maxRegressionPct}%.` }
146
+ : { name: 'max-regression', metric: regressionMetric, status: 'pass', detail: `Metric "${regressionMetric}" ${pctLabel}, within budget ${maxRegressionPct}%.` };
141
147
  }
142
148
 
143
149
  function checkFailOnIntroduced(comparison , band ) {
@@ -158,8 +164,12 @@ function pushComparisonBudget(
158
164
  comparison ,
159
165
  name ,
160
166
  check ,
167
+ metric ,
161
168
  ) {
162
- results.push(comparison ? check(comparison) : { name, status: 'inconclusive', detail: 'No baseline comparison available to evaluate this budget.' });
169
+ results.push(comparison ? check(comparison) : {
170
+ name, status: 'inconclusive', detail: 'No baseline comparison available to evaluate this budget.',
171
+ ...(metric !== undefined ? { metric } : {}),
172
+ });
163
173
  }
164
174
 
165
175
  /** `thresholds`: the overrides the catalog was analyzed with, so a budget that recomputes a
@@ -177,12 +187,16 @@ export function evaluateBudgets({ appModel, catalog, budgets, comparison, thresh
177
187
  // Guarded here so any evaluateBudgets caller benefits: regressionMetric without
178
188
  // maxRegressionPct would otherwise skip the `if` silently, reporting nothing.
179
189
  if (budgets.regressionMetric !== undefined && budgets.maxRegressionPct === undefined) {
180
- results.push({ name: 'max-regression', status: 'inconclusive', detail: `regressionMetric "${budgets.regressionMetric}" was set without maxRegressionPct; the regression budget was not evaluated.` });
190
+ results.push({ name: 'max-regression', metric: budgets.regressionMetric, status: 'inconclusive', detail: `regressionMetric "${budgets.regressionMetric}" was set without maxRegressionPct; the regression budget was not evaluated.` });
181
191
  } else if (budgets.maxRegressionPct !== undefined) {
182
192
  // `!== undefined`, not Number.isFinite: a zero-baseline regression's pct is Infinity,
183
193
  // so an "unlimited" budget is a legitimate input (the CLI already rejects non-finite flags).
184
194
  pushComparisonBudget(results, comparison, 'max-regression',
185
- (c) => checkRegression(c, budgets.maxRegressionPct , budgets.regressionMetric ?? 'wallClock'));
195
+ (c) => checkRegression(c, budgets.maxRegressionPct , budgets.regressionMetric ?? 'wallClock'),
196
+ budgets.regressionMetric ?? 'wallClock');
197
+ }
198
+ for (const { metric, maxPct } of budgets.regressionBudgets ?? []) {
199
+ pushComparisonBudget(results, comparison, 'max-regression', (c) => checkRegression(c, maxPct, metric), metric);
186
200
  }
187
201
  if (budgets.failOnIntroduced !== undefined) {
188
202
  pushComparisonBudget(results, comparison, 'fail-on-introduced',
@@ -5,6 +5,7 @@ import { nodeParseCodecs } from './native-zstd.js';
5
5
  import { createModelCallbacks } from '../model-assembler.js';
6
6
  import { routeMessage, } from '../ingest.js';
7
7
 
8
+ import { mcpError } from '../mcp-error.js';
8
9
 
9
10
  // No whole-file arrayBuffer(): the parser only ever reads bounded slices, and a
10
11
  // whole-file read is what capped local logs at 2 GiB.
@@ -129,7 +130,10 @@ export function collectViaDispatch(
129
130
  return new Promise((resolve, reject) => {
130
131
  const handlers = {
131
132
  ...cb,
132
- onDone: (msg ) => resolve({ appModel, skippedLines: (msg )?.skippedLines ?? 0 }),
133
+ onDone: (msg ) => {
134
+ cb.onDone(msg); // records the parse gaps on the model, as the dashboard's ingest does
135
+ resolve({ appModel, skippedLines: (msg )?.skippedLines ?? 0 });
136
+ },
133
137
  onError: (msg ) => reject(onDecodeError(msg)),
134
138
  };
135
139
  const emit = (msg ) => dispatch(msg, handlers);
@@ -149,10 +153,13 @@ export async function collectRun(inputPath )
149
153
  return file;
150
154
  };
151
155
  try {
156
+ // A file or folder that isn't a decodable event log rejects with invalid-event-log, the code
157
+ // shs-load.ts gives an archive that fails to decode, so MCP reports both the same way. The
158
+ // CLI reads only the message.
152
159
  return await collectViaDispatch((state, emit, reject) => {
153
160
  if (stat.isDirectory()) {
154
161
  if (!isRollingLogDirectory(inputPath)) {
155
- reject(new Error("This isn't a Spark rolling event-log directory. Pass a single event-log file instead."));
162
+ reject(mcpError('invalid-event-log', "This isn't a Spark rolling event-log directory. Pass a single event-log file instead."));
156
163
  return;
157
164
  }
158
165
  const names = readdirSync(inputPath);
@@ -160,7 +167,7 @@ export async function collectRun(inputPath )
160
167
  try {
161
168
  ordered = reassembleRollingEntries(names);
162
169
  } catch (e) {
163
- reject(e);
170
+ reject(mcpError('invalid-event-log', (e ).message));
164
171
  return;
165
172
  }
166
173
  const files = ordered.map((name) => open(join(inputPath, name)));
@@ -170,7 +177,7 @@ export async function collectRun(inputPath )
170
177
  } else {
171
178
  runParse(open(inputPath), state, { emit, ...nodeParseCodecs }).catch(reject);
172
179
  }
173
- }, (msg) => new Error((msg ).message));
180
+ }, (msg) => mcpError('invalid-event-log', (msg ).message));
174
181
  } finally {
175
182
  for (const file of opened) file.close();
176
183
  }
@@ -0,0 +1,83 @@
1
+ // Parses the regression budgets a CLI caller lists: repeated `--regression-budget <metric>:<pct>`
2
+ // flags and a `--budgets <file.json>` file. Every failure throws an Error naming the problem, so
3
+ // the caller refuses to run (exit 2) instead of quietly dropping a budget the user meant to gate on.
4
+ import { readFileSync } from 'node:fs';
5
+ import { resolve } from 'node:path';
6
+ import { COMPARISON_METRIC_KEYS } from '../run-comparison.js';
7
+
8
+
9
+
10
+
11
+ // Plain non-negative decimals only: Number() would also accept '', ' ', '0x10' and '1e3'.
12
+ const PCT_PATTERN = /^\d+(\.\d+)?$/;
13
+
14
+ function assertKnownMetric(metric , origin ) {
15
+ if (!COMPARISON_METRIC_KEYS.includes(metric)) {
16
+ throw new Error(`${origin}: unknown metric "${metric}" (expected one of: ${COMPARISON_METRIC_KEYS.join(', ')}).`);
17
+ }
18
+ }
19
+
20
+ /** One `--regression-budget` value, `<metric>:<pct>`. */
21
+ export function parseRegressionBudgetFlag(spec ) {
22
+ const origin = `--regression-budget "${spec}"`;
23
+ const colon = spec.indexOf(':');
24
+ if (colon === -1) throw new Error(`${origin}: expected <metric>:<pct>.`);
25
+ const metric = spec.slice(0, colon);
26
+ const pct = spec.slice(colon + 1);
27
+ assertKnownMetric(metric, origin);
28
+ if (!PCT_PATTERN.test(pct)) throw new Error(`${origin}: "${pct}" is not a non-negative percentage.`);
29
+ return { metric, maxPct: Number(pct) };
30
+ }
31
+
32
+ /** The parsed `--budgets` JSON: `{"regression": {"<metric>": <pct>}}`. */
33
+ export function parseBudgetsFile(raw , origin ) {
34
+ if (raw === null || typeof raw !== 'object' || Array.isArray(raw)) {
35
+ throw new Error(`${origin}: expected a JSON object like {"regression": {"wallClock": 10}}.`);
36
+ }
37
+ const unknownKeys = Object.keys(raw).filter((k) => k !== 'regression');
38
+ if (unknownKeys.length > 0) {
39
+ throw new Error(`${origin}: unknown key ${unknownKeys.map((k) => `"${k}"`).join(', ')} (the only key is "regression").`);
40
+ }
41
+ const regression = (raw ).regression;
42
+ if (regression === null || typeof regression !== 'object' || Array.isArray(regression)) {
43
+ throw new Error(`${origin}: "regression" must be an object mapping a metric key to a percentage.`);
44
+ }
45
+ return Object.entries(regression).map(([metric, pct]) => {
46
+ assertKnownMetric(metric, origin);
47
+ if (typeof pct !== 'number' || !Number.isFinite(pct) || pct < 0) {
48
+ throw new Error(`${origin}: "${metric}" must be a non-negative number (a percentage), got ${JSON.stringify(pct)}.`);
49
+ }
50
+ return { metric, maxPct: pct };
51
+ });
52
+ }
53
+
54
+ export function loadBudgetsFile(path ) {
55
+ const fullPath = resolve(path);
56
+ let text ;
57
+ try {
58
+ text = readFileSync(fullPath, 'utf8');
59
+ } catch (e) {
60
+ throw new Error(`Cannot read budgets file ${fullPath}: ${(e ).message}`, { cause: e });
61
+ }
62
+ let raw ;
63
+ try {
64
+ raw = JSON.parse(text);
65
+ } catch (e) {
66
+ throw new Error(`Budgets file ${fullPath} is not valid JSON: ${(e ).message}`, { cause: e });
67
+ }
68
+ return parseBudgetsFile(raw, `Budgets file ${fullPath}`);
69
+ }
70
+
71
+ /** Checks that no metric is budgeted twice across the sources (the legacy
72
+ * `--max-regression-pct`/`--regression-metric` pair counts as one), and returns the budgets. */
73
+ export function combineRegressionBudgets(sources ) {
74
+ const seen = new Map ();
75
+ for (const { origin, budget } of sources) {
76
+ const first = seen.get(budget.metric);
77
+ if (first !== undefined) {
78
+ throw new Error(`Metric "${budget.metric}" has two regression budgets (${first} and ${origin}); give each metric one.`);
79
+ }
80
+ seen.set(budget.metric, origin);
81
+ }
82
+ return sources.map((s) => s.budget);
83
+ }
@@ -37,7 +37,7 @@ import { NEUTRAL_METRIC_KEYS, } from './run-comparison.js
37
37
 
38
38
 
39
39
 
40
- /** A change under this share of run A's value reads as "about the same", for
40
+ /** A change under this share of the baseline's value reads as "about the same", for
41
41
  * run time and cost metrics alike: run-to-run noise on a shared cluster easily
42
42
  * moves a job a percent or two. */
43
43
  export const SAME_CHANGE_SHARE = 0.02;
@@ -78,14 +78,14 @@ function namesWhere(net , keep )
78
78
  return [...net].filter(([, change]) => keep(change)).map(([name]) => name);
79
79
  }
80
80
 
81
- /** "Run B had 2 of 5 jobs fail", or, when every job failed, "Run B's only
82
- * job failed" / "All 3 of run B's jobs failed" rather than "1 of 1 jobs". */
83
- function jobsFailed(run , { failedJobs, totalJobs } ) {
84
- if (failedJobs < totalJobs) return `Run ${run} had ${failedJobs} of ${totalJobs} jobs fail`;
85
- return totalJobs === 1 ? `Run ${run}'s only job failed` : `All ${totalJobs} of run ${run}'s jobs failed`;
81
+ /** "The candidate had 2 of 5 jobs fail", or, when every job failed, "The candidate's only
82
+ * job failed" / "All 3 of the candidate's jobs failed" rather than "1 of 1 jobs". */
83
+ function jobsFailed(run , { failedJobs, totalJobs } ) {
84
+ if (failedJobs < totalJobs) return `The ${run} had ${failedJobs} of ${totalJobs} jobs fail`;
85
+ return totalJobs === 1 ? `The ${run}'s only job failed` : `All ${totalJobs} of the ${run}'s jobs failed`;
86
86
  }
87
87
 
88
- /** Run A's failures in the parenthesis after run B's. */
88
+ /** The baseline's failures in the parenthesis after the candidate's. */
89
89
  function baselineFailures({ failedJobs, totalJobs } ) {
90
90
  if (failedJobs === 0) return 'none';
91
91
  if (failedJobs < totalJobs) return `${failedJobs} of ${totalJobs}`;
@@ -93,18 +93,18 @@ function baselineFailures({ failedJobs, totalJobs } )
93
93
  }
94
94
 
95
95
  /** The headline when either run had failed jobs, as the run verdict leads
96
- * with a failure: a faster run B that dropped work is not an improvement,
96
+ * with a failure: a faster candidate that dropped work is not an improvement,
97
97
  * and with equal failure counts the tone stays neutral since a failing run
98
98
  * that ends sooner may just have failed earlier. Null when both runs
99
99
  * completed. */
100
100
  function failureHeadline(base , cand ) {
101
101
  if (base.failedJobs === 0 && cand.failedJobs === 0) return null;
102
102
  const tone = cand.failedJobs > base.failedJobs ? 'worse' : cand.failedJobs < base.failedJobs ? 'better' : 'same';
103
- if (cand.failedJobs === 0) return { title: `${jobsFailed('A', base)}; run B completed`, tone };
104
- return { title: `${jobsFailed('B', cand)} (run A: ${baselineFailures(base)})`, tone };
103
+ if (cand.failedJobs === 0) return { title: `${jobsFailed('baseline', base)}; the candidate completed`, tone };
104
+ return { title: `${jobsFailed('candidate', cand)} (baseline: ${baselineFailures(base)})`, tone };
105
105
  }
106
106
 
107
- /** One plain answer to "did run B get better or worse than run A", from the
107
+ /** One plain answer to "did the candidate get better or worse than the baseline", from the
108
108
  * comparison's own whole-run metrics and finding-category tallies: run time
109
109
  * first, then which cost metrics moved each way past run-to-run noise, then which finding
110
110
  * categories appeared or went away. When either run had failed jobs, that
@@ -119,25 +119,25 @@ export function summarizeComparison(
119
119
  jobs ,
120
120
  ) {
121
121
  const wall = metrics.find((metric) => metric.key === 'wallClock');
122
- let title = 'Run time could not be compared between run A and run B';
122
+ let title = 'Run time could not be compared between the baseline and the candidate';
123
123
  let tone = 'unknown';
124
124
  // A log with no end-of-run record stops where the run was cut off, so its
125
125
  // shorter time is not a speed-up: say how much each log covers, neutrally.
126
- const incompleteRuns = jobs ? (['A', 'B'] ).filter((run) => (run === 'A' ? jobs.baseline : jobs.candidate).incomplete) : [];
126
+ const incompleteRuns = jobs ? (['baseline', 'candidate'] ).filter((run) => jobs[run].incomplete) : [];
127
127
  if (wall && wall.baseline != null && wall.baseline > 0 && wall.candidate != null && incompleteRuns.length > 0) {
128
128
  const change = wall.candidate - wall.baseline;
129
129
  title = Math.abs(change / wall.baseline) < SAME_CHANGE_SHARE
130
- ? "Run B's log covers about as much run time as run A's"
131
- : `Run B's log covers ${formatDuration(Math.abs(change))} ${change < 0 ? 'less' : 'more'} run time than run A's`;
130
+ ? "The candidate's log covers about as much run time as the baseline's"
131
+ : `The candidate's log covers ${formatDuration(Math.abs(change))} ${change < 0 ? 'less' : 'more'} run time than the baseline's`;
132
132
  } else if (wall && wall.baseline != null && wall.baseline > 0 && wall.candidate != null) {
133
133
  const change = wall.candidate - wall.baseline;
134
134
  const share = change / wall.baseline;
135
135
  if (Math.abs(share) < SAME_CHANGE_SHARE) {
136
- title = 'Run B took about as long as run A';
136
+ title = 'The candidate took about as long as the baseline';
137
137
  tone = 'same';
138
138
  } else {
139
139
  const faster = change < 0;
140
- title = `Run B finished ${formatDuration(Math.abs(change))} ${faster ? 'faster' : 'slower'} than run A (${Math.round(Math.abs(share) * 100)}%)`;
140
+ title = `The candidate finished ${formatDuration(Math.abs(change))} ${faster ? 'faster' : 'slower'} than the baseline (${Math.round(Math.abs(share) * 100)}%)`;
141
141
  tone = faster ? 'better' : 'worse';
142
142
  }
143
143
  }
@@ -156,17 +156,17 @@ export function summarizeComparison(
156
156
  if (incompleteRuns.length === 2) {
157
157
  sentences.push('Neither log has an end-of-run record, so their times cover only what each log captured, not how long the runs took.');
158
158
  } else if (incompleteRuns.length === 1) {
159
- sentences.push(`Run ${incompleteRuns[0]}'s log has no end-of-run record, so its time covers only what the log captured, not how long the run took.`);
159
+ sentences.push(`The ${incompleteRuns[0]}'s log has no end-of-run record, so its time covers only what the log captured, not how long the run took.`);
160
160
  }
161
- if (worse.length > 0) sentences.push(`Worse in run B: ${worse.join(', ')}.`);
162
- if (better.length > 0) sentences.push(`Better in run B: ${better.join(', ')}.`);
161
+ if (worse.length > 0) sentences.push(`Worse in the candidate: ${worse.join(', ')}.`);
162
+ if (better.length > 0) sentences.push(`Better in the candidate: ${better.join(', ')}.`);
163
163
  if (cost.length > 0 && worse.length === 0 && better.length === 0) sentences.push('Other measured cost metrics look about the same.');
164
164
 
165
165
  const net = netByCategory(findings);
166
166
  const introduced = namesWhere(net, (change) => change > 0);
167
167
  const resolved = namesWhere(net, (change) => change < 0);
168
- if (introduced.length > 0) sentences.push(`New or more frequent in run B: ${introduced.join(', ')}.`);
169
- if (resolved.length > 0) sentences.push(`Less frequent in run B: ${resolved.join(', ')}.`);
168
+ if (introduced.length > 0) sentences.push(`New or more frequent in the candidate: ${introduced.join(', ')}.`);
169
+ if (resolved.length > 0) sentences.push(`Less frequent in the candidate: ${resolved.join(', ')}.`);
170
170
  return { title, tone, sentences };
171
171
  }
172
172
 
@@ -1 +1 @@
1
- b5b0e0b6178129de4de482d4b59fbf881822655d8911b374c4a60a33b38f63fb
1
+ ab36c5fdcdc1a32a4e6206459e843e8d2267a90e6f1de29671827315d81bc374