sparkforensics-mcp 0.3.0 → 0.5.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (92) hide show
  1. package/bin/sparkforensics-mcp.mjs +41 -10
  2. package/package.json +1 -1
  3. package/vendor-core/allocation.js +106 -0
  4. package/vendor-core/analyzer.js +168 -60
  5. package/vendor-core/check-coverage.js +88 -0
  6. package/vendor-core/cli/budgets.js +54 -27
  7. package/vendor-core/cli/collect-run.js +84 -32
  8. package/vendor-core/cli/regression-budgets.js +83 -0
  9. package/vendor-core/cli/threshold-config.js +28 -0
  10. package/vendor-core/comparison-verdict.js +177 -0
  11. package/vendor-core/core-source-hash.txt +1 -0
  12. package/vendor-core/core-usage-locality.js +56 -2
  13. package/vendor-core/detector-docs.js +58 -0
  14. package/vendor-core/detectors.js +1094 -500
  15. package/vendor-core/docs-config.js +0 -36
  16. package/vendor-core/docs-content/chapters/nav-index.json +31 -0
  17. package/vendor-core/docs-content/detection/cache.md +3 -2
  18. package/vendor-core/docs-content/detection/cfg.md +9 -8
  19. package/vendor-core/docs-content/detection/chrn.md +1 -2
  20. package/vendor-core/docs-content/detection/cold.md +4 -2
  21. package/vendor-core/docs-content/detection/cstor.md +9 -0
  22. package/vendor-core/docs-content/detection/fail.md +3 -2
  23. package/vendor-core/docs-content/detection/gc.md +3 -2
  24. package/vendor-core/docs-content/detection/host.md +2 -1
  25. package/vendor-core/docs-content/detection/local.md +1 -1
  26. package/vendor-core/docs-content/detection/mem.md +5 -2
  27. package/vendor-core/docs-content/detection/plan.md +2 -1
  28. package/vendor-core/docs-content/detection/sfail.md +2 -1
  29. package/vendor-core/docs-content/detection/shape.md +5 -4
  30. package/vendor-core/docs-content/detection/skew.md +3 -1
  31. package/vendor-core/docs-content/detection/slow.md +2 -2
  32. package/vendor-core/docs-content/detection/spec.md +2 -3
  33. package/vendor-core/docs-content/detection/spill.md +1 -1
  34. package/vendor-core/docs-site-config.js +3 -0
  35. package/vendor-core/effective-conf.js +107 -0
  36. package/vendor-core/efficiency-model.js +8 -6
  37. package/vendor-core/event-handlers.js +321 -44
  38. package/vendor-core/event-schemas.js +23 -0
  39. package/vendor-core/evidence-report.js +432 -115
  40. package/vendor-core/export-data.js +79 -6
  41. package/vendor-core/finding-action-label.js +9 -88
  42. package/vendor-core/finding-filter-predicate.js +9 -0
  43. package/vendor-core/finding-generic-recommendation.js +26 -105
  44. package/vendor-core/finding-names.js +28 -45
  45. package/vendor-core/finding-presentation.js +368 -0
  46. package/vendor-core/finding-tag-help.js +110 -0
  47. package/vendor-core/finding-types.js +373 -0
  48. package/vendor-core/findings-of-type.js +11 -0
  49. package/vendor-core/format-utils.js +96 -30
  50. package/vendor-core/html-export.js +51 -0
  51. package/vendor-core/impact-band.js +21 -8
  52. package/vendor-core/impact-estimator.js +25 -520
  53. package/vendor-core/impact-format.js +115 -0
  54. package/vendor-core/impact-model.js +197 -0
  55. package/vendor-core/ingest.js +6 -2
  56. package/vendor-core/intervals.js +13 -0
  57. package/vendor-core/list-runs.js +7 -5
  58. package/vendor-core/load-vendored.js +70 -5
  59. package/vendor-core/mcp-server-factory.js +14 -10
  60. package/vendor-core/mcp-tools.js +105 -45
  61. package/vendor-core/model-assembler.js +35 -1
  62. package/vendor-core/occupancy.js +1 -1
  63. package/vendor-core/parser-worker.js +2 -2
  64. package/vendor-core/plan-graph-model.js +3 -2
  65. package/vendor-core/plan-node-detail.js +1 -1
  66. package/vendor-core/proxy.js +3 -1
  67. package/vendor-core/python-stage.js +25 -0
  68. package/vendor-core/recommendation-rollup.js +70 -3
  69. package/vendor-core/redact.js +96 -37
  70. package/vendor-core/remediation.js +20 -0
  71. package/vendor-core/run-comparison.js +73 -29
  72. package/vendor-core/run-interpretation.js +291 -0
  73. package/vendor-core/run-metrics.js +198 -0
  74. package/vendor-core/run-outcome.js +74 -0
  75. package/vendor-core/run-payload.js +17 -0
  76. package/vendor-core/run-shape.js +40 -0
  77. package/vendor-core/run-totals.js +24 -0
  78. package/vendor-core/run-verdict.js +352 -0
  79. package/vendor-core/scaling-sim.js +4 -5
  80. package/vendor-core/scorecard-estimates.js +63 -0
  81. package/vendor-core/session-snapshot.js +7 -0
  82. package/vendor-core/shs-schemas.js +2 -2
  83. package/vendor-core/spark-memory.js +17 -0
  84. package/vendor-core/sql-stages.js +11 -0
  85. package/vendor-core/stage-plan-nodes.js +18 -0
  86. package/vendor-core/stage-quantiles.js +6 -0
  87. package/vendor-core/threshold-overrides.js +160 -0
  88. package/vendor-core/threshold-summary.js +11 -33
  89. package/vendor-core/types.js +54 -42
  90. package/vendor-core/wall-clock.js +1 -12
  91. package/vendor-core/wasted-core-hours.js +12 -9
  92. package/vendor-core/write-targets.js +312 -0
@@ -1,8 +1,10 @@
1
1
  import { computeEfficiencyModel } from '../efficiency-model.js';
2
- import { computeSkewRatio, DETECTORS } from '../detectors.js';
2
+ import { computeSkewRatio, ENTRY_BY_TYPE, } from '../detectors.js';
3
+ import { effectiveThresholds } from '../threshold-overrides.js';
3
4
  import { IMPACT_BAND_ORDER } from '../format-utils.js';
4
5
 
5
6
 
7
+
6
8
 
7
9
  const IMPACT_BANDS = Object.keys(IMPACT_BAND_ORDER) ;
8
10
 
@@ -10,29 +12,33 @@ const IMPACT_BANDS = Object.keys(IMPACT_BAND_ORDER) ;
10
12
 
11
13
 
12
14
 
15
+
16
+
13
17
 
14
18
 
15
19
 
16
-
20
+
17
21
 
18
22
 
23
+
24
+
19
25
 
20
26
 
21
- const skewDetector = DETECTORS.find((d) => d.type === 'skew');
22
- const SKEW_MIN_TASKS_FOR_P95 = skewDetector .thresholds .minTasksForP95 ;
27
+ // The skew entry's minTasksForP95 under the run's overrides, so the budget measures the same
28
+ // ratio (P95/median or max/median) the skew finding reports.
29
+ function skewMinTasksForP95(thresholds ) {
30
+ return effectiveThresholds(ENTRY_BY_TYPE.get('skew') , thresholds).minTasksForP95 ;
31
+ }
23
32
 
24
33
  function taskDataTrusted(appModel ) {
25
34
  const entry = appModel.evidenceAvailability?.entries?.find((e) => e.key === 'taskCoreTime');
26
35
  return entry?.state === 'present';
27
36
  }
28
37
 
29
- // Finding.value is number|string (some detectors put text there); spill findings are
30
- // always numeric, so the typeof guard narrows without changing behavior for real input.
31
38
  function maxFindingValue(catalog , type ) {
32
39
  const values = catalog
33
40
  .filter((f) => f.type === type)
34
- .map((f) => f.value ?? 0)
35
- .filter((v) => typeof v === 'number');
41
+ .map((f) => f.value ?? 0);
36
42
  return values.length > 0 ? Math.max(...values) : null;
37
43
  }
38
44
 
@@ -58,7 +64,7 @@ function checkSpill(appModel , catalog , maxSpillGb )
58
64
  : { name: 'max-spill', status: 'pass', detail: `Peak stage spill ${maxBytes} bytes within budget ${budgetBytes} bytes.` };
59
65
  }
60
66
 
61
- function checkSkew(appModel , maxSkewRatio ) {
67
+ function checkSkew(appModel , maxSkewRatio , minTasksForP95 ) {
62
68
  if (!taskDataTrusted(appModel)) {
63
69
  return { name: 'max-skew', status: 'inconclusive', detail: 'No trustworthy task-level evidence to measure skew.' };
64
70
  }
@@ -69,7 +75,7 @@ function checkSkew(appModel , maxSkewRatio ) {
69
75
  // Recompute the true ratio per stage: the skew detector floors findings at
70
76
  // thresholds.ratioWarn (3), so a stricter budget can't be enforced from catalog alone.
71
77
  const ratios = stages
72
- .map((stage) => computeSkewRatio(stage, SKEW_MIN_TASKS_FOR_P95))
78
+ .map((stage) => computeSkewRatio(stage, minTasksForP95))
73
79
  .filter((r) => r !== null)
74
80
  .map((r) => r.ratio);
75
81
  if (ratios.length === 0) {
@@ -97,43 +103,47 @@ function checkFailedTaskRate(appModel , catalog , maxPct
97
103
  return { name: 'max-failed-task-rate', status: 'pass', detail: `No job-failure-rate finding: task failure rate is below the detector's reporting floor.` };
98
104
  }
99
105
 
106
+ // Busy core time: the share of available executor core time that ran tasks, 100 minus the
107
+ // dashboard's "Unused core time". Not the dashboard's Efficiency tile (the share of wall-clock with
108
+ // a stage running), so the detail never calls it "Efficiency".
100
109
  function checkEfficiency(appModel , minPct ) {
101
110
  if (!taskDataTrusted(appModel)) {
102
- return { name: 'min-efficiency', status: 'inconclusive', detail: 'No trustworthy task-level evidence to measure efficiency.' };
111
+ return { name: 'min-efficiency', status: 'inconclusive', detail: 'No trustworthy task-level evidence to measure busy core time.' };
103
112
  }
104
113
  const model = computeEfficiencyModel({
105
114
  app: appModel.app, stages: appModel.stages,
106
- executorsAdded: appModel.executors.added, runAggregates: appModel.runAggregates,
115
+ executorsAdded: appModel.executors.added, executorsRemoved: appModel.executors.removed,
116
+ runAggregates: appModel.runAggregates,
107
117
  });
108
118
  if (model.wastagePct == null) {
109
- return { name: 'min-efficiency', status: 'inconclusive', detail: 'Efficiency could not be computed (no available compute hours).' };
119
+ return { name: 'min-efficiency', status: 'inconclusive', detail: 'Busy core time could not be computed (no available compute hours).' };
110
120
  }
111
- const efficiencyPct = 100 - model.wastagePct;
112
- return efficiencyPct < minPct
113
- ? { name: 'min-efficiency', status: 'violation', detail: `Efficiency ${efficiencyPct}% below budget ${minPct}%.` }
114
- : { name: 'min-efficiency', status: 'pass', detail: `Efficiency ${efficiencyPct}% meets budget ${minPct}%.` };
121
+ const busyCorePct = 100 - model.wastagePct;
122
+ return busyCorePct < minPct
123
+ ? { name: 'min-efficiency', status: 'violation', detail: `Busy core time ${busyCorePct}% below budget ${minPct}%.` }
124
+ : { name: 'min-efficiency', status: 'pass', detail: `Busy core time ${busyCorePct}% meets budget ${minPct}%.` };
115
125
  }
116
126
 
117
127
  function checkRegression(comparison , maxRegressionPct , regressionMetric ) {
118
128
  const row = comparison.metrics.find((m) => m.key === regressionMetric);
119
129
  if (!row || row.direction === 'unavailable' || row.baseline == null || row.delta == null) {
120
- return { name: 'max-regression', status: 'inconclusive', detail: `Metric "${regressionMetric}" is unavailable for this comparison.` };
130
+ return { name: 'max-regression', metric: regressionMetric, status: 'inconclusive', detail: `Metric "${regressionMetric}" is unavailable for this comparison.` };
121
131
  }
122
132
  // A neutral-direction metric (inputBytes/outputBytes/taskCount/executorsAdded) measures
123
133
  // workload volume, not performance: an increase isn't a regression, so no direction to check.
124
134
  if (row.direction === 'neutral') {
125
- return { name: 'max-regression', status: 'inconclusive', detail: `Metric "${regressionMetric}" measures workload volume, not performance: it has no regression direction to check.` };
135
+ return { name: 'max-regression', metric: regressionMetric, status: 'inconclusive', detail: `Metric "${regressionMetric}" measures workload volume, not performance: it has no regression direction to check.` };
126
136
  }
127
137
  if (row.direction !== 'regression') {
128
- return { name: 'max-regression', status: 'pass', detail: `Metric "${regressionMetric}" did not regress (${row.direction}).` };
138
+ return { name: 'max-regression', metric: regressionMetric, status: 'pass', detail: `Metric "${regressionMetric}" did not regress (${row.direction}).` };
129
139
  }
130
140
  const pct = row.baseline === 0 ? Infinity : Math.abs(row.delta / row.baseline) * 100;
131
141
  const pctLabel = row.baseline === 0
132
142
  ? `regressed from 0 to ${row.delta} (was absent/zero in baseline)`
133
143
  : `regressed ${pct.toFixed(1)}%`;
134
144
  return pct > maxRegressionPct
135
- ? { name: 'max-regression', status: 'violation', detail: `Metric "${regressionMetric}" ${pctLabel}, exceeding budget ${maxRegressionPct}%.` }
136
- : { name: 'max-regression', status: 'pass', detail: `Metric "${regressionMetric}" ${pctLabel}, within budget ${maxRegressionPct}%.` };
145
+ ? { name: 'max-regression', metric: regressionMetric, status: 'violation', detail: `Metric "${regressionMetric}" ${pctLabel}, exceeding budget ${maxRegressionPct}%.` }
146
+ : { name: 'max-regression', metric: regressionMetric, status: 'pass', detail: `Metric "${regressionMetric}" ${pctLabel}, within budget ${maxRegressionPct}%.` };
137
147
  }
138
148
 
139
149
  function checkFailOnIntroduced(comparison , band ) {
@@ -154,33 +164,50 @@ function pushComparisonBudget(
154
164
  comparison ,
155
165
  name ,
156
166
  check ,
167
+ metric ,
157
168
  ) {
158
- results.push(comparison ? check(comparison) : { name, status: 'inconclusive', detail: 'No baseline comparison available to evaluate this budget.' });
169
+ results.push(comparison ? check(comparison) : {
170
+ name, status: 'inconclusive', detail: 'No baseline comparison available to evaluate this budget.',
171
+ ...(metric !== undefined ? { metric } : {}),
172
+ });
159
173
  }
160
174
 
161
- export function evaluateBudgets({ appModel, catalog, budgets, comparison }
175
+ /** `thresholds`: the overrides the catalog was analyzed with, so a budget that recomputes a
176
+ * detector's figure (--max-skew) uses the same thresholds. */
177
+ export function evaluateBudgets({ appModel, catalog, budgets, comparison, thresholds }
162
178
 
179
+
163
180
  ) {
164
181
  const results = [];
165
182
  if (Number.isFinite(budgets.maxRuntimeMs)) results.push(checkRuntime(appModel, budgets.maxRuntimeMs ));
166
183
  if (Number.isFinite(budgets.maxSpillGb)) results.push(checkSpill(appModel, catalog, budgets.maxSpillGb ));
167
- if (Number.isFinite(budgets.maxSkewRatio)) results.push(checkSkew(appModel, budgets.maxSkewRatio ));
184
+ if (Number.isFinite(budgets.maxSkewRatio)) results.push(checkSkew(appModel, budgets.maxSkewRatio , skewMinTasksForP95(thresholds)));
168
185
  if (Number.isFinite(budgets.maxFailedTaskRatePct)) results.push(checkFailedTaskRate(appModel, catalog, budgets.maxFailedTaskRatePct ));
169
186
  if (Number.isFinite(budgets.minEfficiencyPct)) results.push(checkEfficiency(appModel, budgets.minEfficiencyPct ));
170
187
  // Guarded here so any evaluateBudgets caller benefits: regressionMetric without
171
188
  // maxRegressionPct would otherwise skip the `if` silently, reporting nothing.
172
189
  if (budgets.regressionMetric !== undefined && budgets.maxRegressionPct === undefined) {
173
- results.push({ name: 'max-regression', status: 'inconclusive', detail: `regressionMetric "${budgets.regressionMetric}" was set without maxRegressionPct; the regression budget was not evaluated.` });
190
+ results.push({ name: 'max-regression', metric: budgets.regressionMetric, status: 'inconclusive', detail: `regressionMetric "${budgets.regressionMetric}" was set without maxRegressionPct; the regression budget was not evaluated.` });
174
191
  } else if (budgets.maxRegressionPct !== undefined) {
175
192
  // `!== undefined`, not Number.isFinite: a zero-baseline regression's pct is Infinity,
176
193
  // so an "unlimited" budget is a legitimate input (the CLI already rejects non-finite flags).
177
194
  pushComparisonBudget(results, comparison, 'max-regression',
178
- (c) => checkRegression(c, budgets.maxRegressionPct , budgets.regressionMetric ?? 'wallClock'));
195
+ (c) => checkRegression(c, budgets.maxRegressionPct , budgets.regressionMetric ?? 'wallClock'),
196
+ budgets.regressionMetric ?? 'wallClock');
197
+ }
198
+ for (const { metric, maxPct } of budgets.regressionBudgets ?? []) {
199
+ pushComparisonBudget(results, comparison, 'max-regression', (c) => checkRegression(c, maxPct, metric), metric);
179
200
  }
180
201
  if (budgets.failOnIntroduced !== undefined) {
181
202
  pushComparisonBudget(results, comparison, 'fail-on-introduced',
182
203
  (c) => checkFailOnIntroduced(c, budgets.failOnIntroduced ));
183
204
  }
205
+ // Always checked, unlike the opt-in budgets above: a run with no ApplicationEnd is
206
+ // inconclusive by default, so a passing budget can't hide a truncated log.
207
+ const incompleteRunFinding = catalog.find((f) => f.type === 'incompleteRun');
208
+ if (incompleteRunFinding) {
209
+ results.push({ name: 'run-complete', status: 'inconclusive', detail: incompleteRunFinding.recommendation ?? 'Event log has no ApplicationEnd event.' });
210
+ }
184
211
  return {
185
212
  results,
186
213
  violated: results.some((r) => r.status === 'violation'),
@@ -1,15 +1,19 @@
1
- import { readFileSync, readdirSync, statSync, openSync, readSync, closeSync } from 'node:fs';
1
+ import { readdirSync, statSync, openSync, readSync, closeSync } from 'node:fs';
2
2
  import { join, basename } from 'node:path';
3
3
  import { createState, runParse, runParseFiles, reassembleRollingEntries } from '../parser-worker.js';
4
4
  import { nodeParseCodecs } from './native-zstd.js';
5
5
  import { createModelCallbacks } from '../model-assembler.js';
6
6
  import { routeMessage, } from '../ingest.js';
7
7
 
8
+ import { mcpError } from '../mcp-error.js';
8
9
 
10
+ // No whole-file arrayBuffer(): the parser only ever reads bounded slices, and a
11
+ // whole-file read is what capped local logs at 2 GiB.
9
12
 
10
13
 
11
14
 
12
-
15
+
16
+
13
17
 
14
18
 
15
19
  export function emptyAppModel() {
@@ -24,18 +28,48 @@ export function emptyAppModel() {
24
28
  };
25
29
  }
26
30
 
27
- // File-like shape runParse/runParseFiles need: name, size, slice().arrayBuffer(), arrayBuffer().
31
+ // File-like shape runParse/runParseFiles need (name, size, slice().arrayBuffer()), read with
32
+ // positioned readSync so only the requested slice is ever in memory, whatever the file size.
33
+ // The descriptor opens on the first read and closes once a read reaches the end of the file,
34
+ // so a rolling directory's parts hold at most one open descriptor at a time while streaming.
35
+ // A later read (a zip archive reads its tail first) reopens it; callers must still call
36
+ // close() once parsing settles, to cover reads that stopped early on an error.
28
37
  export function nodeFileFromPath(path ) {
29
- const bytes = readFileSync(path);
30
- const u8 = new Uint8Array(bytes.buffer, bytes.byteOffset, bytes.byteLength);
38
+ const name = basename(path);
39
+ const size = statSync(path).size;
40
+ let fd = null;
41
+ const close = () => {
42
+ if (fd === null) return;
43
+ const open = fd;
44
+ fd = null;
45
+ closeSync(open);
46
+ };
31
47
  return {
32
- name: basename(path),
33
- size: u8.length,
48
+ name,
49
+ size,
34
50
  slice(start , end ) {
35
- const view = u8.subarray(start, end);
36
- return { async arrayBuffer() { return view.slice().buffer; } };
51
+ return {
52
+ async arrayBuffer() {
53
+ const from = Math.max(0, start);
54
+ const to = Math.min(end, size);
55
+ if (to <= from) return new ArrayBuffer(0);
56
+ const buf = new Uint8Array(to - from);
57
+ fd ??= openSync(path, 'r');
58
+ // readSync may return fewer bytes than asked; loop until the slice is full.
59
+ for (let filled = 0; filled < buf.length;) {
60
+ const n = readSync(fd, buf, filled, buf.length - filled, from + filled);
61
+ if (n === 0) {
62
+ close();
63
+ throw new Error(`"${name}" shrank while it was being read`);
64
+ }
65
+ filled += n;
66
+ }
67
+ if (to === size) close();
68
+ return buf.buffer;
69
+ },
70
+ };
37
71
  },
38
- async arrayBuffer() { return u8.slice().buffer; },
72
+ close,
39
73
  };
40
74
  }
41
75
 
@@ -96,7 +130,10 @@ export function collectViaDispatch(
96
130
  return new Promise((resolve, reject) => {
97
131
  const handlers = {
98
132
  ...cb,
99
- onDone: (msg ) => resolve({ appModel, skippedLines: (msg )?.skippedLines ?? 0 }),
133
+ onDone: (msg ) => {
134
+ cb.onDone(msg); // records the parse gaps on the model, as the dashboard's ingest does
135
+ resolve({ appModel, skippedLines: (msg )?.skippedLines ?? 0 });
136
+ },
100
137
  onError: (msg ) => reject(onDecodeError(msg)),
101
138
  };
102
139
  const emit = (msg ) => dispatch(msg, handlers);
@@ -107,26 +144,41 @@ export function collectViaDispatch(
107
144
 
108
145
  export async function collectRun(inputPath ) {
109
146
  const stat = statSync(inputPath);
110
- return collectViaDispatch((state, emit, reject) => {
111
- if (stat.isDirectory()) {
112
- if (!isRollingLogDirectory(inputPath)) {
113
- reject(new Error("This isn't a Spark rolling event-log directory. Pass a single event-log file instead."));
114
- return;
115
- }
116
- const names = readdirSync(inputPath);
117
- let ordered;
118
- try {
119
- ordered = reassembleRollingEntries(names);
120
- } catch (e) {
121
- reject(e);
122
- return;
147
+ // Every file opened for this run, closed once parsing settles on any path (done, parse
148
+ // error, decode error, or a throw past the parser's guards).
149
+ const opened = [];
150
+ const open = (path ) => {
151
+ const file = nodeFileFromPath(path);
152
+ opened.push(file);
153
+ return file;
154
+ };
155
+ try {
156
+ // A file or folder that isn't a decodable event log rejects with invalid-event-log, the code
157
+ // shs-load.ts gives an archive that fails to decode, so MCP reports both the same way. The
158
+ // CLI reads only the message.
159
+ return await collectViaDispatch((state, emit, reject) => {
160
+ if (stat.isDirectory()) {
161
+ if (!isRollingLogDirectory(inputPath)) {
162
+ reject(mcpError('invalid-event-log', "This isn't a Spark rolling event-log directory. Pass a single event-log file instead."));
163
+ return;
164
+ }
165
+ const names = readdirSync(inputPath);
166
+ let ordered;
167
+ try {
168
+ ordered = reassembleRollingEntries(names);
169
+ } catch (e) {
170
+ reject(mcpError('invalid-event-log', (e ).message));
171
+ return;
172
+ }
173
+ const files = ordered.map((name) => open(join(inputPath, name)));
174
+ // .catch(reject), not void: a throw past the parser's guards would otherwise leave
175
+ // this Promise pending forever, surfacing only as an unhandled rejection.
176
+ runParseFiles(files, state, { emit, ...nodeParseCodecs }).catch(reject);
177
+ } else {
178
+ runParse(open(inputPath), state, { emit, ...nodeParseCodecs }).catch(reject);
123
179
  }
124
- const files = ordered.map((name) => nodeFileFromPath(join(inputPath, name)));
125
- // .catch(reject), not void: a throw past the parser's guards would otherwise leave
126
- // this Promise pending forever, surfacing only as an unhandled rejection.
127
- runParseFiles(files, state, { emit, ...nodeParseCodecs }).catch(reject);
128
- } else {
129
- runParse(nodeFileFromPath(inputPath), state, { emit, ...nodeParseCodecs }).catch(reject);
130
- }
131
- }, (msg) => new Error((msg ).message));
180
+ }, (msg) => mcpError('invalid-event-log', (msg ).message));
181
+ } finally {
182
+ for (const file of opened) file.close();
183
+ }
132
184
  }
@@ -0,0 +1,83 @@
1
+ // Parses the regression budgets a CLI caller lists: repeated `--regression-budget <metric>:<pct>`
2
+ // flags and a `--budgets <file.json>` file. Every failure throws an Error naming the problem, so
3
+ // the caller refuses to run (exit 2) instead of quietly dropping a budget the user meant to gate on.
4
+ import { readFileSync } from 'node:fs';
5
+ import { resolve } from 'node:path';
6
+ import { COMPARISON_METRIC_KEYS } from '../run-comparison.js';
7
+
8
+
9
+
10
+
11
+ // Plain non-negative decimals only: Number() would also accept '', ' ', '0x10' and '1e3'.
12
+ const PCT_PATTERN = /^\d+(\.\d+)?$/;
13
+
14
+ function assertKnownMetric(metric , origin ) {
15
+ if (!COMPARISON_METRIC_KEYS.includes(metric)) {
16
+ throw new Error(`${origin}: unknown metric "${metric}" (expected one of: ${COMPARISON_METRIC_KEYS.join(', ')}).`);
17
+ }
18
+ }
19
+
20
+ /** One `--regression-budget` value, `<metric>:<pct>`. */
21
+ export function parseRegressionBudgetFlag(spec ) {
22
+ const origin = `--regression-budget "${spec}"`;
23
+ const colon = spec.indexOf(':');
24
+ if (colon === -1) throw new Error(`${origin}: expected <metric>:<pct>.`);
25
+ const metric = spec.slice(0, colon);
26
+ const pct = spec.slice(colon + 1);
27
+ assertKnownMetric(metric, origin);
28
+ if (!PCT_PATTERN.test(pct)) throw new Error(`${origin}: "${pct}" is not a non-negative percentage.`);
29
+ return { metric, maxPct: Number(pct) };
30
+ }
31
+
32
+ /** The parsed `--budgets` JSON: `{"regression": {"<metric>": <pct>}}`. */
33
+ export function parseBudgetsFile(raw , origin ) {
34
+ if (raw === null || typeof raw !== 'object' || Array.isArray(raw)) {
35
+ throw new Error(`${origin}: expected a JSON object like {"regression": {"wallClock": 10}}.`);
36
+ }
37
+ const unknownKeys = Object.keys(raw).filter((k) => k !== 'regression');
38
+ if (unknownKeys.length > 0) {
39
+ throw new Error(`${origin}: unknown key ${unknownKeys.map((k) => `"${k}"`).join(', ')} (the only key is "regression").`);
40
+ }
41
+ const regression = (raw ).regression;
42
+ if (regression === null || typeof regression !== 'object' || Array.isArray(regression)) {
43
+ throw new Error(`${origin}: "regression" must be an object mapping a metric key to a percentage.`);
44
+ }
45
+ return Object.entries(regression).map(([metric, pct]) => {
46
+ assertKnownMetric(metric, origin);
47
+ if (typeof pct !== 'number' || !Number.isFinite(pct) || pct < 0) {
48
+ throw new Error(`${origin}: "${metric}" must be a non-negative number (a percentage), got ${JSON.stringify(pct)}.`);
49
+ }
50
+ return { metric, maxPct: pct };
51
+ });
52
+ }
53
+
54
+ export function loadBudgetsFile(path ) {
55
+ const fullPath = resolve(path);
56
+ let text ;
57
+ try {
58
+ text = readFileSync(fullPath, 'utf8');
59
+ } catch (e) {
60
+ throw new Error(`Cannot read budgets file ${fullPath}: ${(e ).message}`, { cause: e });
61
+ }
62
+ let raw ;
63
+ try {
64
+ raw = JSON.parse(text);
65
+ } catch (e) {
66
+ throw new Error(`Budgets file ${fullPath} is not valid JSON: ${(e ).message}`, { cause: e });
67
+ }
68
+ return parseBudgetsFile(raw, `Budgets file ${fullPath}`);
69
+ }
70
+
71
+ /** Checks that no metric is budgeted twice across the sources (the legacy
72
+ * `--max-regression-pct`/`--regression-metric` pair counts as one), and returns the budgets. */
73
+ export function combineRegressionBudgets(sources ) {
74
+ const seen = new Map ();
75
+ for (const { origin, budget } of sources) {
76
+ const first = seen.get(budget.metric);
77
+ if (first !== undefined) {
78
+ throw new Error(`Metric "${budget.metric}" has two regression budgets (${first} and ${origin}); give each metric one.`);
79
+ }
80
+ seen.set(budget.metric, origin);
81
+ }
82
+ return sources.map((s) => s.budget);
83
+ }
@@ -0,0 +1,28 @@
1
+ // Reads a --thresholds config file for the CLI and the MCP server. Every failure throws an Error
2
+ // naming the file and the problem, so the caller refuses to run instead of quietly falling back
3
+ // to the defaults the user meant to change.
4
+ import { readFileSync } from 'node:fs';
5
+ import { resolve } from 'node:path';
6
+ import { parseThresholdOverrides } from '../threshold-overrides.js';
7
+
8
+
9
+ export function loadThresholdOverrides(path ) {
10
+ const fullPath = resolve(path);
11
+ let text ;
12
+ try {
13
+ text = readFileSync(fullPath, 'utf8');
14
+ } catch (e) {
15
+ throw new Error(`Cannot read thresholds file ${fullPath}: ${(e ).message}`, { cause: e });
16
+ }
17
+ let raw ;
18
+ try {
19
+ raw = JSON.parse(text);
20
+ } catch (e) {
21
+ throw new Error(`Thresholds file ${fullPath} is not valid JSON: ${(e ).message}`, { cause: e });
22
+ }
23
+ try {
24
+ return parseThresholdOverrides(raw);
25
+ } catch (e) {
26
+ throw new Error(`Thresholds file ${fullPath}: ${(e ).message}`, { cause: e });
27
+ }
28
+ }
@@ -0,0 +1,177 @@
1
+ import { formatDuration, typeTag } from './format-utils.js';
2
+ import { TAG_HELP } from './finding-tag-help.js';
3
+ import { NEUTRAL_METRIC_KEYS, } from './run-comparison.js';
4
+
5
+ /** The slice of a comparison metric row the verdict reads. */
6
+
7
+
8
+
9
+
10
+
11
+
12
+
13
+
14
+
15
+ /** The slice of a finding-category delta the verdict reads. */
16
+
17
+
18
+
19
+
20
+
21
+
22
+ /** How many of a run's ended jobs failed, counted as `summarizeRunOutcome`
23
+ * counts them. */
24
+
25
+
26
+
27
+
28
+
29
+
30
+
31
+
32
+
33
+
34
+
35
+
36
+
37
+
38
+
39
+
40
+ /** A change under this share of the baseline's value reads as "about the same", for
41
+ * run time and cost metrics alike: run-to-run noise on a shared cluster easily
42
+ * moves a job a percent or two. */
43
+ export const SAME_CHANGE_SHARE = 0.02;
44
+
45
+ function measured(metric ) {
46
+ return metric.baseline != null && metric.candidate != null;
47
+ }
48
+
49
+ function movedPastNoise(metric ) {
50
+ if (metric.baseline === 0) return metric.candidate !== 0;
51
+ return Math.abs(metric.candidate - metric.baseline) / Math.abs(metric.baseline) >= SAME_CHANGE_SHARE;
52
+ }
53
+
54
+ /** Plain name for a finding type ("Memory and disk spill"), falling back to
55
+ * its tag when the tag has no help entry. */
56
+ function categoryName(type ) {
57
+ const tag = typeTag(type);
58
+ return TAG_HELP[tag]?.expansion ?? tag;
59
+ }
60
+
61
+ /** Net count change per displayed category name. Categories are tallied per
62
+ * (rule, impact band), and several rules share one name (every Plan Advisor
63
+ * type reads "Plan advisor", every stage-shape sub-rule "Stage shape"), so a
64
+ * rule whose findings moved from critical to warning, or two rules under one
65
+ * name moving opposite ways, would otherwise
66
+ * read as both "introduced" and "resolved". Order follows first appearance,
67
+ * introduced before resolved. */
68
+ function netByCategory(findings ) {
69
+ const net = new Map ();
70
+ for (const item of [...findings.introduced, ...findings.resolved]) {
71
+ const name = categoryName(item.type);
72
+ net.set(name, (net.get(name) ?? 0) + item.candCount - item.baseCount);
73
+ }
74
+ return net;
75
+ }
76
+
77
+ function namesWhere(net , keep ) {
78
+ return [...net].filter(([, change]) => keep(change)).map(([name]) => name);
79
+ }
80
+
81
+ /** "The candidate had 2 of 5 jobs fail", or, when every job failed, "The candidate's only
82
+ * job failed" / "All 3 of the candidate's jobs failed" rather than "1 of 1 jobs". */
83
+ function jobsFailed(run , { failedJobs, totalJobs } ) {
84
+ if (failedJobs < totalJobs) return `The ${run} had ${failedJobs} of ${totalJobs} jobs fail`;
85
+ return totalJobs === 1 ? `The ${run}'s only job failed` : `All ${totalJobs} of the ${run}'s jobs failed`;
86
+ }
87
+
88
+ /** The baseline's failures in the parenthesis after the candidate's. */
89
+ function baselineFailures({ failedJobs, totalJobs } ) {
90
+ if (failedJobs === 0) return 'none';
91
+ if (failedJobs < totalJobs) return `${failedJobs} of ${totalJobs}`;
92
+ return totalJobs === 1 ? 'its only job failed' : `all ${totalJobs} failed`;
93
+ }
94
+
95
+ /** The headline when either run had failed jobs, as the run verdict leads
96
+ * with a failure: a faster candidate that dropped work is not an improvement,
97
+ * and with equal failure counts the tone stays neutral since a failing run
98
+ * that ends sooner may just have failed earlier. Null when both runs
99
+ * completed. */
100
+ function failureHeadline(base , cand ) {
101
+ if (base.failedJobs === 0 && cand.failedJobs === 0) return null;
102
+ const tone = cand.failedJobs > base.failedJobs ? 'worse' : cand.failedJobs < base.failedJobs ? 'better' : 'same';
103
+ if (cand.failedJobs === 0) return { title: `${jobsFailed('baseline', base)}; the candidate completed`, tone };
104
+ return { title: `${jobsFailed('candidate', cand)} (baseline: ${baselineFailures(base)})`, tone };
105
+ }
106
+
107
+ /** One plain answer to "did the candidate get better or worse than the baseline", from the
108
+ * comparison's own whole-run metrics and finding-category tallies: run time
109
+ * first, then which cost metrics moved each way past run-to-run noise, then which finding
110
+ * categories appeared or went away. When either run had failed jobs, that
111
+ * leads instead and run time becomes the first sentence. When either log is
112
+ * incomplete, run time is stated as what each log covers, never as faster or
113
+ * slower, and the tone stays neutral. Volume and count metrics (input, output,
114
+ * tasks, executors) are left out because more or less of them is not
115
+ * inherently better or worse. */
116
+ export function summarizeComparison(
117
+ metrics ,
118
+ findings ,
119
+ jobs ,
120
+ ) {
121
+ const wall = metrics.find((metric) => metric.key === 'wallClock');
122
+ let title = 'Run time could not be compared between the baseline and the candidate';
123
+ let tone = 'unknown';
124
+ // A log with no end-of-run record stops where the run was cut off, so its
125
+ // shorter time is not a speed-up: say how much each log covers, neutrally.
126
+ const incompleteRuns = jobs ? (['baseline', 'candidate'] ).filter((run) => jobs[run].incomplete) : [];
127
+ if (wall && wall.baseline != null && wall.baseline > 0 && wall.candidate != null && incompleteRuns.length > 0) {
128
+ const change = wall.candidate - wall.baseline;
129
+ title = Math.abs(change / wall.baseline) < SAME_CHANGE_SHARE
130
+ ? "The candidate's log covers about as much run time as the baseline's"
131
+ : `The candidate's log covers ${formatDuration(Math.abs(change))} ${change < 0 ? 'less' : 'more'} run time than the baseline's`;
132
+ } else if (wall && wall.baseline != null && wall.baseline > 0 && wall.candidate != null) {
133
+ const change = wall.candidate - wall.baseline;
134
+ const share = change / wall.baseline;
135
+ if (Math.abs(share) < SAME_CHANGE_SHARE) {
136
+ title = 'The candidate took about as long as the baseline';
137
+ tone = 'same';
138
+ } else {
139
+ const faster = change < 0;
140
+ title = `The candidate finished ${formatDuration(Math.abs(change))} ${faster ? 'faster' : 'slower'} than the baseline (${Math.round(Math.abs(share) * 100)}%)`;
141
+ tone = faster ? 'better' : 'worse';
142
+ }
143
+ }
144
+
145
+ const cost = metrics.filter((metric) => metric.key !== 'wallClock' && !NEUTRAL_METRIC_KEYS.has(metric.key)).filter(measured);
146
+ const moved = cost.filter(movedPastNoise);
147
+ const worse = moved.filter((metric) => metric.direction === 'regression').map((metric) => metric.label);
148
+ const better = moved.filter((metric) => metric.direction === 'improvement').map((metric) => metric.label);
149
+ const sentences = [];
150
+ const failure = jobs ? failureHeadline(jobs.baseline, jobs.candidate) : null;
151
+ if (failure) {
152
+ sentences.push(`${title}.`);
153
+ title = failure.title;
154
+ tone = failure.tone;
155
+ }
156
+ if (incompleteRuns.length === 2) {
157
+ sentences.push('Neither log has an end-of-run record, so their times cover only what each log captured, not how long the runs took.');
158
+ } else if (incompleteRuns.length === 1) {
159
+ sentences.push(`The ${incompleteRuns[0]}'s log has no end-of-run record, so its time covers only what the log captured, not how long the run took.`);
160
+ }
161
+ if (worse.length > 0) sentences.push(`Worse in the candidate: ${worse.join(', ')}.`);
162
+ if (better.length > 0) sentences.push(`Better in the candidate: ${better.join(', ')}.`);
163
+ if (cost.length > 0 && worse.length === 0 && better.length === 0) sentences.push('Other measured cost metrics look about the same.');
164
+
165
+ const net = netByCategory(findings);
166
+ const introduced = namesWhere(net, (change) => change > 0);
167
+ const resolved = namesWhere(net, (change) => change < 0);
168
+ if (introduced.length > 0) sentences.push(`New or more frequent in the candidate: ${introduced.join(', ')}.`);
169
+ if (resolved.length > 0) sentences.push(`Less frequent in the candidate: ${resolved.join(', ')}.`);
170
+ return { title, tone, sentences };
171
+ }
172
+
173
+ /** The comparison verdict for a core comparison result: the one the dashboard's comparison page
174
+ * leads with, and the one the CLI's --baseline report and MCP compare_runs carry. */
175
+ export function comparisonVerdict(result ) {
176
+ return summarizeComparison(result.metrics, result.findings, result.jobOutcomes);
177
+ }
@@ -0,0 +1 @@
1
+ ab36c5fdcdc1a32a4e6206459e843e8d2267a90e6f1de29671827315d81bc374