sparkforensics-mcp 0.2.4 → 0.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (71) hide show
  1. package/LICENSE +21 -0
  2. package/README.md +7 -1
  3. package/bin/sparkforensics-mcp.mjs +41 -10
  4. package/package.json +1 -1
  5. package/vendor-core/analyzer.js +156 -48
  6. package/vendor-core/check-coverage.js +88 -0
  7. package/vendor-core/cli/budgets.js +31 -18
  8. package/vendor-core/cli/collect-run.js +76 -31
  9. package/vendor-core/cli/native-zstd.js +2 -2
  10. package/vendor-core/cli/threshold-config.js +28 -0
  11. package/vendor-core/comparison-verdict.js +177 -0
  12. package/vendor-core/core-source-hash.txt +1 -0
  13. package/vendor-core/core-usage-locality.js +56 -2
  14. package/vendor-core/detector-docs.js +58 -0
  15. package/vendor-core/detectors.js +933 -459
  16. package/vendor-core/docs-config.js +0 -36
  17. package/vendor-core/docs-content/chapters/nav-index.json +31 -0
  18. package/vendor-core/docs-content/detection/cstor.md +9 -0
  19. package/vendor-core/docs-content/detection/fail.md +6 -2
  20. package/vendor-core/docs-site-config.js +3 -0
  21. package/vendor-core/event-handlers.js +191 -6
  22. package/vendor-core/event-schemas.js +29 -0
  23. package/vendor-core/evidence-report.js +440 -112
  24. package/vendor-core/export-data.js +79 -6
  25. package/vendor-core/finding-action-label.js +9 -88
  26. package/vendor-core/finding-filter-predicate.js +9 -0
  27. package/vendor-core/finding-generic-recommendation.js +6 -104
  28. package/vendor-core/finding-names.js +21 -45
  29. package/vendor-core/finding-presentation.js +333 -0
  30. package/vendor-core/finding-tag-help.js +110 -0
  31. package/vendor-core/finding-types.js +361 -0
  32. package/vendor-core/findings-of-type.js +11 -0
  33. package/vendor-core/format-utils.js +92 -27
  34. package/vendor-core/html-export.js +51 -0
  35. package/vendor-core/impact-band.js +21 -8
  36. package/vendor-core/impact-estimator.js +8 -521
  37. package/vendor-core/impact-format.js +114 -0
  38. package/vendor-core/impact-model.js +175 -0
  39. package/vendor-core/ingest.js +2 -0
  40. package/vendor-core/intervals.js +13 -0
  41. package/vendor-core/list-runs.js +2 -3
  42. package/vendor-core/load-vendored.js +70 -5
  43. package/vendor-core/mcp-server-factory.js +14 -10
  44. package/vendor-core/mcp-tools.js +105 -45
  45. package/vendor-core/model-assembler.js +12 -0
  46. package/vendor-core/occupancy.js +1 -1
  47. package/vendor-core/parser-worker.js +22 -5
  48. package/vendor-core/plan-graph-model.js +3 -2
  49. package/vendor-core/plan-node-detail.js +1 -1
  50. package/vendor-core/recommendation-rollup.js +63 -3
  51. package/vendor-core/redact.js +68 -28
  52. package/vendor-core/run-comparison.js +40 -7
  53. package/vendor-core/run-interpretation.js +290 -0
  54. package/vendor-core/run-outcome.js +74 -0
  55. package/vendor-core/run-payload.js +17 -0
  56. package/vendor-core/run-shape.js +40 -0
  57. package/vendor-core/run-verdict.js +353 -0
  58. package/vendor-core/scaling-sim.js +4 -5
  59. package/vendor-core/scorecard-estimates.js +62 -0
  60. package/vendor-core/shs-fetch.js +175 -65
  61. package/vendor-core/shs-load.js +1 -1
  62. package/vendor-core/sql-stages.js +11 -0
  63. package/vendor-core/stage-quantiles.js +14 -0
  64. package/vendor-core/task-failure.js +151 -0
  65. package/vendor-core/threshold-overrides.js +160 -0
  66. package/vendor-core/threshold-summary.js +11 -33
  67. package/vendor-core/types.js +6 -42
  68. package/vendor-core/vendor/fflate.js +1 -1
  69. package/vendor-core/wall-clock.js +1 -12
  70. package/vendor-core/wasted-core-hours.js +2 -2
  71. package/vendor-core/zip-archive.js +167 -0
@@ -6,16 +6,21 @@ import { collectRun } from './cli/collect-run.js';
6
6
  import { deriveEvidenceAvailability } from './evidence-availability.js';
7
7
  import { resolveFromShs, DEFAULT_MAX_ARCHIVE_BYTES, DEFAULT_IDLE_TIMEOUT_MS } from './shs-load.js';
8
8
  import { mcpError } from './mcp-error.js';
9
- import { buildEvidenceReport, toFindingsFilter, } from './evidence-report.js';
10
- import { redactAppIdentity, redactComparison } from './redact.js';
9
+ import { buildEvidenceReport, toFindingsFilter, } from './evidence-report.js';
10
+ import { redactComparison } from './redact.js';
11
11
  import { computeWallClock } from './wall-clock.js';
12
12
  import { analyze } from './analyzer.js';
13
13
  import { buildComparison, renderComparisonMarkdown, } from './run-comparison.js';
14
+ import { comparisonVerdict, } from './comparison-verdict.js';
14
15
  import { evaluateBudgets, } from './cli/budgets.js';
15
16
  import { FINDING_NAMES, titleCase } from './finding-names.js';
16
- import { docAnchorForType, tuningDocSlugForAnchor, pageForAnchor } from './docs-config.js';
17
+ import { docAnchorForType } from './detector-docs.js';
18
+ import { DETECTORS, } from './detectors.js';
19
+ import { tunedDetectors } from './threshold-overrides.js';
20
+ import { tuningDocSlugForAnchor, pageForAnchor } from './docs-config.js';
17
21
  import { typeTag } from './format-utils.js';
18
-
22
+
23
+
19
24
 
20
25
 
21
26
  // RunRef uses a nested `source` key, matching every real call site (resolveOrCreateRun's
@@ -33,13 +38,22 @@ import { typeTag } from './format-utils.js';
33
38
 
34
39
 
35
40
 
41
+
42
+
43
+
44
+
45
+
46
+
47
+
36
48
 
37
49
  // compareRuns returns a smaller MCP-facing projection of CompareRunsResult
38
- // (runIdA/runIdB/findingsDelta/metricDeltas/confidence/reason/matchedCoverage), not the full raw
39
- // shape (no baselineLabel/stageSkew/baseStages/candStages).
50
+ // (runIdA/runIdB/verdict/findingsDelta/metricDeltas/confidence/reason/matchedCoverage), not the full
51
+ // raw shape (no baselineLabel/stageSkew/baseStages/candStages/jobOutcomes).
40
52
 
41
53
 
42
54
 
55
+
56
+
43
57
 
44
58
 
45
59
 
@@ -169,20 +183,29 @@ export async function resolveOrCreateRun(
169
183
  return resolution;
170
184
  }
171
185
 
186
+ // `thresholds` on every analyzing tool below: the server's --thresholds overrides, fixed for the
187
+ // process by createMcpServer(). A client can't set them per call; the dashboard never has them.
172
188
  export function diagnoseRun(runId , opts
173
189
 
174
-
190
+
175
191
  )
176
-
177
-
192
+
193
+
194
+
178
195
  {
179
196
  const appModel = getCachedAppModel(runId);
180
197
  const findingsFilter = toFindingsFilter(opts?.impactBand, opts?.type, opts?.stageId);
181
- const { json, markdown } = buildEvidenceReport(appModel, { redact: opts?.redact, markdown: opts?.markdown, findingsFilter });
198
+ const { json, markdown } = buildEvidenceReport(appModel, {
199
+ redact: opts?.redact, markdown: opts?.markdown, findingsFilter, thresholds: opts?.thresholds,
200
+ });
182
201
  const include = opts?.include ?? [];
202
+ const tuned = json.summary.tunedThresholds;
183
203
  return {
184
- runId, findings: json.findings, recommendations: json.recommendations, cleanChecks: json.cleanChecks,
204
+ runId, verdict: json.verdict, findings: json.findings, recommendations: json.recommendations, cleanChecks: json.cleanChecks,
205
+ notRunChecks: json.notRunChecks,
185
206
  runComplete: appModel.app?.endTime != null,
207
+ // Top level too, so a client that never asks for `summary` still sees the run was tuned.
208
+ ...(tuned ? { tunedThresholds: tuned } : {}),
186
209
  ...(include.includes('summary') ? { summary: json.summary } : {}),
187
210
  ...(include.includes('evidenceAvailability') ? { evidenceAvailability: json.evidenceAvailability } : {}),
188
211
  ...(include.includes('detectors') ? { detectors: json.detectors } : {}),
@@ -211,17 +234,26 @@ function extractDocTitle(markdown ) {
211
234
 
212
235
 
213
236
 
237
+ // A finding type documents itself; a detector-level type that never appears on a finding
238
+ // (broadcastSizing) documents the types its entry emits.
239
+ function documentedTypes(type ) {
240
+ if (FINDING_NAMES[type] !== undefined) return [type];
241
+ return DETECTORS.find((entry) => entry.type === type)?.emits ?? [];
242
+ }
243
+
214
244
  /** Detection + tuning reference documentation for one finding `type`, independent of any run
215
- * (documentation is a property of the type: a client fetches it once per type and caches it). */
245
+ * (documentation is a property of the type: a client fetches it once per type and caches it).
246
+ * A detector-level `type` resolves to the documentation of the finding types its entry emits. */
216
247
  export function getFindingDocumentation(type ) {
217
- const label = FINDING_NAMES[type];
218
- if (label === undefined) throw mcpError('invalid-type', `Unknown finding type: ${type}`);
248
+ const types = documentedTypes(type);
249
+ const [docType] = types;
250
+ if (docType === undefined) throw mcpError('invalid-type', `Unknown finding type: ${type}`);
219
251
 
220
- const tag = typeTag(type);
252
+ const tag = typeTag(docType);
221
253
  const detectionContent = readFileSync(join(DOCS_CONTENT_DIR, 'detection', `${tag.toLowerCase()}.md`), 'utf8');
222
254
  const detectionDoc = { tag, title: extractDocTitle(detectionContent), content: detectionContent };
223
255
 
224
- const anchor = docAnchorForType(type);
256
+ const anchor = docAnchorForType(docType);
225
257
  const slug = anchor ? tuningDocSlugForAnchor(anchor) : null;
226
258
  const tuningPath = slug ? join(DOCS_CONTENT_DIR, 'tuning', `${slug}.md`) : null;
227
259
  let tuningDoc = null;
@@ -235,7 +267,8 @@ export function getFindingDocumentation(type ) {
235
267
  if (entry) tuningDoc = { anchor, title: entry.title, content: readNavEntryContent(entry) };
236
268
  }
237
269
 
238
- return { type, name: titleCase(label), detectionDoc, tuningDoc };
270
+ const name = types.map((t) => titleCase(FINDING_NAMES[t] ?? t)).join(' / ');
271
+ return { type, name, detectionDoc, tuningDoc };
239
272
  }
240
273
 
241
274
  const CHAPTERS_NAV_FILE = join(DOCS_CONTENT_DIR, 'chapters', 'nav-index.json');
@@ -266,10 +299,10 @@ export function getReferenceDoc(anchor ) {
266
299
  }
267
300
 
268
301
  export function getFindingEvidence(
269
- runId , findingId , opts ,
302
+ runId , findingId , opts ,
270
303
  ) {
271
304
  const appModel = getCachedAppModel(runId);
272
- const { json } = buildEvidenceReport(appModel, { redact: opts?.redact, markdown: false });
305
+ const { json } = buildEvidenceReport(appModel, { redact: opts?.redact, markdown: false, thresholds: opts?.thresholds });
273
306
  const finding = json.findings.find((f) => f.id === findingId);
274
307
  if (!finding) throw mcpError('finding-not-found', `No finding ${findingId} on run ${runId}.`);
275
308
  return { runId, finding };
@@ -279,15 +312,19 @@ function hasCompleteInterval(app ) {
279
312
  return Number.isFinite(app?.startTime) && Number.isFinite(app?.endTime) && (app?.endTime ?? 0) > (app?.startTime ?? 0);
280
313
  }
281
314
 
282
- export function getRunSummary(runId , opts ) {
315
+ export function getRunSummary(runId , opts ) {
283
316
  const appModel = getCachedAppModel(runId);
284
317
  const { app, stages, jobs, sql, executors } = appModel;
285
318
  const durationMs = hasCompleteInterval(app) ? computeWallClock(app, stages).total : null;
286
- // No buildEvidenceReport call here to redact, so reuse redact.ts's app-id + host-token
287
- // pseudonymization directly. Passing name/sparkVersion (not just id) matters: app.name is free
288
- // text and can itself carry a host/IP token.
319
+ // The failure reason needs the stageFailed findings, so this reads the (cached) evidence report.
320
+ // With redact, the app identity comes from that same redacted report, so a host token in the app
321
+ // name and in Spark's failure reason get the same pseudonym.
322
+ const { summary } = buildEvidenceReport(appModel, { redact: opts?.redact, markdown: false, thresholds: opts?.thresholds }).json;
323
+ const { outcome } = summary;
289
324
  const rawApp = { id: app?.id ?? null, name: app?.name ?? null, sparkVersion: app?.sparkVersion ?? null };
290
- const redactedApp = opts?.redact ? redactAppIdentity(rawApp) : rawApp;
325
+ const redactedApp = opts?.redact
326
+ ? { id: summary.app.id ?? null, name: summary.app.name ?? null, sparkVersion: summary.app.sparkVersion ?? null }
327
+ : rawApp;
291
328
  return {
292
329
  runId,
293
330
  app: redactedApp,
@@ -297,27 +334,31 @@ export function getRunSummary(runId , opts )
297
334
  executorCount: { added: executors.added.length, removed: executors.removed.length },
298
335
  durationMs,
299
336
  runComplete: app?.endTime != null,
337
+ ...outcome,
338
+ runShape: summary.runShape,
300
339
  };
301
340
  }
302
341
 
303
342
  // Shared by compareRuns and evaluateBudgetsForRun: both need a resolved run's finding catalog
304
343
  // (same analyze() call shape) first.
305
- async function resolveAndAnalyze(ref ) {
344
+ async function resolveAndAnalyze(
345
+ ref , thresholds ,
346
+ ) {
306
347
  const { runId, appModel } = await resolveOrCreateRun(ref);
307
348
  const catalog = analyze(
308
349
  appModel.app, appModel.stages, appModel.executors.added, appModel.executors.removed,
309
- appModel.jobs, appModel.sql, appModel.runAggregates,
350
+ appModel.jobs, appModel.sql, appModel.runAggregates, { thresholds },
310
351
  );
311
352
  return { runId, appModel, catalog };
312
353
  }
313
354
 
314
355
  export async function compareRuns(
315
- a , b , opts ,
316
- ) {
356
+ a , b , opts ,
357
+ ) {
317
358
  const [
318
359
  { runId: runIdA, appModel: appModelA, catalog: catalogA },
319
360
  { runId: runIdB, appModel: appModelB, catalog: catalogB },
320
- ] = await Promise.all([resolveAndAnalyze(a), resolveAndAnalyze(b)]);
361
+ ] = await Promise.all([resolveAndAnalyze(a, opts?.thresholds), resolveAndAnalyze(b, opts?.thresholds)]);
321
362
 
322
363
  // buildComparison's captureSnapshot uses an empty taskDataCache: that arg only feeds the
323
364
  // interactive stage-detail drill-down, which none of compare/matchStages/metricDeltas/findingsDelta
@@ -330,44 +371,63 @@ export async function compareRuns(
330
371
  // Stage names throughout `built` carry raw Spark stage text, which can embed a host/IP token as
331
372
  // free text, the same residual redactReport() already scrubs from the evidence report.
332
373
  const result = opts?.redact ? redactComparison(built) : built;
374
+ const verdict = comparisonVerdict(result);
333
375
 
334
376
  return {
335
377
  runIdA,
336
378
  runIdB,
379
+ verdict,
337
380
  findingsDelta: result.findings,
338
381
  metricDeltas: result.metrics,
339
382
  confidence: result.confidence,
340
383
  reason: result.reason,
341
384
  matchedCoverage: result.matchedCoverage,
342
- ...(opts?.markdown ? { markdown: renderComparisonMarkdown(result) } : {}),
385
+ ...tunedField(opts?.thresholds),
386
+ ...(opts?.markdown ? { markdown: renderComparisonMarkdown(result, verdict, tunedDetectors(opts?.thresholds)) } : {}),
343
387
  };
344
388
  }
345
389
 
390
+ // Both runs of a comparison use the same overrides, so a tuned delta compares like with like.
391
+ function tunedField(thresholds ) {
392
+ const tuned = tunedDetectors(thresholds);
393
+ return tuned ? { tunedThresholds: tuned } : {};
394
+ }
395
+
346
396
  export async function evaluateBudgetsForRun(
347
397
  primary ,
348
398
  budgets ,
349
399
  secondary ,
350
- ) {
400
+ opts ,
401
+ )
402
+
403
+
404
+ {
351
405
  // Mirrors the CLI's --regression-metric/--max-regression-pct pairing guard: unlike the CLI, this
352
406
  // tool never defaults regressionMetric, so seeing it set here means the caller asked for a
353
407
  // regression check and forgot the threshold, evaluateBudgets() would otherwise skip it silently.
354
408
  if (budgets.regressionMetric !== undefined && budgets.maxRegressionPct === undefined) {
355
409
  throw mcpError('access-or-upstream-failure', 'regressionMetric requires maxRegressionPct.');
356
410
  }
357
- const [{ runId, appModel, catalog }, second] = await Promise.all([
358
- resolveAndAnalyze(primary),
359
- secondary ? resolveAndAnalyze(secondary) : Promise.resolve(undefined),
411
+ const [first, second] = await Promise.all([
412
+ resolveAndAnalyze(primary, opts?.thresholds),
413
+ secondary ? resolveAndAnalyze(secondary, opts?.thresholds) : Promise.resolve(undefined),
360
414
  ]);
361
415
 
362
- let comparison ;
363
- if (second) {
364
- comparison = buildComparison(
365
- { label: runId, appModel, catalog },
366
- { label: second.runId, appModel: second.appModel, catalog: second.catalog },
367
- );
368
- }
369
-
370
- const { results, violated, inconclusive } = evaluateBudgets({ appModel, catalog, budgets, comparison });
371
-
372
- return { runId, results, violated, inconclusive };
416
+ // Same roles as the CLI's positional run + --baseline: with a second run, `primary` is the
417
+ // baseline and `secondary` the candidate, and absolute budgets plus the run-complete check
418
+ // apply to the candidate.
419
+ const candidate = second ?? first;
420
+ const baseline = second ? first : undefined;
421
+ const comparison = baseline
422
+ ? buildComparison(
423
+ { label: baseline.runId, appModel: baseline.appModel, catalog: baseline.catalog },
424
+ { label: candidate.runId, appModel: candidate.appModel, catalog: candidate.catalog },
425
+ )
426
+ : undefined;
427
+
428
+ const { results, violated, inconclusive } = evaluateBudgets({
429
+ appModel: candidate.appModel, catalog: candidate.catalog, budgets, comparison, thresholds: opts?.thresholds,
430
+ });
431
+
432
+ return { runId: first.runId, results, violated, inconclusive, ...tunedField(opts?.thresholds) };
373
433
  }
@@ -63,6 +63,18 @@ export function createModelCallbacks(
63
63
  if (stage) stage.executorMetrics = execMetrics;
64
64
  }
65
65
  },
66
+ // Patch speculation totals that grew after StageCompleted: Spark kills a losing speculative
67
+ // copy only once its stage finishes. `data` is Map<stageId, { speculationWasteMs, speculationWastedAttempts }>.
68
+ onStageSpeculationWaste(data ) {
69
+ const totalsByStage = data ;
70
+ for (const [stageId, totals] of totalsByStage) {
71
+ const stage = appModel.stages.get(stageId);
72
+ if (stage) {
73
+ stage.speculationWasteMs = totals.speculationWasteMs;
74
+ stage.speculationWastedAttempts = totals.speculationWastedAttempts;
75
+ }
76
+ }
77
+ },
66
78
  onDone, onError,
67
79
  };
68
80
  }
@@ -1,7 +1,7 @@
1
1
  // Occupancy-weighted wall-clock attribution for per-finding impact: every
2
2
  // stage's wall-clock claim is apportioned purely from its own observed window
3
3
  // and how much it overlapped with other stages. No graph, no parentIds traversal.
4
- import { mergeIntervals } from './wall-clock.js';
4
+ import { mergeIntervals } from './intervals.js';
5
5
 
6
6
 
7
7
 
@@ -4,14 +4,15 @@ import { Decompress as ZstdDecompress } from './vendor/fzstd.js';
4
4
  import { createSnappyBlockDecoder } from './snappy-block.js';
5
5
  import { createState, dispatchLine, buildChunkDecoder, emitParseCompletion, } from './event-handlers.js';
6
6
  import { TASK_FIELD_NAMES } from './stage-quantiles.js';
7
- import { runParseFromUrl, sniffCodec } from './shs-fetch.js';
7
+ import { runParseFromUrl, sniffCodec, parseZipArchive } from './shs-fetch.js';
8
+ import { isZip } from './zip-archive.js';
8
9
  import { createWorkerZstdDecoders } from './zstd-worker-client.js';
9
10
 
10
11
  export {
11
12
  buildChunkDecoder, createState, normalizeSparkProperties, parseSparkMemoryMB, extractResources,
12
13
  accumulateTask, resolvePlanTree, startApplication, updateEnvironment, startJob, endJob, submitStage,
13
14
  mergeStageRddInfo, recordStageExecutorMetrics, startSqlExecution, endSqlExecution,
14
- applyDriverAccumUpdates, addExecutor, removeExecutor, processEvent, dispatchLine,
15
+ applyDriverAccumUpdates, addExecutor, removeExecutor, recordBlockUpdate, processEvent, dispatchLine,
15
16
  collectStageExecutorMetrics,
16
17
  } from './event-handlers.js';
17
18
 
@@ -35,6 +36,8 @@ const MIN_PROGRESS_STEPS = 100;
35
36
  const PROGRESS_EMIT_LINES = 300;
36
37
 
37
38
 
39
+ const NOT_AN_EVENT_LOG = 'Not a Spark event log: no application-start event found. Choose a Spark event log file, or check the docs for supported formats.';
40
+
38
41
  // zstdDecoder replaces the vendored fzstd for zstd input; the Node CLI/MCP path passes
39
42
  // cli/native-zstd.ts's native-zlib decoder, which a browser bundle can't import.
40
43
 
@@ -140,6 +143,20 @@ export async function runParse(
140
143
  return;
141
144
  }
142
145
 
146
+ // A Spark History Server download (the UI's download link, or GET
147
+ // /api/v1/applications/<id>/logs) is a zip holding the log file, or a
148
+ // rolling log's parts: unwrap it through the same path the SHS fetch uses.
149
+ if (isZip(new Uint8Array(await file.slice(0, Math.min(4, file.size)).arrayBuffer()))) {
150
+ await parseZipArchive(file, state, emit, {
151
+ zstdDecoder,
152
+ chunkSize,
153
+ progressEvery: PROGRESS_EMIT_LINES,
154
+ reportPct: true,
155
+ onInvalid: (detail) => emit({ type: 'error', message: detail ?? NOT_AN_EVENT_LOG }),
156
+ });
157
+ return;
158
+ }
159
+
143
160
  const decoder = buildChunkDecoder();
144
161
  const joined = [];
145
162
  let linesProcessed = 0;
@@ -168,7 +185,7 @@ export async function runParse(
168
185
  }
169
186
 
170
187
  if (!state.app) {
171
- emit({ type: 'error', message: 'Not a Spark event log: no application-start event found. Choose a Spark event log file, or check the docs for supported formats.' });
188
+ emit({ type: 'error', message: NOT_AN_EVENT_LOG });
172
189
  return;
173
190
  }
174
191
 
@@ -227,7 +244,7 @@ export async function runParseFiles(
227
244
  }
228
245
 
229
246
  if (!state.app) {
230
- emit({ type: 'error', message: 'Not a Spark event log: no application-start event found. Choose a Spark event log file, or check the docs for supported formats.' });
247
+ emit({ type: 'error', message: NOT_AN_EVENT_LOG });
231
248
  return;
232
249
  }
233
250
 
@@ -242,7 +259,7 @@ if (isWorker) {
242
259
  let workerState = null;
243
260
  // Dropped zstd files decompress in a second worker, overlapping with parsing here. The
244
261
  // `new Worker(new URL(...))` stays inline for Vite's worker detection (see ingest.ts). The SHS
245
- // path (runParseFromUrl) decodes whole zip entries synchronously and keeps in-thread fzstd.
262
+ // path (runParseFromUrl) keeps in-thread fzstd.
246
263
  const zstdDecoder = createWorkerZstdDecoders(
247
264
  () => new Worker(new URL('./zstd-worker.js', import.meta.url), { type: 'module' }),
248
265
  (onChunk) => new (ZstdDecompress )(onChunk),
@@ -15,7 +15,7 @@ import {
15
15
  formatPlanMetricValue,
16
16
  } from './plan-node-detail.js';
17
17
  import { computeSegments, mapSegmentsToStagesForDisplay } from './plan-duration-attribution.js';
18
- import { stageIdsForSqlExec } from './detectors.js';
18
+ import { stageIdsForSqlExec } from './sql-stages.js';
19
19
 
20
20
 
21
21
 
@@ -121,7 +121,8 @@ export function buildPlanGraphModel(
121
121
  const findingsByNodeId = new Map ();
122
122
  if (sqlExecutionId != null) {
123
123
  for (const finding of findings) {
124
- if (finding.executionId !== sqlExecutionId) continue;
124
+ if (!('executionId' in finding) || finding.executionId !== sqlExecutionId) continue;
125
+ // `?? []`: a hand-built or foreign finding may still lack the field the type promises.
125
126
  for (const nodeId of finding.planNodeIds ?? []) {
126
127
  const list = findingsByNodeId.get(nodeId);
127
128
  if (list) list.push(finding);
@@ -5,7 +5,7 @@
5
5
 
6
6
  import { pathBasename, formatBytes, formatDuration } from './format-utils.js';
7
7
  import { attributeStageDurationToPlan, attributeStageDurationToPlanInclusive } from './plan-duration-attribution.js';
8
- import { stageIdsForSqlExec } from './detectors.js';
8
+ import { stageIdsForSqlExec } from './sql-stages.js';
9
9
 
10
10
 
11
11
  const BOILERPLATE_PREFIXES = ['serializefromobject', 'deserializetoobject', 'mapelements',
@@ -1,7 +1,7 @@
1
1
  // Explicit .ts extensions: plain Node's ESM resolver (the runtime CLI/MCP path
2
2
  // runs under) requires the exact specifier, unlike a bundler.
3
- import { mergeIntervals } from './wall-clock.js';
4
- import { worstImpactBand, IMPACT_BAND_ORDER } from './format-utils.js';
3
+ import { mergeIntervals } from './intervals.js';
4
+ import { formatWallClockRange, worstImpactBand, IMPACT_BAND_ORDER } from './format-utils.js';
5
5
 
6
6
 
7
7
 
@@ -48,7 +48,7 @@ function groupByType(findings ) {
48
48
  }
49
49
 
50
50
  function stageIdsOf(finding ) {
51
- if (finding.stageIds) return finding.stageIds;
51
+ if ('stageIds' in finding) return finding.stageIds;
52
52
  if (finding.stageId != null) return [finding.stageId];
53
53
  return [];
54
54
  }
@@ -124,6 +124,31 @@ export function buildRecommendationRollup(
124
124
  });
125
125
  }
126
126
 
127
+ /** A group's trailing figure on the Findings board, and its tooltip. `time`
128
+ * and `resource` both lead with the "×N" finding count (the "worth
129
+ * expanding" signal); `count` skips it since the impact-band tally already
130
+ * implies N. The row stays terse ("×2 · 476ms recoverable") to fit a dense
131
+ * right-aligned column; the title spells the shorthand out. */
132
+ export function rollupGroupStat(group ) {
133
+ if (group.kind === 'time') {
134
+ const recoverable = formatWallClockRange(group.recoverableMsHigh, group.recoverableMsHigh);
135
+ return {
136
+ stat: `×${group.findingCount} · ${recoverable} recoverable`,
137
+ statTitle: `${group.findingCount} findings of this type; up to ${recoverable} of run time could be recovered by fixing them`,
138
+ };
139
+ }
140
+ if (group.kind === 'resource') {
141
+ return {
142
+ stat: `×${group.findingCount} · resource-cost projection`,
143
+ statTitle: `${group.findingCount} findings of this type; a resource-cost estimate (not run time) is projected for fixing them`,
144
+ };
145
+ }
146
+ return {
147
+ stat: Object.entries(group.byImpactBand).map(([impactBand, count]) => `${count} ${impactBand}`).join(', '),
148
+ statTitle: `${group.findingCount} findings of this type, by impact`,
149
+ };
150
+ }
151
+
127
152
  /** True when a finding is real evidence of an issue, as opposed to a mere
128
153
  * evidence-unavailable caveat. Shared by `isEligible` below and by anything
129
154
  * else that decides whether a REGISTRY type has "something to show" (an
@@ -150,6 +175,10 @@ export function isEligible(finding ) {
150
175
  // `isRealFinding`, this exclusion doesn't apply beyond the rollup: an
151
176
  // incompleteRun finding still backs its own ordinary active widget card.
152
177
  if (finding.type === 'incompleteRun') return false;
178
+ // A missing-evidence caveat (memoryUtilization's is already excluded by isRealFinding;
179
+ // cacheUtilization's storageUnobserved keeps its widget card, so a run with no block updates
180
+ // never lists Cache Storage as a passed check) is never a fix to rank.
181
+ if ('dataUnavailable' in finding && finding.dataUnavailable) return false;
153
182
  return isRealFinding(finding);
154
183
  }
155
184
 
@@ -191,3 +220,34 @@ export function rankFindings(findings ) {
191
220
  return (IMPACT_BAND_ORDER[a.impactBand] ?? 9) - (IMPACT_BAND_ORDER[b.impactBand] ?? 9);
192
221
  });
193
222
  }
223
+
224
+ /** A board group ready to render: members ranked representative first, the representative's
225
+ * band (the heading the group sits under) and the group's trailing figure. */
226
+
227
+
228
+
229
+
230
+
231
+
232
+
233
+
234
+
235
+
236
+ /** The Findings board's groups over `eligible`, in fix-first order. The run's interpretation
237
+ * carries this for every eligible finding; the board recomputes it only over a filtered subset. */
238
+ export function rankedRollup(
239
+ eligible ,
240
+ stages ,
241
+ ) {
242
+ return buildRecommendationRollup(eligible, stages).map((group) => {
243
+ const members = rankFindings(group.findings);
244
+ return {
245
+ kind: group.kind,
246
+ type: group.type,
247
+ unit: group.kind === 'resource' ? group.unit : null,
248
+ band: members[0].impactBand,
249
+ members,
250
+ ...rollupGroupStat(group),
251
+ };
252
+ });
253
+ }