sparkforensics-cli 0.1.0 → 0.2.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (361) hide show
  1. package/README.md +6 -0
  2. package/bin/sparkforensics-analyze.mjs +113 -48
  3. package/export-template/docs/404.html +25 -0
  4. package/export-template/docs/assets/app.DQTZyGL1.js +1 -0
  5. package/export-template/docs/assets/aqe-loop.IwQSATHw.svg +1 -0
  6. package/export-template/docs/assets/aqe-loop.dark.DGbaxqJE.svg +1 -0
  7. package/export-template/docs/assets/broadcast-vs-shuffle.Db4WY1XK.svg +1 -0
  8. package/export-template/docs/assets/broadcast-vs-shuffle.dark.C7Bxs0mG.svg +1 -0
  9. package/export-template/docs/assets/cache-lifecycle.dark.B-hS7AgU.svg +1 -0
  10. package/export-template/docs/assets/cache-lifecycle.rEOVYQNU.svg +1 -0
  11. package/export-template/docs/assets/chunks/@localSearchIndexroot.DppnXnDE.js +1 -0
  12. package/export-template/docs/assets/chunks/VPLocalSearchBox.BkBIPFs6.js +9 -0
  13. package/export-template/docs/assets/chunks/duplicate-plan-subtree.dark.Cdp70QhV.js +1 -0
  14. package/export-template/docs/assets/chunks/framework.DSg0KOwT.js +20 -0
  15. package/export-template/docs/assets/chunks/retry-escalation-ladder.dark.DHipdJgZ.js +1 -0
  16. package/export-template/docs/assets/chunks/theme.DP0u1AUq.js +2 -0
  17. package/export-template/docs/assets/cold-start-timeline.DxC_Sc7w.svg +1 -0
  18. package/export-template/docs/assets/cold-start-timeline.dark.CZ17YcAG.svg +1 -0
  19. package/export-template/docs/assets/columnar-layout.PghGeOEA.svg +1 -0
  20. package/export-template/docs/assets/columnar-layout.dark.BVNlz0ff.svg +1 -0
  21. package/export-template/docs/assets/container-memory.DIO0AnIm.svg +1 -0
  22. package/export-template/docs/assets/container-memory.dark.CP-5zuCl.svg +1 -0
  23. package/export-template/docs/assets/contributor-guide_architecture_board-widgets.md.CWpj01WU.js +1 -0
  24. package/export-template/docs/assets/contributor-guide_architecture_board-widgets.md.CWpj01WU.lean.js +1 -0
  25. package/export-template/docs/assets/contributor-guide_architecture_detector-contract.md.CgzUsQ6W.js +1 -0
  26. package/export-template/docs/assets/contributor-guide_architecture_detector-contract.md.CgzUsQ6W.lean.js +1 -0
  27. package/export-template/docs/assets/contributor-guide_architecture_drill-down.md.BtPdlM7r.js +1 -0
  28. package/export-template/docs/assets/contributor-guide_architecture_drill-down.md.BtPdlM7r.lean.js +1 -0
  29. package/export-template/docs/assets/contributor-guide_architecture_impact-estimation.md.CooslVJt.js +1 -0
  30. package/export-template/docs/assets/contributor-guide_architecture_impact-estimation.md.CooslVJt.lean.js +1 -0
  31. package/export-template/docs/assets/contributor-guide_architecture_index.md.3TO9ic6w.js +1 -0
  32. package/export-template/docs/assets/contributor-guide_architecture_index.md.3TO9ic6w.lean.js +1 -0
  33. package/export-template/docs/assets/contributor-guide_architecture_overview.md.CehiRmGn.js +1 -0
  34. package/export-template/docs/assets/contributor-guide_architecture_overview.md.CehiRmGn.lean.js +1 -0
  35. package/export-template/docs/assets/contributor-guide_architecture_state-and-history.md.C-xxn0q7.js +1 -0
  36. package/export-template/docs/assets/contributor-guide_architecture_state-and-history.md.C-xxn0q7.lean.js +1 -0
  37. package/export-template/docs/assets/contributor-guide_architecture_widget-rendering.md.R27gQrgY.js +1 -0
  38. package/export-template/docs/assets/contributor-guide_architecture_widget-rendering.md.R27gQrgY.lean.js +1 -0
  39. package/export-template/docs/assets/contributor-guide_architecture_worker-protocol.md.IbnfNrV3.js +6 -0
  40. package/export-template/docs/assets/contributor-guide_architecture_worker-protocol.md.IbnfNrV3.lean.js +1 -0
  41. package/export-template/docs/assets/contributor-guide_contributing.md.CvRsdr6J.js +1 -0
  42. package/export-template/docs/assets/contributor-guide_contributing.md.CvRsdr6J.lean.js +1 -0
  43. package/export-template/docs/assets/contributor-guide_development-setup.md.DvAN_9mK.js +12 -0
  44. package/export-template/docs/assets/contributor-guide_development-setup.md.DvAN_9mK.lean.js +1 -0
  45. package/export-template/docs/assets/contributor-guide_testing.md.6rIKqSyY.js +1 -0
  46. package/export-template/docs/assets/contributor-guide_testing.md.6rIKqSyY.lean.js +1 -0
  47. package/export-template/docs/assets/dag-stages.DSz_S937.svg +1 -0
  48. package/export-template/docs/assets/dag-stages.dark.F72UzxH4.svg +1 -0
  49. package/export-template/docs/assets/driver-executor.D5pQ7YN1.svg +1 -0
  50. package/export-template/docs/assets/driver-executor.dark.BmX9cPvh.svg +1 -0
  51. package/export-template/docs/assets/duplicate-plan-subtree.B4cvN6fj.svg +1 -0
  52. package/export-template/docs/assets/duplicate-plan-subtree.dark.Dw8wS0Ag.svg +1 -0
  53. package/export-template/docs/assets/index.md.CHJVslga.js +1 -0
  54. package/export-template/docs/assets/index.md.CHJVslga.lean.js +1 -0
  55. package/export-template/docs/assets/inter-italic-cyrillic-ext.r48I6akx.woff2 +0 -0
  56. package/export-template/docs/assets/inter-italic-cyrillic.By2_1cv3.woff2 +0 -0
  57. package/export-template/docs/assets/inter-italic-greek-ext.1u6EdAuj.woff2 +0 -0
  58. package/export-template/docs/assets/inter-italic-greek.DJ8dCoTZ.woff2 +0 -0
  59. package/export-template/docs/assets/inter-italic-latin-ext.CN1xVJS-.woff2 +0 -0
  60. package/export-template/docs/assets/inter-italic-latin.C2AdPX0b.woff2 +0 -0
  61. package/export-template/docs/assets/inter-italic-vietnamese.BSbpV94h.woff2 +0 -0
  62. package/export-template/docs/assets/inter-roman-cyrillic-ext.BBPuwvHQ.woff2 +0 -0
  63. package/export-template/docs/assets/inter-roman-cyrillic.C5lxZ8CY.woff2 +0 -0
  64. package/export-template/docs/assets/inter-roman-greek-ext.CqjqNYQ-.woff2 +0 -0
  65. package/export-template/docs/assets/inter-roman-greek.BBVDIX6e.woff2 +0 -0
  66. package/export-template/docs/assets/inter-roman-latin-ext.4ZJIpNVo.woff2 +0 -0
  67. package/export-template/docs/assets/inter-roman-latin.Di8DUHzh.woff2 +0 -0
  68. package/export-template/docs/assets/inter-roman-vietnamese.BjW4sHH5.woff2 +0 -0
  69. package/export-template/docs/assets/join-strategy.C_FvrCEo.svg +1 -0
  70. package/export-template/docs/assets/join-strategy.dark.ChMLnNII.svg +1 -0
  71. package/export-template/docs/assets/memory-borrowing.BqQRJg0u.svg +1 -0
  72. package/export-template/docs/assets/memory-borrowing.dark.Yhh20O9C.svg +1 -0
  73. package/export-template/docs/assets/memory-regions.XHvO7jHG.svg +1 -0
  74. package/export-template/docs/assets/memory-regions.dark.D4TP9_08.svg +1 -0
  75. package/export-template/docs/assets/repartition-vs-coalesce.BovLRrpj.svg +1 -0
  76. package/export-template/docs/assets/repartition-vs-coalesce.dark.BhAczKZQ.svg +1 -0
  77. package/export-template/docs/assets/retry-escalation-ladder.DyTKJJmZ.svg +1 -0
  78. package/export-template/docs/assets/retry-escalation-ladder.dark.BdsabtU3.svg +1 -0
  79. package/export-template/docs/assets/shuffle-map-reduce.KuOEZVmg.svg +1 -0
  80. package/export-template/docs/assets/shuffle-map-reduce.dark.BgQZnFSb.svg +1 -0
  81. package/export-template/docs/assets/spill-classification.BU2euYDO.svg +1 -0
  82. package/export-template/docs/assets/spill-classification.dark.D7i1M40d.svg +1 -0
  83. package/export-template/docs/assets/style.DSixAiZE.css +1 -0
  84. package/export-template/docs/assets/tuning-reference_anti-patterns.md.Df1YMIHu.js +1 -0
  85. package/export-template/docs/assets/tuning-reference_anti-patterns.md.Df1YMIHu.lean.js +1 -0
  86. package/export-template/docs/assets/tuning-reference_aqe.md.BIsCtLzm.js +1 -0
  87. package/export-template/docs/assets/tuning-reference_aqe.md.BIsCtLzm.lean.js +1 -0
  88. package/export-template/docs/assets/tuning-reference_bottleneck-broadcast-sizing.md.CEstB3Ia.js +1 -0
  89. package/export-template/docs/assets/tuning-reference_bottleneck-broadcast-sizing.md.CEstB3Ia.lean.js +1 -0
  90. package/export-template/docs/assets/tuning-reference_bottleneck-cold-start.md.CEuy-72y.js +7 -0
  91. package/export-template/docs/assets/tuning-reference_bottleneck-cold-start.md.CEuy-72y.lean.js +1 -0
  92. package/export-template/docs/assets/tuning-reference_bottleneck-duplicate-plan-subtree.md.CIohQDfn.js +1 -0
  93. package/export-template/docs/assets/tuning-reference_bottleneck-duplicate-plan-subtree.md.CIohQDfn.lean.js +1 -0
  94. package/export-template/docs/assets/tuning-reference_bottleneck-failures.md.4z5BXGJ2.js +6 -0
  95. package/export-template/docs/assets/tuning-reference_bottleneck-failures.md.4z5BXGJ2.lean.js +1 -0
  96. package/export-template/docs/assets/tuning-reference_bottleneck-gc.md.DSxzZRK7.js +6 -0
  97. package/export-template/docs/assets/tuning-reference_bottleneck-gc.md.DSxzZRK7.lean.js +1 -0
  98. package/export-template/docs/assets/tuning-reference_bottleneck-job-failure-rate.md.BaJl__1W.js +8 -0
  99. package/export-template/docs/assets/tuning-reference_bottleneck-job-failure-rate.md.BaJl__1W.lean.js +1 -0
  100. package/export-template/docs/assets/tuning-reference_bottleneck-memory-utilization.md.DbP-SJZc.js +7 -0
  101. package/export-template/docs/assets/tuning-reference_bottleneck-memory-utilization.md.DbP-SJZc.lean.js +1 -0
  102. package/export-template/docs/assets/tuning-reference_bottleneck-retry-waste.md.D5JMjVOt.js +1 -0
  103. package/export-template/docs/assets/tuning-reference_bottleneck-retry-waste.md.D5JMjVOt.lean.js +1 -0
  104. package/export-template/docs/assets/tuning-reference_bottleneck-shuffle.md.CM-nTmIH.js +12 -0
  105. package/export-template/docs/assets/tuning-reference_bottleneck-shuffle.md.CM-nTmIH.lean.js +1 -0
  106. package/export-template/docs/assets/tuning-reference_bottleneck-skew.md.BdUwiDhn.js +14 -0
  107. package/export-template/docs/assets/tuning-reference_bottleneck-skew.md.BdUwiDhn.lean.js +1 -0
  108. package/export-template/docs/assets/tuning-reference_bottleneck-slow-host.md.BlIo6UDW.js +7 -0
  109. package/export-template/docs/assets/tuning-reference_bottleneck-slow-host.md.BlIo6UDW.lean.js +1 -0
  110. package/export-template/docs/assets/tuning-reference_bottleneck-small-files.md.B8kloyx8.js +5 -0
  111. package/export-template/docs/assets/tuning-reference_bottleneck-small-files.md.B8kloyx8.lean.js +1 -0
  112. package/export-template/docs/assets/tuning-reference_bottleneck-spill.md.PNH7mITt.js +6 -0
  113. package/export-template/docs/assets/tuning-reference_bottleneck-spill.md.PNH7mITt.lean.js +1 -0
  114. package/export-template/docs/assets/tuning-reference_bottleneck-straggler.md.DY36fHN5.js +7 -0
  115. package/export-template/docs/assets/tuning-reference_bottleneck-straggler.md.DY36fHN5.lean.js +1 -0
  116. package/export-template/docs/assets/tuning-reference_bottleneck-tiny-tasks.md.QTV7O8kU.js +8 -0
  117. package/export-template/docs/assets/tuning-reference_bottleneck-tiny-tasks.md.QTV7O8kU.lean.js +1 -0
  118. package/export-template/docs/assets/tuning-reference_bottleneck-utilization.md.DTiueZC3.js +5 -0
  119. package/export-template/docs/assets/tuning-reference_bottleneck-utilization.md.DTiueZC3.lean.js +1 -0
  120. package/export-template/docs/assets/tuning-reference_caching.md.B7aQ8asB.js +1 -0
  121. package/export-template/docs/assets/tuning-reference_caching.md.B7aQ8asB.lean.js +1 -0
  122. package/export-template/docs/assets/tuning-reference_cluster-config.md.ZVmDGsQ3.js +1 -0
  123. package/export-template/docs/assets/tuning-reference_cluster-config.md.ZVmDGsQ3.lean.js +1 -0
  124. package/export-template/docs/assets/tuning-reference_config.md.UvveiWG3.js +1 -0
  125. package/export-template/docs/assets/tuning-reference_config.md.UvveiWG3.lean.js +1 -0
  126. package/export-template/docs/assets/tuning-reference_data-formats.md.bjCAWH3N.js +1 -0
  127. package/export-template/docs/assets/tuning-reference_data-formats.md.bjCAWH3N.lean.js +1 -0
  128. package/export-template/docs/assets/tuning-reference_index.md.BQ_NooMV.js +1 -0
  129. package/export-template/docs/assets/tuning-reference_index.md.BQ_NooMV.lean.js +1 -0
  130. package/export-template/docs/assets/tuning-reference_intro.md.CobD-lGB.js +1 -0
  131. package/export-template/docs/assets/tuning-reference_intro.md.CobD-lGB.lean.js +1 -0
  132. package/export-template/docs/assets/tuning-reference_joins.md.BtKs_CuW.js +1 -0
  133. package/export-template/docs/assets/tuning-reference_joins.md.BtKs_CuW.lean.js +1 -0
  134. package/export-template/docs/assets/tuning-reference_memory-model.md.DhT-n4y3.js +1 -0
  135. package/export-template/docs/assets/tuning-reference_memory-model.md.DhT-n4y3.lean.js +1 -0
  136. package/export-template/docs/assets/tuning-reference_metrics.md.mLOh7Apj.js +1 -0
  137. package/export-template/docs/assets/tuning-reference_metrics.md.mLOh7Apj.lean.js +1 -0
  138. package/export-template/docs/assets/tuning-reference_partitioning.md.q0zKF_8X.js +1 -0
  139. package/export-template/docs/assets/tuning-reference_partitioning.md.q0zKF_8X.lean.js +1 -0
  140. package/export-template/docs/assets/tuning-reference_pyspark.md.DDCfvN9t.js +6 -0
  141. package/export-template/docs/assets/tuning-reference_pyspark.md.DDCfvN9t.lean.js +1 -0
  142. package/export-template/docs/assets/tuning-reference_shuffle.md.BZZ7R4Ix.js +1 -0
  143. package/export-template/docs/assets/tuning-reference_shuffle.md.BZZ7R4Ix.lean.js +1 -0
  144. package/export-template/docs/assets/tuning-reference_spark-architecture.md.Dwzm5avO.js +1 -0
  145. package/export-template/docs/assets/tuning-reference_spark-architecture.md.Dwzm5avO.lean.js +1 -0
  146. package/export-template/docs/assets/tuning-reference_table-formats.md.D6wj-2dX.js +1 -0
  147. package/export-template/docs/assets/tuning-reference_table-formats.md.D6wj-2dX.lean.js +1 -0
  148. package/export-template/docs/assets/udf-execution-models.BUFDICuG.svg +1 -0
  149. package/export-template/docs/assets/udf-execution-models.dark.YTNS6GDq.svg +1 -0
  150. package/export-template/docs/assets/user-guide_alternative-log-retrieval.md.B4tPGIal.js +1 -0
  151. package/export-template/docs/assets/user-guide_alternative-log-retrieval.md.B4tPGIal.lean.js +1 -0
  152. package/export-template/docs/assets/user-guide_getting-started.md.BJvwLEIM.js +3 -0
  153. package/export-template/docs/assets/user-guide_getting-started.md.BJvwLEIM.lean.js +1 -0
  154. package/export-template/docs/assets/user-guide_mcp-tools.md.Vi3RoflJ.js +125 -0
  155. package/export-template/docs/assets/user-guide_mcp-tools.md.Vi3RoflJ.lean.js +1 -0
  156. package/export-template/docs/assets/user-guide_run-comparison.md.CQc1aoU8.js +1 -0
  157. package/export-template/docs/assets/user-guide_run-comparison.md.CQc1aoU8.lean.js +1 -0
  158. package/export-template/docs/assets/user-guide_understanding-findings.md.DL1UDhvR.js +1 -0
  159. package/export-template/docs/assets/user-guide_understanding-findings.md.DL1UDhvR.lean.js +1 -0
  160. package/export-template/docs/contributor-guide/architecture/board-widgets.html +25 -0
  161. package/export-template/docs/contributor-guide/architecture/detector-contract.html +25 -0
  162. package/export-template/docs/contributor-guide/architecture/drill-down.html +25 -0
  163. package/export-template/docs/contributor-guide/architecture/impact-estimation.html +25 -0
  164. package/export-template/docs/contributor-guide/architecture/index.html +25 -0
  165. package/export-template/docs/contributor-guide/architecture/overview.html +25 -0
  166. package/export-template/docs/contributor-guide/architecture/state-and-history.html +25 -0
  167. package/export-template/docs/contributor-guide/architecture/widget-rendering.html +25 -0
  168. package/export-template/docs/contributor-guide/architecture/worker-protocol.html +30 -0
  169. package/export-template/docs/contributor-guide/contributing.html +25 -0
  170. package/export-template/docs/contributor-guide/development-setup.html +36 -0
  171. package/export-template/docs/contributor-guide/testing.html +25 -0
  172. package/export-template/docs/favicon.svg +4 -0
  173. package/export-template/docs/hashmap.json +1 -0
  174. package/export-template/docs/index.html +25 -0
  175. package/export-template/docs/package.json +1 -0
  176. package/export-template/docs/tuning-reference/anti-patterns.html +25 -0
  177. package/export-template/docs/tuning-reference/aqe.html +25 -0
  178. package/export-template/docs/tuning-reference/bottleneck-broadcast-sizing.html +25 -0
  179. package/export-template/docs/tuning-reference/bottleneck-cold-start.html +31 -0
  180. package/export-template/docs/tuning-reference/bottleneck-duplicate-plan-subtree.html +25 -0
  181. package/export-template/docs/tuning-reference/bottleneck-failures.html +30 -0
  182. package/export-template/docs/tuning-reference/bottleneck-gc.html +30 -0
  183. package/export-template/docs/tuning-reference/bottleneck-job-failure-rate.html +32 -0
  184. package/export-template/docs/tuning-reference/bottleneck-memory-utilization.html +31 -0
  185. package/export-template/docs/tuning-reference/bottleneck-retry-waste.html +25 -0
  186. package/export-template/docs/tuning-reference/bottleneck-shuffle.html +36 -0
  187. package/export-template/docs/tuning-reference/bottleneck-skew.html +38 -0
  188. package/export-template/docs/tuning-reference/bottleneck-slow-host.html +31 -0
  189. package/export-template/docs/tuning-reference/bottleneck-small-files.html +29 -0
  190. package/export-template/docs/tuning-reference/bottleneck-spill.html +30 -0
  191. package/export-template/docs/tuning-reference/bottleneck-straggler.html +31 -0
  192. package/export-template/docs/tuning-reference/bottleneck-tiny-tasks.html +32 -0
  193. package/export-template/docs/tuning-reference/bottleneck-utilization.html +29 -0
  194. package/export-template/docs/tuning-reference/caching.html +25 -0
  195. package/export-template/docs/tuning-reference/cluster-config.html +25 -0
  196. package/export-template/docs/tuning-reference/config.html +25 -0
  197. package/export-template/docs/tuning-reference/data-formats.html +25 -0
  198. package/export-template/docs/tuning-reference/index.html +25 -0
  199. package/export-template/docs/tuning-reference/intro.html +25 -0
  200. package/export-template/docs/tuning-reference/joins.html +25 -0
  201. package/export-template/docs/tuning-reference/memory-model.html +25 -0
  202. package/export-template/docs/tuning-reference/metrics.html +25 -0
  203. package/export-template/docs/tuning-reference/partitioning.html +25 -0
  204. package/export-template/docs/tuning-reference/pyspark.html +30 -0
  205. package/export-template/docs/tuning-reference/shuffle.html +25 -0
  206. package/export-template/docs/tuning-reference/spark-architecture.html +25 -0
  207. package/export-template/docs/tuning-reference/table-formats.html +25 -0
  208. package/export-template/docs/user-guide/alternative-log-retrieval.html +25 -0
  209. package/export-template/docs/user-guide/getting-started.html +27 -0
  210. package/export-template/docs/user-guide/mcp-tools.html +149 -0
  211. package/export-template/docs/user-guide/run-comparison.html +25 -0
  212. package/export-template/docs/user-guide/understanding-findings.html +25 -0
  213. package/export-template/docs/vp-icons.css +0 -0
  214. package/export-template/favicon.svg +4 -0
  215. package/export-template/index.html +111 -0
  216. package/export-template/parser-worker-DyjiQvfP.js +112 -0
  217. package/export-template/sample-runs/sample-run.ndjson.gz +0 -0
  218. package/package.json +20 -6
  219. package/vendor-core/analyzer.js +74 -74
  220. package/vendor-core/cli/budgets.js +13 -27
  221. package/vendor-core/cli/collect-run.js +43 -19
  222. package/vendor-core/core-count.js +25 -27
  223. package/vendor-core/core-locality-ratio.js +4 -11
  224. package/vendor-core/core-time-series.js +6 -12
  225. package/vendor-core/core-usage-locality.js +3 -4
  226. package/vendor-core/detectors.js +395 -389
  227. package/vendor-core/docs-config.js +69 -21
  228. package/vendor-core/docs-content/chapters/01-intro.md +32 -0
  229. package/vendor-core/docs-content/chapters/02-spark-architecture.md +76 -0
  230. package/vendor-core/docs-content/chapters/03-memory-model.md +73 -0
  231. package/vendor-core/docs-content/chapters/04-partitioning.md +65 -0
  232. package/vendor-core/docs-content/chapters/05-joins.md +62 -0
  233. package/vendor-core/docs-content/chapters/06-shuffle.md +59 -0
  234. package/vendor-core/docs-content/chapters/07-data-formats.md +81 -0
  235. package/vendor-core/docs-content/chapters/07b-table-formats.md +56 -0
  236. package/vendor-core/docs-content/chapters/08-caching.md +58 -0
  237. package/vendor-core/docs-content/chapters/09-pyspark.md +78 -0
  238. package/vendor-core/docs-content/chapters/10-aqe.md +167 -0
  239. package/vendor-core/docs-content/chapters/11-cluster-config.md +170 -0
  240. package/vendor-core/docs-content/chapters/12-anti-patterns.md +171 -0
  241. package/vendor-core/docs-content/chapters/14-metrics.md +87 -0
  242. package/vendor-core/docs-content/chapters/15-config.md +93 -0
  243. package/vendor-core/docs-content/chapters/nav-index.json +370 -0
  244. package/vendor-core/docs-content/detection/cache.md +7 -0
  245. package/vendor-core/docs-content/detection/cfg.md +15 -0
  246. package/vendor-core/docs-content/detection/chrn.md +9 -0
  247. package/vendor-core/docs-content/detection/cold.md +4 -0
  248. package/vendor-core/docs-content/detection/cstor.md +4 -0
  249. package/vendor-core/docs-content/detection/fail.md +5 -0
  250. package/vendor-core/docs-content/detection/gc.md +4 -0
  251. package/vendor-core/docs-content/detection/host.md +5 -0
  252. package/vendor-core/docs-content/detection/incmp.md +6 -0
  253. package/vendor-core/docs-content/detection/jobs.md +4 -0
  254. package/vendor-core/docs-content/detection/local.md +6 -0
  255. package/vendor-core/docs-content/detection/mem.md +10 -0
  256. package/vendor-core/docs-content/detection/part.md +5 -0
  257. package/vendor-core/docs-content/detection/plan.md +14 -0
  258. package/vendor-core/docs-content/detection/retry.md +4 -0
  259. package/vendor-core/docs-content/detection/sfail.md +5 -0
  260. package/vendor-core/docs-content/detection/shape.md +5 -0
  261. package/vendor-core/docs-content/detection/shfl.md +4 -0
  262. package/vendor-core/docs-content/detection/skew.md +6 -0
  263. package/vendor-core/docs-content/detection/slow.md +6 -0
  264. package/vendor-core/docs-content/detection/spec.md +8 -0
  265. package/vendor-core/docs-content/detection/spill.md +7 -0
  266. package/vendor-core/docs-content/detection/strag.md +5 -0
  267. package/vendor-core/docs-content/detection/tiny.md +4 -0
  268. package/vendor-core/docs-content/detection/util.md +4 -0
  269. package/vendor-core/docs-content/diagrams/aqe-loop.dark.svg +1 -0
  270. package/vendor-core/docs-content/diagrams/aqe-loop.svg +1 -0
  271. package/vendor-core/docs-content/diagrams/broadcast-vs-shuffle.dark.svg +1 -0
  272. package/vendor-core/docs-content/diagrams/broadcast-vs-shuffle.svg +1 -0
  273. package/vendor-core/docs-content/diagrams/cache-lifecycle.dark.svg +1 -0
  274. package/vendor-core/docs-content/diagrams/cache-lifecycle.svg +1 -0
  275. package/vendor-core/docs-content/diagrams/cold-start-timeline.dark.svg +1 -0
  276. package/vendor-core/docs-content/diagrams/cold-start-timeline.svg +1 -0
  277. package/vendor-core/docs-content/diagrams/columnar-layout.dark.svg +1 -0
  278. package/vendor-core/docs-content/diagrams/columnar-layout.svg +1 -0
  279. package/vendor-core/docs-content/diagrams/container-memory.dark.svg +1 -0
  280. package/vendor-core/docs-content/diagrams/container-memory.svg +1 -0
  281. package/vendor-core/docs-content/diagrams/dag-stages.dark.svg +1 -0
  282. package/vendor-core/docs-content/diagrams/dag-stages.svg +1 -0
  283. package/vendor-core/docs-content/diagrams/driver-executor.dark.svg +1 -0
  284. package/vendor-core/docs-content/diagrams/driver-executor.svg +1 -0
  285. package/vendor-core/docs-content/diagrams/duplicate-plan-subtree.dark.svg +1 -0
  286. package/vendor-core/docs-content/diagrams/duplicate-plan-subtree.svg +1 -0
  287. package/vendor-core/docs-content/diagrams/join-strategy.dark.svg +1 -0
  288. package/vendor-core/docs-content/diagrams/join-strategy.svg +1 -0
  289. package/vendor-core/docs-content/diagrams/memory-borrowing.dark.svg +1 -0
  290. package/vendor-core/docs-content/diagrams/memory-borrowing.svg +1 -0
  291. package/vendor-core/docs-content/diagrams/memory-regions.dark.svg +1 -0
  292. package/vendor-core/docs-content/diagrams/memory-regions.svg +1 -0
  293. package/vendor-core/docs-content/diagrams/repartition-vs-coalesce.dark.svg +1 -0
  294. package/vendor-core/docs-content/diagrams/repartition-vs-coalesce.svg +1 -0
  295. package/vendor-core/docs-content/diagrams/retry-escalation-ladder.dark.svg +1 -0
  296. package/vendor-core/docs-content/diagrams/retry-escalation-ladder.svg +1 -0
  297. package/vendor-core/docs-content/diagrams/shuffle-map-reduce.dark.svg +1 -0
  298. package/vendor-core/docs-content/diagrams/shuffle-map-reduce.svg +1 -0
  299. package/vendor-core/docs-content/diagrams/spill-classification.dark.svg +1 -0
  300. package/vendor-core/docs-content/diagrams/spill-classification.svg +1 -0
  301. package/vendor-core/docs-content/diagrams/udf-execution-models.dark.svg +1 -0
  302. package/vendor-core/docs-content/diagrams/udf-execution-models.svg +1 -0
  303. package/vendor-core/docs-content/tuning/broadcast-sizing.md +78 -0
  304. package/vendor-core/docs-content/tuning/cold-start.md +81 -0
  305. package/vendor-core/docs-content/tuning/duplicate-plan-subtree.md +45 -0
  306. package/vendor-core/docs-content/tuning/failures.md +124 -0
  307. package/vendor-core/docs-content/tuning/gc.md +110 -0
  308. package/vendor-core/docs-content/tuning/job-failure-rate.md +101 -0
  309. package/vendor-core/docs-content/tuning/memory-utilization.md +58 -0
  310. package/vendor-core/docs-content/tuning/retry-waste.md +90 -0
  311. package/vendor-core/docs-content/tuning/shuffle.md +154 -0
  312. package/vendor-core/docs-content/tuning/skew.md +123 -0
  313. package/vendor-core/docs-content/tuning/slow-host.md +117 -0
  314. package/vendor-core/docs-content/tuning/small-files.md +99 -0
  315. package/vendor-core/docs-content/tuning/spill.md +114 -0
  316. package/vendor-core/docs-content/tuning/straggler.md +103 -0
  317. package/vendor-core/docs-content/tuning/tiny-tasks.md +94 -0
  318. package/vendor-core/docs-content/tuning/utilization.md +90 -0
  319. package/vendor-core/docs-site-config.js +10 -17
  320. package/vendor-core/efficiency-model.js +7 -13
  321. package/vendor-core/etl-phases.js +3 -5
  322. package/vendor-core/event-handlers.js +232 -134
  323. package/vendor-core/event-schemas.js +48 -114
  324. package/vendor-core/evidence-availability.js +5 -10
  325. package/vendor-core/evidence-report.js +73 -123
  326. package/vendor-core/export-data.js +48 -0
  327. package/vendor-core/finding-action-label.js +4 -10
  328. package/vendor-core/finding-filter-predicate.js +3 -7
  329. package/vendor-core/finding-generic-recommendation.js +112 -0
  330. package/vendor-core/finding-names.js +51 -0
  331. package/vendor-core/format-utils.js +112 -38
  332. package/vendor-core/impact-band.js +18 -24
  333. package/vendor-core/impact-estimator.js +38 -74
  334. package/vendor-core/ingest.js +7 -13
  335. package/vendor-core/job-groups.js +3 -6
  336. package/vendor-core/list-runs.js +278 -0
  337. package/vendor-core/load-vendored.js +6 -12
  338. package/vendor-core/log-header-peek.js +81 -0
  339. package/vendor-core/lz4-block.js +4 -6
  340. package/vendor-core/mcp-server-factory.js +38 -8
  341. package/vendor-core/mcp-tools.js +105 -76
  342. package/vendor-core/model-assembler.js +8 -16
  343. package/vendor-core/occupancy.js +5 -9
  344. package/vendor-core/parser-worker.js +20 -29
  345. package/vendor-core/plan-dot.js +2 -5
  346. package/vendor-core/plan-duration-attribution.js +78 -29
  347. package/vendor-core/plan-graph-model.js +126 -69
  348. package/vendor-core/plan-node-detail.js +31 -17
  349. package/vendor-core/plan-summary.js +19 -8
  350. package/vendor-core/recommendation-rollup.js +35 -39
  351. package/vendor-core/redact.js +72 -16
  352. package/vendor-core/rolling-log-reassembly.js +4 -6
  353. package/vendor-core/run-comparison.js +86 -72
  354. package/vendor-core/scaling-sim.js +5 -7
  355. package/vendor-core/session-snapshot.js +1 -1
  356. package/vendor-core/shs-fetch.js +4 -6
  357. package/vendor-core/shs-load.js +9 -13
  358. package/vendor-core/shs-request.js +1 -1
  359. package/vendor-core/stage-quantiles.js +14 -0
  360. package/vendor-core/types.js +78 -18
  361. package/vendor-core/wasted-core-hours.js +7 -12
@@ -1,33 +1,23 @@
1
1
  import { pathBasename, formatBytes, nsToMs, IMPACT_BAND_ORDER } from './format-utils.js';
2
2
  import { scanRelationId } from './plan-summary.js';
3
- import { computeTotalCores } from './core-count.js';
3
+ import { computePeakConcurrentCores, computePeakConcurrentExecutorCount } from './core-count.js';
4
4
  import { walkPlanTree } from './plan-tree-walk.js';
5
5
  import { computeCoreLocalityRatio } from './core-locality-ratio.js';
6
6
  import { estimateSingleStage, } from './occupancy.js';
7
+ import { isExchangeNode, isBroadcastExchangeNode } from './plan-node-detail.js';
7
8
 
8
9
 
9
10
  const MB = 1024 * 1024;
10
11
  const GB = 1024 * MB;
11
12
  const TB = 1024 * GB;
12
13
 
13
- // ---------------------------------------------------------------------------
14
14
  // Local runtime shapes.
15
15
  //
16
- // `types.ts`'s `Stage`/`SqlExecution`/`SparkAppInfo`/`AppModel` describe the
17
- // *posted* AppModel surface at a level of detail that suits the view layer
18
- // (both carry a catch-all `[key: string]: unknown` index signature for
19
- // anything beyond their few named fields). This file needs the FULL set of
20
- // fields `finalizeStage` (src/stage-quantiles.ts) actually computes and
21
- // `event-handlers.ts`'s stage/sql/app records actually carry, accessed
22
- // directly (arithmetic, comparisons) rather than read-and-display, so an
23
- // index-signature-shaped type would force an `unknown` cast at nearly every
24
- // field access. Every field below is verified against a real access in this
25
- // file (or a helper it calls); none are speculative.
26
- //
27
- // `analyzer.js` (the only real caller of `detect()`, still plain JS) passes
28
- // whatever the parser actually produced, so these types describe reality,
29
- // not a narrowing of some existing stricter type; there is nothing unsound
30
- // about them being independent of `types.ts`'s `Stage`/`SqlExecution`.
16
+ // types.ts's Stage/SqlExecution/SparkAppInfo describe the posted AppModel surface for the view
17
+ // layer (with a catch-all index signature). This file needs the FULL set of fields finalizeStage
18
+ // computes and event-handlers.ts records carry, accessed directly (arithmetic, comparisons), so
19
+ // an index-signature type would force an `unknown` cast at nearly every access. Every field below
20
+ // is verified against a real access here.
31
21
 
32
22
 
33
23
 
@@ -35,9 +25,18 @@ const TB = 1024 * GB;
35
25
 
36
26
 
37
27
 
38
- // Per-executor snapshot from a StageExecutorMetrics event (mergeStageRddInfo/
39
- // onStageExecutorMetrics): a loose bag of Spark's ExecutorMetrics field
40
- // names, only a few of which any detector reads.
28
+
29
+
30
+
31
+
32
+
33
+
34
+
35
+
36
+
37
+
38
+ // Per-executor snapshot from a StageExecutorMetrics event: a loose bag of Spark's
39
+ // ExecutorMetrics field names, only a few of which any detector reads.
41
40
 
42
41
 
43
42
 
@@ -81,6 +80,8 @@ const TB = 1024 * GB;
81
80
 
82
81
 
83
82
 
83
+
84
+
84
85
 
85
86
 
86
87
 
@@ -116,27 +117,19 @@ const TB = 1024 * GB;
116
117
 
117
118
 
118
119
 
119
-
120
-
120
+
121
121
 
122
122
 
123
123
 
124
124
 
125
- // The full context object analyzer.js's `analyze()` builds and passes as
126
- // every stage/sql detect()'s second argument, and as the sole argument for
127
- // 'app' scope (`d.detect(ctx)`). 'config' scope gets a narrower `{ app }`
128
- // (see auditConfig in analyzer.js), typed per-entry below instead of here.
125
+ // The full context analyze() passes as every stage/sql detect()'s second arg, and as the sole
126
+ // arg for 'app' scope. 'config' scope gets a narrower `{ app }` (auditConfig), typed per-entry.
129
127
  //
130
- // `app` is non-nullable here (unlike `AppModel.app: SparkAppInfo | null`):
131
- // `analyze()` only ever runs after a full parse, by which point `app` is
132
- // always populated; the `SparkAppInfo | null` nullability models the
133
- // in-progress-parse window this file never observes. This matches every
134
- // 'app'-scope entry below: most defensively write `ctx.app?.foo` anyway
135
- // (harmless on a non-nullable value), and `coldStart` reads `app.startTime`
136
- // with no guard at all, which only type-checks if `app` is non-nullable.
137
- // `auditConfig`'s separate config-scope target type keeps `app` nullable
138
- // instead, since `auditConfig(appModel.app)` (src/analyzer.js) really can be
139
- // called with a `null` app and every config-scope entry optional-chains it.
128
+ // `app` is typed as non-nullable here (unlike AppModel.app), but incompleteRun, coldStart,
129
+ // utilization, and autoscalingChurn defensively guard against null at runtime to tolerate
130
+ // malformed/incomplete logs. Despite the type annotation, app may be null in edge cases, and
131
+ // these detectors handle it gracefully. auditConfig's config-scope target keeps `app` nullable
132
+ // instead.
140
133
 
141
134
 
142
135
 
@@ -145,17 +138,13 @@ const TB = 1024 * GB;
145
138
 
146
139
 
147
140
 
148
-
149
-
150
-
151
-
152
-
141
+
142
+
153
143
 
154
144
 
155
145
 
156
- // `auditConfig(app)` (src/analyzer.js) calls every 'config'-scope entry's
157
- // `detect({ app })` directly with whatever `appModel.app` is at call time
158
- // (`SparkAppInfo | null`), independent of the full `DetectorCtx` above.
146
+ // auditConfig(app) calls every config-scope detect({ app }) with whatever appModel.app is
147
+ // (SparkAppInfo | null), independent of DetectorCtx.
159
148
 
160
149
 
161
150
 
@@ -167,8 +156,7 @@ const TB = 1024 * GB;
167
156
 
168
157
 
169
158
 
170
- // sparkDoctor SpillPressureDetector (5a) + SpillSkewDetector (5b).
171
- // Returns { magnitude } | null. Magnitude ∈ 'severe'|'high'|'medium'.
159
+ // SpillPressureDetector (5a) + SpillSkewDetector (5b).
172
160
  function computeSpillMagnitude(
173
161
  stage ,
174
162
  t ,
@@ -195,15 +183,9 @@ function pickDominantReason(reasons )
195
183
  return [...reasons].sort((a, b) => b.count - a.count)[0].reason;
196
184
  }
197
185
 
198
- // Shared by every scope:'sql' detector below. `sql.get(executionId).stageIds`
199
- // (src/model-assembler.js) is always empty because parser-worker.js never
200
- // populates it, so stage linkage is derived the other way round, from each
201
- // stage's own (correctly populated) `sqlExecutionId`. Parameter type is
202
- // intentionally the minimal shape needed (not `DetectorStage`): callers
203
- // outside this file (Topbar.tsx, plan-node-detail.ts, plan-graph-model.ts)
204
- // pass `appModel.stages`, typed `Map<StageId, Stage>` per types.ts, which
205
- // carries `id`/`sqlExecutionId` but not this file's fuller `DetectorStage`
206
- // shape.
186
+ // Shared by every scope:'sql' detector. sql.get(id).stageIds is always empty (parser-worker
187
+ // never populates it), so stage linkage is derived from each stage's own sqlExecutionId.
188
+ // Minimal param shape (not DetectorStage): external callers pass Map<StageId, Stage>.
207
189
  export function stageIdsForSqlExec(
208
190
  executionId ,
209
191
  stages ,
@@ -213,38 +195,23 @@ export function stageIdsForSqlExec(
213
195
  return out;
214
196
  }
215
197
 
216
- // Shared by the three Plan Advisor detectors below (duplicatePlanSubtree,
217
- // smallFiles, broadcastSizing): union the given nodes' own `stageIds`, or
218
- // fall back to the whole execution's stage set when none of them have any
219
- // coverage. A finding never partially blends a narrowed set with the
220
- // execution-wide one (see docs/architecture.md's plan-node-to-stage mapping
221
- // section). `fallback` is a param, not computed here, so callers can compute
222
- // `stageIdsForSqlExec` once per `detect()` call and reuse it across every
223
- // finding in that call instead of re-walking `stages` per finding.
198
+ // Shared by the three Plan Advisor detectors: union the given nodes' own stageIds, or fall back
199
+ // to the whole execution's stage set when none have coverage (never a partial blend). `fallback`
200
+ // is a param so callers compute stageIdsForSqlExec once per detect() and reuse it per finding.
224
201
  export function unionStageIds(nodes , fallback ) {
225
202
  const union = new Set ();
226
203
  for (const node of nodes) for (const sid of node.stageIds ?? []) union.add(sid);
227
204
  return union.size > 0 ? [...union].sort((a, b) => a - b) : fallback;
228
205
  }
229
206
 
230
- // Bottom-up shape computation for duplicate-subtree detection (sparkDoctor
231
- // idea, src/detectors.js `duplicatePlanSubtree`) and for the cachingOpportunity
232
- // composite (join/union) detector's anchor-fingerprint contract. `size` is the
233
- // node count of the subtree rooted at each node; the default fingerprint
234
- // encodes operator name + sorted metric *names* (never values, per spec) +
235
- // children fingerprints in original order, so two subtrees with the same shape
236
- // but different row counts/literals still collide, which is the point (we
237
- // have no expr/plan/codegen IDs to strip in the first place, since `metrics`
238
- // never carried them).
207
+ // Bottom-up shape computation for duplicate-subtree detection and the
208
+ // cachingOpportunity composite detector. `size` is the subtree node count; the default
209
+ // fingerprint encodes operator name + sorted metric NAMES (never values, per spec) + child
210
+ // fingerprints, so two subtrees with the same shape but different values still collide.
239
211
  //
240
- // `opts.includeDetail` (default false) folds `root`'s own `detail` text
241
- // (normalized via `opts.normalizeDetail`) into ONLY `root`'s fingerprint,
242
- // never into any recursive child call's fingerprint, regardless of `opts`.
243
- // This is what lets a caller ask "does this specific node's own detail +
244
- // structural shape match another node's", without descendant filter/scan
245
- // detail (literals, paths) ever entering the comparison. The existing
246
- // `duplicatePlanSubtree` call site passes no options: identical behavior,
247
- // zero regression to its existing tests.
212
+ // opts.includeDetail (default false) folds root's own detail (via opts.normalizeDetail) into
213
+ // ONLY root's fingerprint, never a child's: lets a caller match "this node's own detail + shape"
214
+ // without descendant scan detail entering the comparison.
248
215
 
249
216
 
250
217
  export function computePlanShapes(
@@ -256,9 +223,22 @@ export function computePlanShapes(
256
223
  const allNodes = [];
257
224
  function visit(node , isRoot ) {
258
225
  allNodes.push(node);
259
- const childShapes = (node.children ?? []).map((c) => visit(c, false));
226
+ // A read half wrapping a write half (the Exchange split from
227
+ // resolvePlanTree, see event-handlers.ts) is one logical Spark operator
228
+ // for subtree-shape purposes. Without this, every real Exchange in a
229
+ // matched subtree would count twice, inflating duplicatePlanSubtree's
230
+ // reported subtreeSize and shifting its groupIndex-derived findingIds
231
+ // for an otherwise-unchanged plan. Skip straight through the write
232
+ // wrapper: size comes from its real children, and metricNames from its
233
+ // real metrics (the read half's own metrics are always empty), so
234
+ // fingerprint distinctiveness between different real Exchanges is
235
+ // preserved too.
236
+ const writeHalf = node.exchangeRole === 'read' ? node.children[0] : null;
237
+ const realChildren = writeHalf ? writeHalf.children : (node.children ?? []);
238
+ const realMetrics = writeHalf ? writeHalf.metrics : node.metrics;
239
+ const childShapes = realChildren.map((c) => visit(c, false));
260
240
  const size = 1 + childShapes.reduce((sum, c) => sum + c.size, 0);
261
- const metricNames = (node.metrics ?? []).map((m) => m.name).sort().join(',');
241
+ const metricNames = (realMetrics ?? []).map((m) => m.name).sort().join(',');
262
242
  const childFingerprints = childShapes.map((c) => c.fingerprint).join(',');
263
243
  const fingerprint = isRoot && includeDetail
264
244
  ? `${node.name}[${metricNames}]<${normalize(node.detail ?? '')}>{${childFingerprints}}`
@@ -271,16 +251,10 @@ export function computePlanShapes(
271
251
  return { shapeOf, allNodes };
272
252
  }
273
253
 
274
- // Normalizes an anchor join/union node's own `detail` text for the
275
- // cachingOpportunity composite detector's structural fingerprint
276
- // (computePlanShapes's opts.includeDetail path). Strips per-analysis
277
- // numbering noise that is never part of a join/union's logical identity:
278
- // expr ids, plan/codegen-stage ids, and AQE's runtime BuildLeft/BuildRight
279
- // broadcast-side choice (which can flip between executions of the logically
280
- // identical join based on runtime stats); it then canonicalizes commutative
281
- // equality operand order so `A.x = B.y` and `B.y = A.x` collide. Everything
282
- // else (join type, columns, literal values) is kept: that's the
283
- // semantically meaningful part a leaf-relation-set identity would miss.
254
+ // Normalizes an anchor join/union node's detail for the cachingOpportunity composite
255
+ // fingerprint. Strips per-analysis numbering (expr ids, plan/codegen ids) and AQE's runtime
256
+ // BuildLeft/BuildRight choice (can flip between runs), then canonicalizes commutative equality
257
+ // operand order so `A.x = B.y` and `B.y = A.x` collide. Join type, columns, literals are kept.
284
258
  export function normalizeDetail(detail ) {
285
259
  let s = detail
286
260
  .replace(/#\d+L?/g, '')
@@ -296,34 +270,21 @@ export function normalizeDetail(detail ) {
296
270
 
297
271
  const JOIN_NAME_RE = /Join/i;
298
272
 
299
- // Structural operator kind for cachingOpportunity's composite detection:
300
- // 'join' covers every Spark join physical operator (SortMergeJoin,
301
- // BroadcastHashJoin, ShuffledHashJoin, BroadcastNestedLoopJoin, the same set
302
- // plan-summary.js's visitJoin recognizes); 'union' is Spark's exact `Union`
303
- // node name. CartesianProduct is deliberately excluded (out of scope, same
304
- // as plan-summary.js's join handling).
273
+ // Structural operator kind for cachingOpportunity's composite detection: 'join' covers every
274
+ // Spark join physical operator; 'union' is Spark's exact `Union` node. CartesianProduct is
275
+ // deliberately excluded (out of scope, as in plan-summary.ts).
305
276
  export function planOperatorKind(name ) {
306
277
  if (JOIN_NAME_RE.test(name)) return 'join';
307
278
  if (name === 'Union') return 'union';
308
279
  return null;
309
280
  }
310
281
 
311
- // Single bottom-up pass over a resolved planTree producing one composite
312
- // candidate per join/union node, in O(n) total (see the design doc's
313
- // "Compute cost" section; this deliberately does NOT call
314
- // computePlanShapes per join/union node, since subtrees overlap and that
315
- // would be O(n·k) on multi-way star/snowflake joins). For every node it
316
- // computes the plain (detail-free) fingerprint once (identical formula to
317
- // computePlanShapes's default path) and merges each subtree's leaf-relation
318
- // byte map (via scanRelationId, same identity cachingOpportunity's existing
319
- // leaf aggregation uses) bottom-up. When a node is a join/union, its anchor
320
- // fingerprint folds only its OWN normalized detail (per computePlanShapes's
321
- // opts.includeDetail contract), computed here inline, in O(1), from the
322
- // node's own detail plus its already-computed child fingerprints, not via a
323
- // nested computePlanShapes call. `ancestorNodes` (strict ancestors, root
324
- // first) is threaded down for free via the recursion's own call stack, so
325
- // later nested-composite dedupe (cachingOpportunity.detect()) doesn't need a
326
- // separate tree walk to determine containment.
282
+ // Single bottom-up O(n) pass producing one composite candidate per join/union node. Deliberately
283
+ // does NOT call computePlanShapes per node (subtrees overlap, that would be O(n·k) on
284
+ // star/snowflake joins): computes the plain fingerprint once and merges each subtree's
285
+ // leaf-relation byte map bottom-up. A join/union's anchor fingerprint folds only its OWN
286
+ // normalized detail, inline in O(1). `ancestorNodes` threads down via the call stack so
287
+ // nested-composite dedupe needs no separate tree walk.
327
288
 
328
289
 
329
290
 
@@ -374,17 +335,10 @@ export function findCompositeCandidates(root ) {
374
335
  }
375
336
 
376
337
 
377
- // First scanned table/relation identity found in a subtree (pre-order, per
378
- // walkPlanTree's contract), or null when the subtree touches no named
379
- // relation. Surfaced on duplicatePlanSubtree findings (as `sampleRelation`,
380
- // see findDuplicateSubtrees below) so a user can tell apart two same-shaped
381
- // duplicate groups that scan different tables (real-log bug: two unrelated
382
- // "BroadcastExchange over a Project/Filter/Scan" patterns, one per dimension
383
- // table, produced byte-identical findings). This is best-effort/informational
384
- // only, not a uniqueness guarantee: it's null for scan-less subtrees (JDBC/
385
- // Kafka/LocalRelation) and can coincide when two groups share their first-
386
- // encountered leaf. See findDuplicateSubtrees's groupIndex for the actual
387
- // discriminator findingId() relies on.
338
+ // First scanned relation identity in a subtree (pre-order), or null when none. Surfaced on
339
+ // duplicatePlanSubtree findings as `sampleRelation` so a user can tell apart same-shaped groups
340
+ // that scan different tables. Best-effort only: null for scan-less subtrees, can coincide across
341
+ // groups; findDuplicateSubtrees's groupIndex is the actual discriminator findingId relies on.
388
342
  function firstLeafRelationId(node ) {
389
343
  let found = null;
390
344
  walkPlanTree(node, (n) => {
@@ -394,14 +348,10 @@ function firstLeafRelationId(node ) {
394
348
  return found;
395
349
  }
396
350
 
397
- // Groups nodes by fingerprint, keeping only groups of size >= minOccurrences
398
- // whose shared subtree size is >= minSubtreeSize. De-overlap: processes
399
- // candidate groups largest-subtree-first, and once a group is accepted every
400
- // node inside each of its matched occurrences is marked "claimed" so a
401
- // smaller, fully-nested duplicate group inside an already-accepted match is
402
- // dropped (a 5-node duplicate should not also emit findings for its 3-node
403
- // sub-subtrees); occurrences of a smaller group that fall OUTSIDE any
404
- // accepted larger match still count normally.
351
+ // Groups nodes by fingerprint, keeping groups of size >= minOccurrences whose subtree size is
352
+ // >= minSubtreeSize. De-overlap: process largest-subtree-first, and once a group is accepted mark
353
+ // every node in its matches "claimed" so a smaller fully-nested duplicate is dropped; occurrences
354
+ // of a smaller group OUTSIDE any accepted match still count.
405
355
 
406
356
 
407
357
 
@@ -417,7 +367,12 @@ export function findDuplicateSubtrees(
417
367
  { minSubtreeSize, minOccurrences } ,
418
368
  ) {
419
369
  const { shapeOf, allNodes } = computePlanShapes(root);
420
- const eligible = allNodes.filter(n => shapeOf.get(n) .size >= minSubtreeSize);
370
+ // Defensive, not load-bearing: computePlanShapes's visit() already skips
371
+ // straight through a read node to its write half's real children, so no
372
+ // write-half node is ever pushed into allNodes in the first place, this
373
+ // filter can structurally never exclude anything. Kept in case that
374
+ // invariant ever changes upstream.
375
+ const eligible = allNodes.filter(n => shapeOf.get(n) .size >= minSubtreeSize && n.exchangeRole !== 'write');
421
376
 
422
377
  const groups = new Map ();
423
378
  for (const n of eligible) {
@@ -445,14 +400,10 @@ export function findDuplicateSubtrees(
445
400
  rootName: unclaimed[0].name,
446
401
  subtreeSize: shapeOf.get(unclaimed[0]) .size,
447
402
  occurrences: unclaimed.length,
448
- isExchangeRoot: /Exchange/i.test(unclaimed[0].name),
403
+ isExchangeRoot: isExchangeNode(unclaimed[0]),
449
404
  sampleRelation: firstLeafRelationId(unclaimed[0]),
450
- // Deterministic position within this execution's group list. `sampleRelation`
451
- // is best-effort (null when the subtree touches no named catalog relation, e.g.
452
- // JDBC/Kafka/LocalRelation sources, or identical when two groups happen to share
453
- // their first-encountered leaf) and is NOT sufficient on its own to guarantee two
454
- // structurally-distinct groups get distinct finding ids; groupIndex is the actual
455
- // uniqueness guarantee findingId() relies on.
405
+ // Deterministic position within this execution's group list: the actual uniqueness
406
+ // guarantee findingId relies on, since sampleRelation is best-effort (null or coincident).
456
407
  groupIndex: results.length,
457
408
  nodes: unclaimed,
458
409
  });
@@ -460,21 +411,32 @@ export function findDuplicateSubtrees(
460
411
  return results;
461
412
  }
462
413
 
463
- // Exact metric names Spark emits: verified against a real SQLExecutionStart
464
- // event's sparkPlanInfo in examples/big-application_1777489669889_51251_1.
465
- // NOTE: the write-side byte metric is "written output", not "size of written
466
- // files" as an earlier draft of this detector's spec assumed.
414
+ // Fingerprint matching compares operator + metric names only, not literal values or expr IDs
415
+ // (see the finding's validationRequired text), so a small pattern repeated the bare minimum
416
+ // number of times is the case most likely to be coincidental rather than real duplicated work.
417
+ // A bigger matched subtree, or more repeats, are each on their own strong corroborating evidence
418
+ // that the match is real: the odds of two semantically-different query branches producing an
419
+ // identical operator-name sequence shrink fast as the sequence grows or repeats.
420
+ function duplicateSubtreeConfidence(
421
+ subtreeSize ,
422
+ occurrences ,
423
+ thresholds ,
424
+ ) {
425
+ if (subtreeSize <= thresholds.minSubtreeSize && occurrences <= thresholds.minOccurrences) return 'low';
426
+ if (subtreeSize >= thresholds.minSubtreeSize * 2 || occurrences >= thresholds.minOccurrences + 2) return 'high';
427
+ return 'medium';
428
+ }
429
+
430
+ // Exact metric names Spark emits, verified against a real SQLExecutionStart's sparkPlanInfo.
431
+ // The write-side byte metric is "written output", not "size of written files".
467
432
  const FILES_READ_COUNT = 'number of files read';
468
433
  const FILES_READ_BYTES = 'size of files read';
469
434
  const FILES_WRITTEN_COUNT = 'number of written files';
470
435
  const FILES_WRITTEN_BYTES = 'written output';
471
436
 
472
- // Byte size of a join-side subtree for broadcast sizing (dataflint
473
- // JoinToBroadcastAlert). Stops descending the instant a node carries a
474
- // "data size" metric: that node's value already aggregates everything
475
- // beneath it (e.g. an Exchange's "data size" already reflects everything it
476
- // shuffled), so summing further down would double-count. Only recurses into
477
- // children when the current node carries no such metric.
437
+ // Byte size of a join-side subtree for broadcast sizing. Stops
438
+ // descending at a node with a "data size" metric: that value already aggregates everything
439
+ // beneath it, so summing further would double-count. Only recurses when no such metric.
478
440
  function sumBoundarySize(node ) {
479
441
  const m = (node.metrics ?? []).find(x => x.name === 'data size');
480
442
  if (m) return m.value;
@@ -483,10 +445,8 @@ function sumBoundarySize(node ) {
483
445
  return sum;
484
446
  }
485
447
 
486
- // Nodes that actually fed sumBoundarySize's total for this subtree: mirrors
487
- // its recursion exactly (same "data size" metric check, same stop condition)
488
- // so the implicated stageIds line up with the size that was actually
489
- // compared, however deep that turns out to live.
448
+ // Nodes that fed sumBoundarySize's total: mirrors its recursion exactly so implicated stageIds
449
+ // line up with the size actually compared.
490
450
  function boundarySizeContributors(node ) {
491
451
  const m = (node.metrics ?? []).find((x) => x.name === 'data size');
492
452
  if (m) return [node];
@@ -516,10 +476,8 @@ function maxMedianRatio(
516
476
 
517
477
 
518
478
 
519
- // Machine-readable detector metadata for the evidence report: type, version,
520
- // scope, thresholds, and doc anchor per entry (no `detect` closure). Lets a
521
- // portable report record exactly which detector + threshold set produced each
522
- // finding, so evidence stays reproducible as detectors evolve.
479
+ // Machine-readable detector metadata for the evidence report (no `detect` closure), so a
480
+ // portable report records which detector + thresholds produced each finding.
523
481
  export function detectorCatalog() {
524
482
  return DETECTORS.map((d) => ({
525
483
  type: d.type,
@@ -530,21 +488,11 @@ export function detectorCatalog() {
530
488
  }));
531
489
  }
532
490
 
533
- // True task-duration skew ratio for a stage: P95/median once there are enough
534
- // tasks to trust a P95 estimate, otherwise max/median. Returns null when the
535
- // stage has no measurable median (p50 === 0). Exported so the CLI budget gate
536
- // (src/cli/budgets.ts) can recompute the same ratio the detector uses, rather
537
- // than reading the detector's own findings (which are floored at ratioWarn).
538
- //
539
- // Fields widened to optional: the `skew` detector below only ever calls this
540
- // with an already-finalized `DetectorStage` (all four always numeric by the
541
- // time `analyze()` runs), but `cli/budgets.ts`'s `checkSkew` calls it
542
- // directly against `AppModel.stages`: real `Stage` records for a stage that
543
- // never received a `StageCompleted` event (e.g. an unfinished run) genuinely
544
- // lack these fields (see `types.ts`'s `Stage`). The `as number` casts below
545
- // preserve the original behavior byte-for-byte: dividing through an absent
546
- // field still naturally produces `NaN` (as it always has for untyped JS
547
- // callers), rather than adding a new guard that would change the result.
491
+ // True task-duration skew ratio: P95/median once enough tasks to trust P95, else max/median.
492
+ // Null when no measurable median (p50 === 0). Exported so cli/budgets.ts recomputes the same
493
+ // ratio rather than reading findings floored at ratioWarn.
494
+ // Fields optional: budgets.ts calls this against raw AppModel.stages, whose stages may lack
495
+ // these fields (unfinished run). The `as number` casts keep behavior: an absent field yields NaN.
548
496
  export function computeSkewRatio(
549
497
  stage ,
550
498
  minTasksForP95 ,
@@ -556,18 +504,10 @@ export function computeSkewRatio(
556
504
  : { ratio: (max ) / (p50 ), metric: 'max/median' };
557
505
  }
558
506
 
559
- // Absolute-magnitude floor expressed as a % of total app runtime rather than
560
- // a fixed ms constant (mirrors computeSpillMagnitude's/slowHost's ratio+floor
561
- // pattern above/below): a skew/straggler/taskStageSkew ratio computed on a
562
- // handful of milliseconds is noise in a run that took hours, but the same
563
- // absolute waste is real in a run that only took seconds; a fixed-ms floor
564
- // can't scale between those. Used by the `skew` and `straggler` entries
565
- // below, each gated via `clippedWasteMs` (below) against the *same
566
- // occupancy-clipped* wall-clock figure
567
- // src/impact-estimator.ts displays as that finding's savings, not the raw
568
- // pre-clip delta, which can stay large after clipping collapses the
569
- // recoverable time to near zero (the stage's own longest task already
570
- // accounts for nearly all of its wall-clock window).
507
+ // Absolute-magnitude floor as a % of app runtime, not a fixed ms constant: a skew/straggler
508
+ // ratio on a few ms is noise in an hours-long run but real in a seconds-long one; a fixed-ms
509
+ // floor can't scale. Used by skew/straggler, gated via clippedWasteMs against the same
510
+ // occupancy-clipped figure impact-estimator.ts displays as savings.
571
511
  // NOT SOURCED: floor percentages are our own noise floor, unvalidated.
572
512
  function computeAppDurationMs(ctx ) {
573
513
  const app = ctx?.app;
@@ -576,37 +516,38 @@ function computeAppDurationMs(ctx ) {
576
516
  return durationMs > 0 ? durationMs : null;
577
517
  }
578
518
 
579
- // Unknown app timing (computeAppDurationMs returned null) never suppresses a
580
- // finding; it just skips the floor gate, preserving prior ratio-only
581
- // behavior when total runtime can't be computed.
519
+ // Unknown app timing never suppresses a finding; it just skips the floor gate.
582
520
  function meetsRuntimeFloor(wasteMs , appDurationMs , floorPct ) {
583
521
  return appDurationMs == null || wasteMs >= appDurationMs * floorPct;
584
522
  }
585
523
 
586
- // Runs a raw waste delta through the same ceiling/occupancy clip
587
- // src/impact-estimator.ts's singleStageImpact applies before display, so the
588
- // runtime floor above is checked against a stage's actual recoverable
589
- // wall-clock time rather than a delta that a physical floor (the stage's own
590
- // longest task) may leave almost entirely unrecoverable. Falls back to the
591
- // raw delta when occupancy data isn't available for this stage (ctx omitted,
592
- // or the stage was excluded from the occupancy sweep for having <= 0
593
- // duration), same as the pre-existing ratio-only behavior for unknown app
594
- // timing above.
524
+ // Runs a raw waste delta through the same occupancy clip impact-estimator.ts applies before
525
+ // display, so the runtime floor checks recoverable wall-clock, not a delta a physical floor
526
+ // leaves unrecoverable. Falls back to the raw delta when occupancy data is unavailable.
595
527
  function clippedWasteMs(wasteMs , stageId , ctx ) {
596
528
  if (!ctx) return wasteMs;
597
529
  const est = estimateSingleStage(wasteMs, stageId, ctx.stages , ctx.occupancy);
598
530
  return est ? est.wallClock.high : wasteMs;
599
531
  }
600
532
 
601
- // cacheUtilization's per-RDD copy (private to this file), following the
602
- // existing pattern of small per-detector formatting helpers (e.g.
603
- // pickDominantReason above). Both variants share the same confidence/
604
- // validationRequired text: the ratio is a point-in-time storage snapshot
605
- // from stage-submission events (src/event-handlers.js's mergeStageRddInfo),
606
- // not a runtime block-access read-count.
533
+ // Shared by cacheUtilization's two variants: the ratio is a point-in-time storage snapshot from
534
+ // stage-submission events, not a runtime block-access read-count.
607
535
  const CACHE_UTILIZATION_VALIDATION =
608
536
  "This ratio is a point-in-time storage snapshot from stage-submission events, not a runtime read-count. Confirm against the Spark UI's Storage tab before acting.";
609
537
 
538
+ // Both cachedRatio and diskRatio are percentages of an RDD's partition/byte count: with few
539
+ // partitions, one partition flipping cached/evicted (or disk-resident/memory-resident) swings the
540
+ // reported percentage by a large amount, so the point estimate is noisy. More partitions average
541
+ // that noise out into a stable ratio. numPartitions is the only sample-size signal
542
+ // DetectorRddInfo carries, so it drives confidence for both variants rather than a flat guess.
543
+ // 10/50 mirror this file's other "trust the sample" cutoffs (skew's minTasksForP95: 20,
544
+ // coreLocality's minTasks: 50).
545
+ function cacheSampleConfidence(numPartitions ) {
546
+ if (numPartitions < 10) return 'low';
547
+ if (numPartitions >= 50) return 'high';
548
+ return 'medium';
549
+ }
550
+
610
551
  function partialCacheFinding(rdd , cachedRatio , impactBand ) {
611
552
  const rddName = rdd.name || `RDD ${rdd.id}`;
612
553
  const cachedPct = Math.round(cachedRatio * 100);
@@ -615,7 +556,7 @@ function partialCacheFinding(rdd , cachedRatio , impactBa
615
556
  type: 'cacheUtilization', variant: 'partialCache', stageId: null,
616
557
  rddId: rdd.id, rddName,
617
558
  impactBand, metric: 'cachedRatio', value: cachedPct,
618
- confidence: 'medium',
559
+ confidence: cacheSampleConfidence(rdd.numPartitions),
619
560
  validationRequired: CACHE_UTILIZATION_VALIDATION,
620
561
  memorySize: rdd.memorySize, diskSize: rdd.diskSize,
621
562
  numCachedPartitions: rdd.numCachedPartitions, numPartitions: rdd.numPartitions,
@@ -630,7 +571,7 @@ function diskSpilloverFinding(rdd , diskRatio , impactBan
630
571
  type: 'cacheUtilization', variant: 'diskSpillover', stageId: null,
631
572
  rddId: rdd.id, rddName,
632
573
  impactBand, metric: 'diskRatio', value: diskPct,
633
- confidence: 'medium',
574
+ confidence: cacheSampleConfidence(rdd.numPartitions),
634
575
  validationRequired: CACHE_UTILIZATION_VALIDATION,
635
576
  memorySize: rdd.memorySize, diskSize: rdd.diskSize,
636
577
  numCachedPartitions: rdd.numCachedPartitions, numPartitions: rdd.numPartitions,
@@ -638,19 +579,11 @@ function diskSpilloverFinding(rdd , diskRatio , impactBan
638
579
  };
639
580
  }
640
581
 
641
- // Entry shape for every item in DETECTORS. `TTarget` stays `unknown` at the
642
- // array level (kept as the default, never instantiated per-scope): `detect`'s
643
- // real first-argument shape varies by `scope` (a `DetectorStage` for
644
- // 'stage', a `DetectorSqlExec` for 'sql', a `DetectorCtx` for 'app', or a
645
- // narrower `{ app }` for 'config'; see analyzer.js's `analyze`/`auditConfig`,
646
- // the only real callers), and unifying those four into one `TTarget` would
647
- // need either an unsound cast or a discriminated-union-of-detectors redesign
648
- // this migration task doesn't ask for. Each entry below still gets a
649
- // precisely-typed `detect` by annotating its own `target`/`ctx` parameters
650
- // directly: object-literal methods (this `detect(target) {}` shorthand, not
651
- // an arrow function assigned to a property) are checked bivariantly against
652
- // an interface's method parameter types, so a narrower, concrete annotation
653
- // here does not conflict with `Detector`'s `unknown` declaration.
582
+ // Entry shape for every DETECTORS item. TTarget stays `unknown` at the array level: detect's
583
+ // real first-arg varies by scope (DetectorStage/DetectorSqlExec/DetectorCtx/{ app }), and
584
+ // unifying them would need an unsound cast or a discriminated-union redesign. Each entry gets a
585
+ // precise detect by annotating its own params: object-literal method params are checked
586
+ // bivariantly, so a narrower annotation here doesn't conflict with the `unknown` declaration.
654
587
 
655
588
 
656
589
 
@@ -658,14 +591,11 @@ function diskSpilloverFinding(rdd , diskRatio , impactBan
658
591
 
659
592
 
660
593
 
661
-
662
-
663
-
664
-
665
-
594
+
666
595
 
667
596
 
668
597
 
598
+
669
599
 
670
600
 
671
601
 
@@ -678,12 +608,108 @@ function diskSpilloverFinding(rdd , diskRatio , impactBan
678
608
  // *Ratio = multiplicative factor
679
609
  // *Share/*Rate/*Util = 0–1 fraction (normalized)
680
610
 
681
- // `straggler`'s own noise-floor thresholds (NOT SOURCED: unvalidated),
682
- // exported so src/impact-band.ts can reuse the same figures as its global
683
- // impact-band floor instead of hand-copying the literals.
611
+ // straggler's noise-floor thresholds (NOT SOURCED: unvalidated), exported so impact-band.ts
612
+ // reuses the same figures instead of hand-copying.
684
613
  export const STRAGGLER_FLOOR_PCT_WARN = 0.005;
685
614
  export const STRAGGLER_FLOOR_PCT_CRIT = 0.02;
686
615
 
616
+ // A ratio just past ratioWarn is the case most likely to be ordinary task-duration variance
617
+ // rather than real skew; a ratio many multiples past it (a 50x P95/median vs. a 3.1x one) is
618
+ // unambiguous. 1.5x/5x mirror this file's other "just past the floor vs. clearly past it" splits
619
+ // (duplicateSubtreeConfidence's 2x, cacheSampleConfidence's 10/50-partition cutoffs).
620
+ function skewConfidence(ratio , ratioWarn ) {
621
+ if (ratio <= ratioWarn * 1.5) return 'low';
622
+ if (ratio >= ratioWarn * 5) return 'high';
623
+ return 'medium';
624
+ }
625
+
626
+ // warnPct100/lowInfoPct100 are this detector's own two thresholds; scale confidence as a multiple
627
+ // of whichever one gates the branch that fired, the same way skewConfidence scales off ratioWarn.
628
+ // High-GC: a pct just past warnPct100 (10%) is likely normal variance, 3x past it is unambiguous.
629
+ // Low-GC ("cost", over-provisioned): a pct just under lowInfoPct100 (5%) is borderline, a pct near
630
+ // zero is unambiguous idle GC.
631
+ function gcConfidence(
632
+ pct ,
633
+ thresholds ,
634
+ direction ,
635
+ ) {
636
+ if (direction === 'high') {
637
+ if (pct <= thresholds.warnPct100 * 1.5) return 'low';
638
+ if (pct >= thresholds.warnPct100 * 3) return 'high';
639
+ return 'medium';
640
+ }
641
+ if (pct >= thresholds.lowInfoPct100 * 0.66) return 'low';
642
+ if (pct <= thresholds.lowInfoPct100 * 0.2) return 'high';
643
+ return 'medium';
644
+ }
645
+
646
+ // warnFloor gates the finding, so a value just past it is the weakest evidence this detector can
647
+ // produce. highFloor is critPct: straggler's own second (speculative-share) tier, 2x warnPct
648
+ // (0.10 -> 0.20); reused as the high-confidence bar for the stragglerShare path too since that
649
+ // metric has no dedicated critical tier of its own (see the "no dedicated critical tier" comment
650
+ // on the straggler detector) but is the same 0-1 task-share magnitude.
651
+ function stragglerConfidence(shareValue , warnFloor , highFloor ) {
652
+ if (shareValue < warnFloor * 1.5) return 'low';
653
+ if (shareValue >= highFloor) return 'high';
654
+ return 'medium';
655
+ }
656
+
657
+ // minWasteMs is speculationWaste's own floor; a run that barely clears it (under 1.5x) is the
658
+ // weakest case, one that clears it several times over (4x, i.e. 4 minutes against a 1-minute
659
+ // floor) is unambiguous.
660
+ function speculationWasteConfidence(wastedMs , minWasteMs ) {
661
+ if (wastedMs <= minWasteMs * 1.5) return 'low';
662
+ if (wastedMs >= minWasteMs * 4) return 'high';
663
+ return 'medium';
664
+ }
665
+
666
+ // The finding already gates on wastedMBSeconds > wasteBufferMultiplier * usedMBSeconds, i.e. a
667
+ // ratio of 1 at the floor; scale confidence off that same ratio the way skewConfidence scales off
668
+ // ratioWarn, instead of introducing a second, unrelated multiplier.
669
+ function memoryWasteConfidence(wastedMBSeconds , usedMBSeconds , wasteBufferMultiplier ) {
670
+ const ratio = usedMBSeconds > 0 ? wastedMBSeconds / (wasteBufferMultiplier * usedMBSeconds) : Infinity;
671
+ if (ratio <= 1.5) return 'low';
672
+ if (ratio >= 3) return 'high';
673
+ return 'medium';
674
+ }
675
+
676
+ // Two independent weak spots can each undercut this finding: a ratio just past warnRatio (could
677
+ // be one bad stage), or too few sampled tasks (mirrors cacheSampleConfidence's use of
678
+ // numPartitions as a sample-size signal). Report whichever signal is weaker rather than
679
+ // averaging them away. critRatio is this detector's own existing second tier, reused directly as
680
+ // the ratio high-bar; minTasks*2/*4 mirror the same "2x floor is still weak, 4x is strong" spread
681
+ // used elsewhere in this file.
682
+ function coreLocalityConfidence(
683
+ ratio ,
684
+ totalTasks ,
685
+ thresholds ,
686
+ ) {
687
+ const rank = { low: 0, medium: 1, high: 2 } ;
688
+ const ratioTier = ratio < thresholds.warnRatio * 1.5 ? 'low' : ratio >= thresholds.critRatio ? 'high' : 'medium';
689
+ const sampleTier = totalTasks < thresholds.minTasks * 2 ? 'low' : totalTasks >= thresholds.minTasks * 4 ? 'high' : 'medium';
690
+ return rank[ratioTier] <= rank[sampleTier] ? ratioTier : sampleTier;
691
+ }
692
+
693
+ // warningPct/criticalPct are this detector's own two tiers (0.30/0.60); a churn rate just past
694
+ // warningPct is the borderline call the impact-band split already treats as the weaker tier, so
695
+ // reuse criticalPct directly as the high-confidence bar instead of inventing a third figure.
696
+ function autoscalingChurnConfidence(shortLivedPct , warningPct , criticalPct ) {
697
+ if (shortLivedPct <= warningPct * 1.5) return 'low';
698
+ if (shortLivedPct >= criticalPct) return 'high';
699
+ return 'medium';
700
+ }
701
+
702
+ // minExecutions is the bare minimum occurrence count this detector will even emit a finding for;
703
+ // a match at exactly that count is the weakest reuse signal (as likely to be coincidental overlap
704
+ // as real shared work), while 3x the floor is several independent executions all hitting the same
705
+ // relation/composite shape, unambiguous. Mirrors duplicateSubtreeConfidence's occurrences handling
706
+ // for the same reason: repetition count is the strength signal for a structural-match detector.
707
+ function cachingReuseConfidence(occurrences , minExecutions ) {
708
+ if (occurrences <= minExecutions) return 'low';
709
+ if (occurrences >= minExecutions * 3) return 'high';
710
+ return 'medium';
711
+ }
712
+
687
713
  export const DETECTORS = [
688
714
  {
689
715
  type: 'skew', scope: 'stage', order: 30, fixEffort: 'code', version: 1,
@@ -700,9 +726,8 @@ export const DETECTORS = [
700
726
  if (result === null) return null;
701
727
  const { ratio, metric } = result;
702
728
  if (ratio <= this.thresholds.ratioWarn) return null;
703
- // Same absolute delta src/impact-estimator.ts's 'skew' case reports as
704
- // this finding's savings; clipped the same way before the floor check
705
- // so the gate agrees with what's actually displayed.
729
+ // Same absolute delta impact-estimator.ts's 'skew' case reports as savings; clipped the
730
+ // same way before the floor check so the gate agrees with what's displayed.
706
731
  const wasteMs = Math.max(0, metric === 'P95/median' ? stage.taskDurationP95 - stage.taskDurationP50 : stage.taskDurationMax - stage.taskDurationP50);
707
732
  const appDurationMs = computeAppDurationMs(ctx);
708
733
  const floorWasteMs = clippedWasteMs(wasteMs, stage.id, ctx);
@@ -712,6 +737,8 @@ export const DETECTORS = [
712
737
  type: 'skew', stageId: stage.id,
713
738
  impactBand: 'warning',
714
739
  metric, value,
740
+ confidence: skewConfidence(ratio, this.thresholds.ratioWarn),
741
+ validationRequired: 'This finding is gated by a 0.5% runtime-floor threshold, our own noise floor for this metric.',
715
742
  recommendation: `Task duration ratio (${metric}) is ${value}×: for join-driven skew, enable AQE skew-join handling (spark.sql.adaptive.skewJoin.enabled); otherwise salt the key or repartition on a better key to reduce task skew.`,
716
743
  };
717
744
  },
@@ -753,11 +780,9 @@ export const DETECTORS = [
753
780
  });
754
781
  }
755
782
  }
756
- // TaskStageSkew: straggler cost vs stage wall-clock. Skip near-zero duration.
757
- // Always info, like its lowParallelism/dataExplosion siblings above: satisfying
758
- // this trigger mathematically forces the occupancy-clipped wall-clock estimate to
759
- // exactly zero on every firing (see impact-estimator.ts's costOnly branch below),
760
- // so there is no wall-clock-backed impact-band tier left to gate on.
783
+ // TaskStageSkew: straggler cost vs stage wall-clock. Skip near-zero duration. Always info
784
+ // like its siblings: this trigger forces the occupancy-clipped estimate to exactly zero on
785
+ // every firing, so there's no wall-clock-backed tier left to gate on.
761
786
  const stageDurationMs = (stage.completedAt ?? 0) - (stage.submittedAt ?? 0);
762
787
  if (stageDurationMs > 0) {
763
788
  const ratio = stage.taskDurationMax / stageDurationMs;
@@ -808,9 +833,8 @@ export const DETECTORS = [
808
833
  const out = [];
809
834
  const { shuffleReadP50: p50, shuffleReadMax: max, shuffleReadBytes: total, taskCount } = stage;
810
835
  if (max > this.thresholds.skewRatio * p50 && max > this.thresholds.skewFloorBytes) {
811
- // p50 can be 0 (more than half the shuffle partitions empty): a
812
- // ratio against zero renders the literal string "Infinity×", so fall
813
- // back to median-free phrasing instead of dividing by p50.
836
+ // p50 can be 0 (over half the shuffle partitions empty): a ratio against zero renders
837
+ // "Infinity×", so fall back to median-free phrasing.
814
838
  const ratioText = p50 > 0
815
839
  ? `${Math.round(max / p50 * 10) / 10}× the median (${formatBytes(p50)})`
816
840
  : `far larger than the median (${formatBytes(p50)}, effectively empty)`;
@@ -828,6 +852,10 @@ export const DETECTORS = [
828
852
  });
829
853
  }
830
854
  if (max >= this.thresholds.maxPartBytes) {
855
+ // Fixed 'critical': an OOM/crash-risk safety signal, not a time-waste one. Exempted in
856
+ // impact-band.ts's deriveImpactBand from the wall-clock-based overwrite every other
857
+ // finding here gets, so a long-running job can't demote an active crash risk to 'info'
858
+ // just because the modeled time savings are a small fraction of total runtime.
831
859
  out.push({
832
860
  type: 'partitionSizing', stageId: stage.id, impactBand: 'critical',
833
861
  rule: 'maxPartitionTooBig', metric: 'shuffleReadMax', value: max,
@@ -864,18 +892,18 @@ export const DETECTORS = [
864
892
  {
865
893
  type: 'gc', scope: 'stage', order: 50, fixEffort: 'config', version: 1,
866
894
  docAnchor: '#bottleneck-gc',
895
+ validationRequired: 'This finding is gated by a 10-second minimum-runtime floor, our own noise floor for this metric.',
867
896
  thresholds: {
868
897
  warnPct100: 10,
869
- // Descending tier: Dr. Elephant ExecutorGcHeuristic, ported as-is.
898
+ // Descending tier: ExecutorGcHeuristic, ported as-is.
870
899
  lowInfoPct100: 5,
871
- // NOT SOURCED: our own noise floor so a stage that barely ran (gcPct
872
- // near 0 or wildly inflated by a tiny denominator) does not flag,
873
- // in either direction.
900
+ // NOT SOURCED: our own noise floor so a stage that barely ran doesn't flag either direction.
874
901
  minRunTimeMs: 10000,
875
902
  },
876
903
  detect(
877
904
 
878
905
 
906
+
879
907
 
880
908
  stage ,
881
909
  ) {
@@ -887,6 +915,7 @@ export const DETECTORS = [
887
915
  type: 'gc', stageId: stage.id,
888
916
  impactBand: 'warning',
889
917
  metric: 'gcPct', value,
918
+ confidence: gcConfidence(pct, this.thresholds, 'high'), validationRequired: this.validationRequired,
890
919
  recommendation: `GC consumed ${value}% of executor run time: reduce object creation, use primitive types, avoid UDFs, increase executor memory.`,
891
920
  };
892
921
  }
@@ -898,6 +927,7 @@ export const DETECTORS = [
898
927
  type: 'gc', stageId: stage.id, direction: 'low',
899
928
  impactBand: 'info',
900
929
  metric: 'gcPct', value,
930
+ confidence: gcConfidence(pct, this.thresholds, 'low'), validationRequired: this.validationRequired,
901
931
  recommendation: `GC consumed only ${value}% of executor run time: memory may be over-provisioned; consider reducing spark.executor.memory for cost savings.`,
902
932
  };
903
933
  }
@@ -909,12 +939,9 @@ export const DETECTORS = [
909
939
  docAnchor: '#bottleneck-slow-host',
910
940
  thresholds: {
911
941
  minHosts: 3, minTasks: 15, ratioWarn: 2.0, minShare: 0.20, shareWarn: 0.75, taskShareWarn: 0.50, ratioTiers: [1.33, 1.78, 3.16, 10],
912
- // Absolute-magnitude floors (mirrors computeSpillMagnitude's ratio+floor
913
- // pattern): on short stages, sub-second/sub-64MB differences between
914
- // hosts or executors produce huge ratios that are pure noise, not a
915
- // real slow-host problem. 1s is well above typical per-task scheduling
916
- // jitter but well below the tens-of-seconds+ means genuine slow-host
917
- // stages exhibit; 64MB mirrors the spill detector's disk-skew floor.
942
+ // Absolute-magnitude floors (mirrors computeSpillMagnitude's ratio+floor pattern): on short
943
+ // stages, sub-second/sub-64MB host differences produce huge noise ratios. 1s is above
944
+ // per-task jitter but below genuine slow-host stages; 64MB mirrors the spill disk-skew floor.
918
945
  floorMs: 1000, floorBytes: 64 * MB,
919
946
  },
920
947
  detect(
@@ -946,7 +973,7 @@ export const DETECTORS = [
946
973
  // `value` is a ratio; the estimator needs the absolute per-host mean.
947
974
  hostMeanMs: h.mean,
948
975
  host: h.host, hostTaskShare: Math.round(share * 100) / 100,
949
- recommendation: `Check executor logs for ${h.host}: possible bad node, disk pressure, or NUMA misalignment.`,
976
+ recommendation: `${h.host} may just hold data locality for its tasks or carry one heavy stage, not necessarily a hardware fault: check what it was running, and consider enabling spark.speculation to relaunch a lagging task automatically.`,
950
977
  });
951
978
  }
952
979
  }
@@ -1012,10 +1039,8 @@ export const DETECTORS = [
1012
1039
  type: 'stageSlowness', scope: 'stage', order: 65, fixEffort: 'code', version: 2,
1013
1040
  docAnchor: '#bottleneck-stage-slowness',
1014
1041
  thresholds: { infoMin: 15 },
1015
- // Cross-detector suppression (new pattern; see "Detector contract" in
1016
- // docs-site/contributor-guide/architecture/detector-contract.md).
1017
- // Requires this entry to be declared AFTER slowHost in DETECTORS so
1018
- // slowHost findings are already in `out`.
1042
+ // Cross-detector suppression (see "Detector contract" in detector-contract.md). Requires this
1043
+ // entry to be declared AFTER slowHost in DETECTORS so slowHost findings are already in `out`.
1019
1044
  suppressWhen(finding, out) {
1020
1045
  return out.some(o => o.type === 'slowHost' && o.stageId === finding.stageId);
1021
1046
  },
@@ -1023,9 +1048,8 @@ export const DETECTORS = [
1023
1048
 
1024
1049
  stage ,
1025
1050
  ) {
1026
- // Basis is real wall-clock stage duration, not per-executor average
1027
- // (Decision 7); Task 11's impact-estimator formula reuses this exact
1028
- // stageDurationMs computation.
1051
+ // Basis is real wall-clock stage duration, not per-executor average; the impact-estimator
1052
+ // formula reuses this exact stageDurationMs computation.
1029
1053
  const stageDurationMs = (stage.completedAt ?? 0) - (stage.submittedAt ?? 0);
1030
1054
  if (!(stageDurationMs > 0)) return null;
1031
1055
  const durationMinutes = stageDurationMs / 60000;
@@ -1036,7 +1060,7 @@ export const DETECTORS = [
1036
1060
  return {
1037
1061
  type: 'stageSlowness', stageId: stage.id, impactBand,
1038
1062
  metric: 'stageDurationMinutes', value,
1039
- recommendation: `This stage ran ${value} minutes with no more specific cause flagged: profile its query plan and check for wide shuffles or expensive UDFs.`,
1063
+ recommendation: `This stage ran ${value} minutes with no more specific cause flagged: often a partition-count problem, raise parallelism via spark.sql.shuffle.partitions or spark.default.parallelism, or check for a large per-task data volume driving heavy shuffle and spill.`,
1040
1064
  };
1041
1065
  },
1042
1066
  },
@@ -1050,6 +1074,9 @@ export const DETECTORS = [
1050
1074
  type: 'stageFailed', stageId: stage.id, impactBand: 'critical',
1051
1075
  variant: 'stageFailure',
1052
1076
  metric: 'stageFailureReason', value: stage.stageFailureReason,
1077
+ numTasks: stage.taskCount,
1078
+ memoryBytesSpilled: stage.memoryBytesSpilled,
1079
+ failedTaskDetails: stage.failedTaskSamples ?? [],
1053
1080
  recommendation: `This stage attempt failed outright. Inspect the driver log for the failure reason and the job that triggered it.`,
1054
1081
  };
1055
1082
  },
@@ -1081,9 +1108,8 @@ export const DETECTORS = [
1081
1108
  {
1082
1109
  type: 'straggler', scope: 'stage', order: 70, fixEffort: 'code', version: 1,
1083
1110
  docAnchor: '#bottleneck-straggler',
1084
- // floorPctWarn/floorPctCrit are re-exported as STRAGGLER_FLOOR_PCT_WARN/CRIT
1085
- // below and reused as src/impact-band.ts's global noise floor: keep the two
1086
- // in sync, don't hand-edit one without the other.
1111
+ // floorPctWarn/floorPctCrit are re-exported as STRAGGLER_FLOOR_PCT_WARN/CRIT and reused as
1112
+ // impact-band.ts's global noise floor: keep the two in sync.
1087
1113
  thresholds: { minTasks: 10, shareWarn: 0.05, warnPct: 0.10, critPct: 0.20, floorPctWarn: STRAGGLER_FLOOR_PCT_WARN, floorPctCrit: STRAGGLER_FLOOR_PCT_CRIT },
1088
1114
  detect(
1089
1115
 
@@ -1097,12 +1123,9 @@ export const DETECTORS = [
1097
1123
  if ((stage.speculativeTasks ?? 0) === 0 && stragglerShare <= this.thresholds.shareWarn) return null;
1098
1124
  const useSpeculative = (stage.speculativeTasks ?? 0) > 0;
1099
1125
  const speculativeShare = useSpeculative ? stage.speculativeTasks / stage.taskCount : 0;
1100
- // Same absolute delta src/impact-estimator.ts's shared straggler/
1101
- // stageShape case reports as this finding's savings: a high share of
1102
- // stragglers/speculative retries on a stage whose tasks barely vary in
1103
- // duration models near-zero savings, so it must not outrank 'info'.
1104
- // Clipped the same way before the floor check so the gate agrees with
1105
- // what's actually displayed.
1126
+ // Same absolute delta impact-estimator.ts's straggler/stageShape case reports as savings: a
1127
+ // high straggler/speculative share on a stage whose tasks barely vary models near-zero
1128
+ // savings, so it must not outrank 'info'. Clipped the same way before the floor check.
1106
1129
  const wasteMs = Math.max(0, stage.taskDurationMax - stage.taskDurationP50);
1107
1130
  const appDurationMs = computeAppDurationMs(ctx);
1108
1131
  const floorWasteMs = clippedWasteMs(wasteMs, stage.id, ctx);
@@ -1110,21 +1133,14 @@ export const DETECTORS = [
1110
1133
  const meetsCritFloor = meetsRuntimeFloor(floorWasteMs, appDurationMs, this.thresholds.floorPctCrit);
1111
1134
  const speculativeTier = speculativeShare >= this.thresholds.critPct && meetsCritFloor ? 'critical'
1112
1135
  : speculativeShare >= this.thresholds.warnPct && meetsWarnFloor ? 'warning' : 'info';
1113
- // Straggler share has no dedicated critical tier per
1114
- // docs-site/contributor-guide/architecture/detector-contract.md; it can only push to warning.
1136
+ // Straggler share has no dedicated critical tier per detector-contract.md; only warning.
1115
1137
  const stragglerTier = stragglerShare > this.thresholds.shareWarn && meetsWarnFloor ? 'warning' : 'info';
1116
- // Fixed fallback: overwritten by deriveImpactBand (src/impact-band.ts) whenever
1117
- // this finding gets a real wallClock estimate, which is the common case. Only
1118
- // surfaces on the rare miss (stage excluded from the occupancy sweep).
1138
+ // Fixed fallback: overwritten by deriveImpactBand when this finding gets a real wallClock
1139
+ // estimate (the common case). Only surfaces on the rare occupancy-sweep miss.
1119
1140
  const impactBand = 'info';
1120
- // Report whichever signal actually drove the finding, not just whether
1121
- // speculative execution happened to be on: a high stragglerShare with
1122
- // few/no speculative retries must not be reported as a low-value
1123
- // speculativeTasks count (real-log bug: a 50%-straggler-share stage
1124
- // with 1 speculative task reported an impact band of 'warning' but metric
1125
- // 'speculativeTasks: 1', hiding the actual cause). Ties keep the prior
1126
- // default (speculative-driven) so existing speculative-only findings
1127
- // are unaffected.
1141
+ // Report whichever signal actually drove the finding, not just whether speculation was on:
1142
+ // a high stragglerShare with few speculative retries must not be reported as a low-value
1143
+ // speculativeTasks count. Ties keep the speculative-driven default.
1128
1144
  const useSpeculativeMetric = useSpeculative && !(IMPACT_BAND_ORDER[stragglerTier] < IMPACT_BAND_ORDER[speculativeTier]);
1129
1145
  const value = useSpeculativeMetric ? stage.speculativeTasks : Math.round(stragglerShare * 100);
1130
1146
  const detail = useSpeculativeMetric
@@ -1137,18 +1153,21 @@ export const DETECTORS = [
1137
1153
  unit: useSpeculativeMetric ? 'count' : 'pct',
1138
1154
  speculativeTasks: stage.speculativeTasks ?? 0,
1139
1155
  stragglerCount: stage.stragglerCount ?? 0,
1140
- recommendation: `${detail}: investigate stragglers, likely candidates for AQE skewJoin or data locality issues.`,
1156
+ confidence: useSpeculativeMetric
1157
+ ? stragglerConfidence(speculativeShare, this.thresholds.warnPct, this.thresholds.critPct)
1158
+ : stragglerConfidence(stragglerShare, this.thresholds.shareWarn, this.thresholds.critPct),
1159
+ validationRequired: 'This finding is gated by 0.5%/2% runtime-floor thresholds, our own noise floor for this metric.',
1160
+ recommendation: `${detail}: rule out a GC pause or a slow shuffle fetch before assuming a hardware issue; if a skewed key is the real cause, that's a candidate for AQE's skew-join handling.`,
1141
1161
  };
1142
1162
  },
1143
1163
  },
1144
1164
  {
1145
1165
  type: 'speculationWaste', scope: 'stage', order: 71, fixEffort: 'config', version: 1,
1146
- docAnchor: '#bottleneck-straggler', confidence: 'low',
1166
+ docAnchor: '#bottleneck-straggler',
1147
1167
  thresholds: { minWasted: 5, minWasteMs: 60000 },
1148
1168
  detect(
1149
1169
 
1150
1170
 
1151
-
1152
1171
 
1153
1172
  stage ,
1154
1173
  ) {
@@ -1159,7 +1178,7 @@ export const DETECTORS = [
1159
1178
  type: 'speculationWaste', stageId: stage.id,
1160
1179
  impactBand: 'warning',
1161
1180
  metric: 'speculationWasteMs', value: wastedMs,
1162
- confidence: this.confidence,
1181
+ confidence: speculationWasteConfidence(wastedMs, this.thresholds.minWasteMs),
1163
1182
  recommendation: `Speculative execution discarded ${Math.round(wastedMs / 1000)}s of executor time in this stage; if task durations are naturally variable rather than genuine stragglers, consider tuning spark.speculation.multiplier/quantile.`,
1164
1183
  };
1165
1184
  },
@@ -1179,6 +1198,9 @@ export const DETECTORS = [
1179
1198
  type: 'retryWaste', stageId: stage.id,
1180
1199
  impactBand: 'warning',
1181
1200
  metric: 'retryWasteMs', value: wastedMs,
1201
+ numTasks: stage.taskCount,
1202
+ memoryBytesSpilled: stage.memoryBytesSpilled,
1203
+ retriedTaskDetails: stage.retryTaskSamples ?? [],
1182
1204
  recommendation: `Retried task attempts wasted ${Math.round(wastedMs / 1000)}s of executor time (${wasted} attempt${wasted === 1 ? '' : 's'}) even though the stage completed: investigate executor loss or fetch failures.`,
1183
1205
  extended: `${wasted} task attempts were superseded by a later retry, wasting ${Math.round(wastedMs / 1000)}s of executor time. Common causes: executor loss (OOM-kill, node death) or shuffle FetchFailed forcing a stage-map recompute. Check driver logs for the dominant reason (see the Failures widget) even if the final failure rate looks low; retries hide the true cost.`,
1184
1206
  };
@@ -1206,14 +1228,13 @@ export const DETECTORS = [
1206
1228
  },
1207
1229
  },
1208
1230
  {
1209
- // No docAnchor set (documented deviation, like autoscalingChurn above):
1210
- // the upstream `shuffle-works/spark-tuning-reference` docs repo has no
1211
- // section for this tool-specific "capture stopped early" signal.
1231
+ // No docAnchor: the upstream spark-tuning-reference docs have no section for this
1232
+ // tool-specific "capture stopped early" signal.
1212
1233
  type: 'incompleteRun', scope: 'app', order: 5, fixEffort: 'code', version: 1,
1213
1234
  thresholds: {},
1214
1235
  recommendation: 'This event log never recorded an ApplicationEnd event: the capture stopped before the run finished (an in-flight job, a rotated-away log, or a cut-short capture). Findings and metrics elsewhere on this board reflect only what was captured up to that point, not the full run.',
1215
1236
  detect( ctx ) {
1216
- if (ctx.app.startTime == null || ctx.app.endTime != null) return null;
1237
+ if (!ctx.app || ctx.app.startTime == null || ctx.app.endTime != null) return null;
1217
1238
  return {
1218
1239
  type: 'incompleteRun', stageId: null, impactBand: 'warning',
1219
1240
  metric: 'applicationEnd', value: 'missing',
@@ -1230,18 +1251,22 @@ export const DETECTORS = [
1230
1251
  ctx ,
1231
1252
  ) {
1232
1253
  const { app, stages } = ctx;
1233
- if (!app.startTime || stages.size === 0) return null;
1254
+ // Nullish (not falsy) check: a literal startTime:0 must not be treated as "missing".
1255
+ if (!app || app.startTime == null || stages.size === 0) return null;
1234
1256
  let firstTaskLaunch = Infinity;
1235
1257
  for (const stage of stages.values()) {
1236
1258
  if (stage.submittedAt > 0 && stage.submittedAt < firstTaskLaunch) firstTaskLaunch = stage.submittedAt;
1237
1259
  }
1260
+ // No stage ever recorded a submission timestamp: no basis to measure a startup gap against.
1261
+ // Exposed now that a literal app.startTime:0 no longer short-circuits this detector entirely.
1262
+ if (!Number.isFinite(firstTaskLaunch)) return null;
1238
1263
  const gapSeconds = (firstTaskLaunch - app.startTime) / 1000;
1239
1264
  if (gapSeconds <= this.thresholds.gapSeconds) return null;
1240
1265
  const value = Math.round(gapSeconds);
1241
1266
  return {
1242
1267
  type: 'coldStart', stageId: null, impactBand: 'warning',
1243
1268
  metric: 'startupGapSeconds', value,
1244
- recommendation: `Executor startup took ${value}s: consider pre-warming the cluster or using dynamic allocation.`,
1269
+ recommendation: `The first task waited ${value}s for executors to become available: keep a warm pool of idle executors, or if using dynamic allocation, raise the minimum/initial executor count so it doesn't scale up from zero.`,
1245
1270
  };
1246
1271
  },
1247
1272
  },
@@ -1253,25 +1278,27 @@ export const DETECTORS = [
1253
1278
 
1254
1279
  ctx ,
1255
1280
  ) {
1256
- const { app, executorsAdded, executorsRemoved } = ctx;
1257
- if (executorsAdded.length === 0 || !app.startTime || !app.endTime) return null;
1281
+ const { app, executorsAdded, executorsRemoved, runAggregates } = ctx;
1282
+ // Nullish (not falsy) check: a literal startTime:0 must not be treated as "missing".
1283
+ if (!app || executorsAdded.length === 0 || app.startTime == null || app.endTime == null) return null;
1258
1284
  const appDuration = app.endTime - app.startTime;
1259
1285
  if (appDuration <= 0) return null;
1260
- const peakExecutors = executorsAdded.length;
1261
- let totalActiveMs = 0;
1262
- const removed = new Map ();
1263
- for (const ev of executorsRemoved) removed.set(ev.executorId, ev.timestamp);
1264
- for (const ev of executorsAdded) {
1265
- const addedAt = Math.max(ev.timestamp, app.startTime);
1266
- const removedAt = removed.has(ev.executorId) ? removed.get(ev.executorId) : app.endTime;
1267
- totalActiveMs += Math.max(0, removedAt - addedAt);
1268
- }
1269
- const utilization = (totalActiveMs / appDuration) / peakExecutors;
1286
+ // computePeakConcurrentCores (not executorsAdded.length/computeTotalCores): real concurrent
1287
+ // capacity, not a cumulative sum that double-counts a churned-through executor against its
1288
+ // replacement's (spot preemption, dynamicAllocation replacement).
1289
+ const totalCores = computePeakConcurrentCores(app, executorsAdded, executorsRemoved);
1290
+ if (totalCores <= 0) return null;
1291
+ const capacityCoreMs = totalCores * appDuration;
1292
+ // Busy core-time (from the whole-run core-time-series, same signal memoryUtilization's
1293
+ // idleCores variant already uses), not executor lifetime: an executor that exists for the
1294
+ // whole run but sits fully idle must not score as 100% used. Missing runAggregates (older
1295
+ // callers, synthetic fixtures) reads as 0 busy time rather than falling back to the
1296
+ // lifetime-based measure this replaces.
1297
+ const busyCoreMs = runAggregates?.busyCoreMs ?? 0;
1298
+ const utilization = busyCoreMs / capacityCoreMs;
1270
1299
  if (utilization >= this.thresholds.minUtil) return null;
1271
1300
 
1272
1301
  // CPU-time-based utilization (sparkMeasure): metric only, no threshold.
1273
- // Total cores: prefer executor-added Total Cores (real), else config cores.
1274
- const totalCores = computeTotalCores(app, executorsAdded);
1275
1302
  let cpuUtilizationPct = null;
1276
1303
  if (totalCores > 0) {
1277
1304
  let cpuMs = 0;
@@ -1296,10 +1323,10 @@ export const DETECTORS = [
1296
1323
  type: 'memoryUtilization', scope: 'app', order: 102, fixEffort: 'config', version: 1,
1297
1324
  docAnchor: '#bottleneck-memory-utilization',
1298
1325
  thresholds: {
1299
- idleCoreWarn: 0.50, // dataflint WastedCoresAlertsReducer
1300
- bandTooSmall: 0.95, // dataflint MemoryAlertsReducer: used/allocated
1326
+ idleCoreWarn: 0.50, // WastedCoresAlertsReducer
1327
+ bandTooSmall: 0.95, // MemoryAlertsReducer: used/allocated
1301
1328
  bandTooHigh: 0.70, // below this => over-provisioned (cost signal)
1302
- wasteBufferMultiplier: 1.5, // Dr. Elephant SparkMetricsAggregator: UNVERIFIED
1329
+ wasteBufferMultiplier: 1.5, // UNVERIFIED
1303
1330
  },
1304
1331
  detect(
1305
1332
 
@@ -1309,19 +1336,20 @@ export const DETECTORS = [
1309
1336
 
1310
1337
  ctx ,
1311
1338
  ) {
1312
- const { app, executorsAdded, runAggregates, stages } = ctx;
1339
+ const { app, executorsAdded, executorsRemoved, runAggregates, stages } = ctx;
1313
1340
  const out = [];
1314
- // Nullish (not falsy) check: unlike the older coldStart/utilization
1315
- // detectors, a literal startTime:0 must not be treated as "missing".
1341
+ // Nullish (not falsy) check: a literal startTime:0 must not be treated as "missing".
1316
1342
  if (app?.startTime == null || app?.endTime == null) return out;
1317
1343
  const appDurationMs = app.endTime - app.startTime;
1318
1344
  if (appDurationMs <= 0) return out;
1319
1345
 
1320
- // Total cores: prefer real Executor-Added Total Cores, else config.
1321
- const peakExecutors = executorsAdded.length;
1322
- const totalCores = computeTotalCores(app, executorsAdded);
1323
- // Hoisted above 1a (it is also 1b/1c's input) so the idle-cores finding can carry
1324
- // the allocated memory its MB-seconds impact estimate needs.
1346
+ // Peak concurrent executors/cores (not executorsAdded.length/computeTotalCores): a
1347
+ // cumulative sum or count double-counts a churned-through executor against its replacement's
1348
+ // (spot preemption, dynamicAllocation replacement), inflating idle-rate and waste-model figures.
1349
+ const peakExecutors = computePeakConcurrentExecutorCount(executorsAdded, executorsRemoved);
1350
+ const totalCores = computePeakConcurrentCores(app, executorsAdded, executorsRemoved);
1351
+ // Hoisted above 1a (also 1b/1c's input) so the idle-cores finding carries the allocated
1352
+ // memory its MB-seconds estimate needs.
1325
1353
  const allocatedMB = app.resources?.executor?.memoryMB ?? null;
1326
1354
 
1327
1355
  // ── 1a idle-cores rate ────────────────────────────────────────────────
@@ -1333,8 +1361,7 @@ export const DETECTORS = [
1333
1361
  out.push({
1334
1362
  type: 'memoryUtilization', variant: 'idleCores', stageId: null,
1335
1363
  impactBand: 'warning', metric: 'idleCoreRate', value,
1336
- // Raw (unrounded) rate plus the sizing inputs, for the impact estimator's
1337
- // wasted-MB-seconds model: `value` above is a rounded percentage.
1364
+ // Raw (unrounded) rate plus sizing inputs for the impact estimator: `value` is rounded pct.
1338
1365
  idleRateFraction: idleRate, allocatedMB, peakExecutors, appDurationMs,
1339
1366
  recommendation: `${value}% of allocated core-time ran no task: reduce cluster size or enable dynamic allocation.`,
1340
1367
  });
@@ -1362,10 +1389,8 @@ export const DETECTORS = [
1362
1389
  const allocatedBytes = allocatedMB * 1024 * 1024;
1363
1390
  for (const [execId, heap] of peakHeapByExec) {
1364
1391
  const ratio = heap / allocatedBytes;
1365
- // The two bands are opposite signals, not two degrees of one: an explicit
1366
- // `rule` discriminator (same pattern as stageShape) lets consumers tell the
1367
- // OOM-risk case from the over-provisioning waste case without re-deriving
1368
- // the ratio against the thresholds.
1392
+ // The two bands are opposite signals: an explicit `rule` discriminator lets consumers
1393
+ // tell OOM-risk from over-provisioning without re-deriving the ratio.
1369
1394
  if (ratio > this.thresholds.bandTooSmall) {
1370
1395
  out.push({
1371
1396
  type: 'memoryUtilization', variant: 'memoryBand', rule: 'heapNearCapacity',
@@ -1387,7 +1412,7 @@ export const DETECTORS = [
1387
1412
  }
1388
1413
  }
1389
1414
 
1390
- // ── 1c Spark Memory Limit waste model (Dr. Elephant, UNVERIFIED buffer) ─
1415
+ // ── 1c Spark Memory Limit waste model (UNVERIFIED buffer) ─
1391
1416
  if (allocatedMB != null && peakExecutors > 0) {
1392
1417
  const allocatedMBSeconds = peakExecutors * allocatedMB * (appDurationMs / 1000);
1393
1418
  let usedRunTimeMs = 0;
@@ -1399,8 +1424,8 @@ export const DETECTORS = [
1399
1424
  out.push({
1400
1425
  type: 'memoryUtilization', variant: 'wasteModel', stageId: null,
1401
1426
  impactBand: 'info', metric: 'wastedMBSeconds', value,
1402
- confidence: 'low',
1403
- validationRequired: 'Memory-waste estimate uses allocated-vs-used memory-time and an unverified 1.5x buffer ported from Dr. Elephant: confirm against the Spark UI before acting.',
1427
+ confidence: memoryWasteConfidence(wastedMBSeconds, usedMBSeconds, this.thresholds.wasteBufferMultiplier),
1428
+ validationRequired: 'Memory-waste estimate uses allocated-vs-used memory-time and a 1.5x buffer: confirm against the Spark UI before acting.',
1404
1429
  recommendation: `Allocated executor memory sat largely idle over the run (~${value.toLocaleString('en-US')} MB-seconds wasted): review spark.executor.memory and executor count.`,
1405
1430
  });
1406
1431
  }
@@ -1410,13 +1435,9 @@ export const DETECTORS = [
1410
1435
  },
1411
1436
  },
1412
1437
  {
1413
- // Per-RDD cache-utilization proxies (this repo's own design: Spark
1414
- // event logs carry no block-access/read-count events, so a literal
1415
- // cache hit rate isn't derivable; see the design doc for GitHub issue
1416
- // #85). Two independent, per-RDD tiered checks over `ctx.app.rddInfo`
1417
- // storage snapshots: partial caching (numCachedPartitions < numPartitions)
1418
- // and disk spillover (diskSize share of a MEMORY_AND_DISK*-requesting
1419
- // RDD's cached footprint). An RDD can produce both findings in one pass.
1438
+ // Per-RDD cache-utilization proxies (this repo's own design: Spark event logs carry no
1439
+ // block-access events, so a literal cache hit rate isn't derivable). Two per-RDD tiered
1440
+ // checks over rddInfo snapshots: partial caching and disk spillover. An RDD can produce both.
1420
1441
  type: 'cacheUtilization', scope: 'app', order: 103, fixEffort: 'code', version: 1,
1421
1442
  docAnchor: '#memory-model',
1422
1443
  thresholds: {
@@ -1458,11 +1479,9 @@ export const DETECTORS = [
1458
1479
  },
1459
1480
  },
1460
1481
  {
1461
- // Non-local task ratio across stage.localityStats (RACK_LOCAL + ANY vs.
1462
- // all tasks), the other half of dataflint's "Wasted Cores Ratio" alert;
1463
- // the idle-core half already lives in memoryUtilization's idleCores
1464
- // variant above. NO_PREF stays in the denominator only: it's what
1465
- // shuffle-read stages legitimately report with no locality problem.
1482
+ // Non-local task ratio across stage.localityStats (RACK_LOCAL + ANY vs all tasks), the other
1483
+ // half of the "Wasted Cores Ratio" (idle-core half is memoryUtilization's idleCores).
1484
+ // NO_PREF stays in the denominator only: shuffle-read stages legitimately report it.
1466
1485
  type: 'coreLocality', scope: 'app', order: 103, fixEffort: 'config', version: 1,
1467
1486
  docAnchor: '#bottleneck-utilization',
1468
1487
  thresholds: { minTasks: 50, warnRatio: 0.15, critRatio: 0.35 },
@@ -1472,11 +1491,8 @@ export const DETECTORS = [
1472
1491
  ) {
1473
1492
  const { totalTasks, nonLocalTasks, ratio } = computeCoreLocalityRatio([...ctx.stages.values()]);
1474
1493
  if (totalTasks == null || totalTasks < this.thresholds.minTasks) return null;
1475
- // computeCoreLocalityRatio (src/core-locality-ratio.ts) only ever returns
1476
- // `ratio: null` together with `totalTasks: null` (its EMPTY sentinel sets
1477
- // both at once); the `totalTasks == null` guard above already rules that
1478
- // out, so `ratio` is guaranteed non-null here even though the function's
1479
- // declared return type keeps the two nullable independently.
1494
+ // computeCoreLocalityRatio only returns ratio:null together with totalTasks:null (shared
1495
+ // EMPTY sentinel); the totalTasks guard above rules that out, so ratio is non-null here.
1480
1496
  if (ratio < this.thresholds.warnRatio) return null;
1481
1497
 
1482
1498
  const value = Math.round(ratio * 100);
@@ -1484,33 +1500,28 @@ export const DETECTORS = [
1484
1500
  type: 'coreLocality', stageId: null,
1485
1501
  impactBand: ratio >= this.thresholds.critRatio ? 'critical' : 'warning',
1486
1502
  metric: 'nonLocalRatio', value,
1487
- // Raw count behind the ratio, for the impact estimator's network-fetch-penalty
1488
- // figure. Non-null whenever totalTasks is (both come from the same EMPTY sentinel).
1503
+ // Raw count behind the ratio, for the impact estimator. Non-null whenever totalTasks is.
1489
1504
  nonLocalTaskCount: nonLocalTasks ,
1490
- confidence: 'low',
1491
- validationRequired: 'The 15%/35% non-local-ratio thresholds (and the 50-task minimum) are unvalidated design-spike values: no external tool publishes an equivalent metric to calibrate against. Confirm against known-good/known-bad real logs before trusting the impact-band split.',
1505
+ confidence: coreLocalityConfidence(ratio , totalTasks, this.thresholds),
1506
+ validationRequired: 'This finding is gated by 15%/35% non-local-ratio thresholds (and a 50-task minimum), our own noise floor for this metric.',
1492
1507
  recommendation: `${value}% of tasks (${nonLocalTasks }) ran without process- or node-local data placement: check spark.locality.wait settings and executor/data colocation.`,
1493
1508
  };
1494
1509
  },
1495
1510
  },
1496
1511
  {
1497
- // Short-lived executors: an executor stood up and torn down before it
1498
- // could do useful work: wasteful re-provisioning, not normal scale-down.
1499
- // Reuses the same executorsAdded/executorsRemoved matching logic as
1500
- // `utilization` above (no new data extraction), but measures lifetime
1501
- // against a threshold instead of aggregate active-time.
1512
+ // Short-lived executors: stood up and torn down before doing useful work (wasteful
1513
+ // re-provisioning, not normal scale-down). Reuses utilization's add/remove matching, but
1514
+ // measures lifetime against a threshold instead of aggregate active-time.
1502
1515
  type: 'autoscalingChurn', scope: 'app', order: 103, fixEffort: 'config', version: 1,
1503
- confidence: 'low',
1504
1516
  thresholds: { shortLivedMs: 120_000, warningPct: 0.30, criticalPct: 0.60, minExecutors: 5 },
1505
1517
  detect(
1506
1518
 
1507
1519
 
1508
-
1509
1520
 
1510
1521
  ctx ,
1511
1522
  ) {
1512
1523
  const { app, executorsAdded, executorsRemoved } = ctx;
1513
- if (executorsAdded.length === 0 || app.endTime == null) return null;
1524
+ if (!app || executorsAdded.length === 0 || app.endTime == null) return null;
1514
1525
  if (executorsAdded.length < this.thresholds.minExecutors) return null;
1515
1526
 
1516
1527
  const removedAt = new Map ();
@@ -1534,16 +1545,14 @@ export const DETECTORS = [
1534
1545
  metric: 'shortLivedExecutorPct', value: pct,
1535
1546
  // Raw count behind the percentage, for the impact estimator's startup-overhead figure.
1536
1547
  shortLivedExecutorCount: shortLivedCount,
1537
- confidence: this.confidence,
1548
+ confidence: autoscalingChurnConfidence(shortLivedPct, this.thresholds.warningPct, this.thresholds.criticalPct),
1538
1549
  recommendation: `${pct}% of executors ran for under 2 minutes before being removed. This looks like wasteful re-provisioning rather than normal scale-down; consider raising spark.dynamicAllocation.executorIdleTimeout or widening the minExecutors/maxExecutors bounds to reduce flapping.`,
1539
1550
  };
1540
1551
  },
1541
1552
  },
1542
1553
  {
1543
- // Cross-execution relation reuse (repurposed from the old RDD-lineage
1544
- // heuristic, which surfaced only internal query-engine RDDs on DataFrame/
1545
- // SQL workloads; see docs/adr/0009-caching-opportunity-relation-reuse.md).
1546
- // Flags an input relation scanned by two or more SQL executions in one run.
1554
+ // Cross-execution relation reuse: flags an input relation scanned by two or more SQL
1555
+ // executions in one run, firing on real relation names (parquet:..., jdbc:...).
1547
1556
  type: 'cachingOpportunity', scope: 'app', order: 105, fixEffort: 'code', version: 1,
1548
1557
  docAnchor: '#bottleneck-utilization',
1549
1558
  thresholds: { minExecutions: 2 },
@@ -1566,12 +1575,10 @@ export const DETECTORS = [
1566
1575
 
1567
1576
 
1568
1577
 
1569
- // relationId -> { format, relation, executionIds:Set, executionBytes: Map<execId, bytes> }
1570
1578
  const byRelation = new Map ();
1571
1579
  for (const exec of sql.values()) {
1572
1580
  if (!exec.planTree) continue;
1573
- // Dedupe relations within one execution (self-joins count once), summing
1574
- // this execution's read bytes per relation across its scan nodes.
1581
+ // Dedupe relations within one execution (self-joins count once), summing read bytes per relation.
1575
1582
  const perExec = new Map ();
1576
1583
  walkPlanTree(exec.planTree, (node) => {
1577
1584
  const rid = scanRelationId(node.name ?? '', node.detail ?? '');
@@ -1591,15 +1598,13 @@ export const DETECTORS = [
1591
1598
  }
1592
1599
  }
1593
1600
 
1594
- // fingerprint -> { operator, exampleNode, executionIds:Set, executionBytes:Map<execId,bytes>, ancestorFingerprints:Set<fingerprint>, leafRelationRids:Set<rid> }
1595
1601
  const byComposite = new Map ();
1596
1602
  for (const exec of sql.values()) {
1597
1603
  if (!exec.planTree) continue;
1598
1604
  const candidates = findCompositeCandidates(exec.planTree);
1599
1605
  const fingerprintByNode = new Map (candidates.map((c) => [c.node, c.fingerprint]));
1600
1606
 
1601
- // Dedupe identical fingerprints within this execution (repeated
1602
- // identical composite counts once, mirroring the leaf perExec dedupe).
1607
+ // Dedupe identical fingerprints within this execution (repeated composite counts once).
1603
1608
  const perExecComposite = new Map ();
1604
1609
  for (const c of candidates) {
1605
1610
  let agg = perExecComposite.get(c.fingerprint);
@@ -1636,12 +1641,10 @@ export const DETECTORS = [
1636
1641
  }
1637
1642
  }
1638
1643
 
1639
- // Qualifying = has enough distinct executions on its own. Nested-dedupe:
1640
- // if a qualifying composite has a qualifying ANCESTOR composite, it is
1641
- // subsumed: fully (equal execution sets) or partially (residual).
1644
+ // Qualifying = enough distinct executions on its own. Nested-dedupe: a qualifying composite
1645
+ // with a qualifying ANCESTOR is subsumed, fully (equal sets) or partially (residual).
1642
1646
  const isQualifying = (fp ) =>
1643
1647
  byComposite.has(fp) && byComposite.get(fp) .executionIds.size >= this.thresholds.minExecutions;
1644
- // fingerprint -> { finalExecutionIds:Set, suppressed:boolean }
1645
1648
  const compositeResolutions = new Map ();
1646
1649
  for (const [fingerprint, agg] of byComposite) {
1647
1650
  if (!isQualifying(fingerprint)) { compositeResolutions.set(fingerprint, { finalExecutionIds: agg.executionIds, suppressed: true }); continue; }
@@ -1658,7 +1661,7 @@ export const DETECTORS = [
1658
1661
  const compositeVerb = { join: ['joined', 'join'], union: ['unioned', 'union'] };
1659
1662
 
1660
1663
  const out = [];
1661
- // rid -> Set<execId> covered by an emitted composite (for leaf suppression, Task 6)
1664
+ // rid -> Set<execId> covered by an emitted composite, for leaf suppression.
1662
1665
  const coveredExecutionsByRid = new Map ();
1663
1666
  for (const [fingerprint, agg] of byComposite) {
1664
1667
  const resolution = compositeResolutions.get(fingerprint) ;
@@ -1683,7 +1686,7 @@ export const DETECTORS = [
1683
1686
  metric: 'executionReuse', value,
1684
1687
  format: 'derived', relations, operator: agg.operator, relation: relationDisplay,
1685
1688
  executionIds: finalExecutionIds, totalReadBytes,
1686
- confidence: 'low',
1689
+ confidence: cachingReuseConfidence(value, this.thresholds.minExecutions),
1687
1690
  validationRequired:
1688
1691
  'Composite reuse is inferred from a structural plan-shape match (operator + normalized ' +
1689
1692
  'join/filter condition + child shapes) across SQL executions; confirm these executions ' +
@@ -1714,7 +1717,7 @@ export const DETECTORS = [
1714
1717
  relation: agg.relation, format: agg.format,
1715
1718
  executionIds: residualExecutionIds.sort((a, b) => a - b),
1716
1719
  totalReadBytes,
1717
- confidence: 'low',
1720
+ confidence: cachingReuseConfidence(value, this.thresholds.minExecutions),
1718
1721
  validationRequired:
1719
1722
  'Relation-reuse is inferred from the pre-AQE plan scan identity across SQL ' +
1720
1723
  'executions; confirm the reads are the same data and cacheable within one ' +
@@ -1744,10 +1747,8 @@ export const DETECTORS = [
1744
1747
  let totalTasks = 0, failedTasks = 0;
1745
1748
  for (const s of stages.values()) { totalTasks += s.taskCount ?? 0; failedTasks += s.failedTasks ?? 0; }
1746
1749
  const taskFailureRate = totalTasks > 0 ? failedTasks / totalTasks : 0;
1747
- // Average wall-clock duration of the failed jobs, for the impact estimator's
1748
- // cost-only "core-hours burned on work that was thrown away" figure. Jobs
1749
- // missing either timestamp are excluded rather than counted as zero-length;
1750
- // with no timed failed job at all the average is 0 (never NaN).
1750
+ // Average wall-clock of failed jobs, for the impact estimator's cost-only figure. Jobs
1751
+ // missing either timestamp are excluded (not counted as zero); with none timed the average is 0.
1751
1752
  const timedFailedJobs = failedJobList.filter(j => j.submissionTime != null && j.completionTime != null);
1752
1753
  const avgJobDurationMs = timedFailedJobs.length > 0
1753
1754
  ? timedFailedJobs.reduce((s, j) => s + (j.completionTime - j.submissionTime ), 0) / timedFailedJobs.length
@@ -1858,15 +1859,17 @@ export const DETECTORS = [
1858
1859
  const nodes = [];
1859
1860
  for (const n of g.nodes) walkPlanTree(n, (node) => nodes.push(node));
1860
1861
  const stageIds = unionStageIds(nodes, fallbackStageIds);
1862
+ // resolvePlanTree always sets id; safe downstream of it.
1863
+ const planNodeIds = nodes.map((n) => n.id ).filter(Boolean);
1861
1864
  const touching = g.sampleRelation ? ` (touching ${g.sampleRelation})` : '';
1862
1865
  return {
1863
- type: 'duplicatePlanSubtree', executionId: sqlExec.id, stageIds,
1864
- // Fixed fallback: overwritten by deriveImpactBand whenever this finding
1865
- // gets a real wallClock estimate, which is the common case. Only
1866
- // surfaces on the rare miss (stage excluded from the occupancy sweep).
1866
+ type: 'duplicatePlanSubtree', executionId: sqlExec.id, stageIds, planNodeIds,
1867
+ // Fixed fallback: overwritten by deriveImpactBand when this finding gets a real
1868
+ // wallClock estimate (the common case). Only surfaces on the rare occupancy-sweep miss.
1867
1869
  impactBand: 'warning', metric: 'subtreeOccurrences', value: g.occurrences,
1868
1870
  rootName: g.rootName, subtreeSize: g.subtreeSize, sampleRelation: g.sampleRelation,
1869
- groupIndex: g.groupIndex, confidence: 'medium',
1871
+ groupIndex: g.groupIndex,
1872
+ confidence: duplicateSubtreeConfidence(g.subtreeSize, g.occurrences, this.thresholds),
1870
1873
  validationRequired: 'Duplicate-subtree matching compares operator names and metric names only, not literal values or expr IDs: confirm the repeated work is real in the Spark SQL plan tab before acting.',
1871
1874
  recommendation: g.isExchangeRoot
1872
1875
  ? `A ${g.subtreeSize}-node subtree rooted at ${pathBasename(g.rootName)} repeats ${g.occurrences}x in this plan${touching}: this looks like a possible missed exchange reuse; check whether the same shuffle could be computed once and reused.`
@@ -1908,8 +1911,9 @@ export const DETECTORS = [
1908
1911
  const fallbackStageIds = stageIdsForSqlExec(sqlExec.id, ctx.stages);
1909
1912
  return hits.map(h => {
1910
1913
  const stageIds = unionStageIds([h.node], fallbackStageIds);
1914
+ const planNodeIds = h.node.id ? [h.node.id] : [];
1911
1915
  return {
1912
- type: 'smallFiles', executionId: sqlExec.id, stageIds,
1916
+ type: 'smallFiles', executionId: sqlExec.id, stageIds, planNodeIds,
1913
1917
  impactBand: 'warning',
1914
1918
  metric: 'avgFileSizeBytes', value: Math.round(h.avgBytes),
1915
1919
  fileCount: h.fileCount, direction: h.direction, nodeName: h.nodeName,
@@ -1921,10 +1925,9 @@ export const DETECTORS = [
1921
1925
  },
1922
1926
  },
1923
1927
  {
1924
- // Entry-level type is an identifier only; it never appears on an
1925
- // emitted finding. Findings carry their own type ('underBroadcast' or
1926
- // 'overBroadcast') since one shared plan-walk covers both opposite-
1927
- // direction rules (dataflint JoinToBroadcastAlert / BroadcastTooLargeAlert).
1928
+ // Entry-level type is an identifier only; it never appears on an emitted finding. Findings
1929
+ // carry 'underBroadcast'/'overBroadcast' since one shared plan-walk covers both
1930
+ // opposite-direction rules (JoinToBroadcastAlert / BroadcastTooLargeAlert).
1928
1931
  type: 'broadcastSizing', scope: 'sql', order: 132, fixEffort: 'config', version: 2,
1929
1932
  docAnchor: '#bottleneck-broadcast-sizing',
1930
1933
  thresholds: {
@@ -1959,6 +1962,8 @@ export const DETECTORS = [
1959
1962
  const contributors = [...boundarySizeContributors(childA), ...boundarySizeContributors(childB)];
1960
1963
  out.push({
1961
1964
  type: 'underBroadcast', executionId: sqlExec.id, stageIds: unionStageIds(contributors, fallbackStageIds),
1965
+ // resolvePlanTree always sets id; safe downstream of it.
1966
+ planNodeIds: contributors.map((n) => n.id ).filter(Boolean),
1962
1967
  impactBand: 'info', metric: 'smallerSideBytes', value: smaller,
1963
1968
  largerSideBytes: larger,
1964
1969
  recommendation: `The smaller input to this Sort Merge Join (${formatBytes(smaller)}) is well under the broadcast threshold relative to the larger side (${formatBytes(larger)}): this could have been a broadcast join. Consider a broadcast() hint or raising spark.sql.autoBroadcastJoinThreshold.`,
@@ -1966,17 +1971,18 @@ export const DETECTORS = [
1966
1971
  }
1967
1972
  }
1968
1973
  }
1969
- if (node.name === 'BroadcastExchange') {
1974
+ if (isBroadcastExchangeNode(node.name)) {
1970
1975
  const m = (node.metrics ?? []).find(x => x.name === 'data size');
1971
1976
  if (m && m.value > overBroadcastBytes) {
1972
- // The BroadcastExchange node's own metrics are computed on the
1973
- // driver and never appear on any TaskEnd (node.stageIds is
1974
- // always empty in real data); its child carries the
1975
- // executor-side metrics, so only the child is unioned in.
1977
+ // BroadcastExchange's own metrics are driver-computed and never on any TaskEnd
1978
+ // (node.stageIds always empty in real data); its child carries the executor-side
1979
+ // metrics, so only the child unions in.
1976
1980
  const child = (node.children ?? [])[0];
1977
1981
  out.push({
1978
1982
  type: 'overBroadcast', executionId: sqlExec.id,
1979
1983
  stageIds: unionStageIds(child ? [child] : [], fallbackStageIds),
1984
+ // resolvePlanTree always sets id; safe downstream of it.
1985
+ planNodeIds: [node.id ].filter(Boolean),
1980
1986
  impactBand: 'warning', metric: 'broadcastBytes', value: m.value,
1981
1987
  recommendation: `This broadcast (${formatBytes(m.value)}) exceeds the 1 GB threshold: check for a misapplied broadcast hint or a misconfigured spark.sql.autoBroadcastJoinThreshold.`,
1982
1988
  });