sparkforensics-cli 0.1.0 → 0.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (360) hide show
  1. package/README.md +6 -0
  2. package/bin/sparkforensics-analyze.mjs +113 -48
  3. package/export-template/docs/404.html +25 -0
  4. package/export-template/docs/assets/app.CndaAS6v.js +1 -0
  5. package/export-template/docs/assets/aqe-loop.IwQSATHw.svg +1 -0
  6. package/export-template/docs/assets/aqe-loop.dark.DGbaxqJE.svg +1 -0
  7. package/export-template/docs/assets/broadcast-vs-shuffle.Db4WY1XK.svg +1 -0
  8. package/export-template/docs/assets/broadcast-vs-shuffle.dark.C7Bxs0mG.svg +1 -0
  9. package/export-template/docs/assets/cache-lifecycle.dark.B-hS7AgU.svg +1 -0
  10. package/export-template/docs/assets/cache-lifecycle.rEOVYQNU.svg +1 -0
  11. package/export-template/docs/assets/chunks/@localSearchIndexroot.DNY8bVcl.js +1 -0
  12. package/export-template/docs/assets/chunks/VPLocalSearchBox.yJbZbsEo.js +9 -0
  13. package/export-template/docs/assets/chunks/duplicate-plan-subtree.dark.Cdp70QhV.js +1 -0
  14. package/export-template/docs/assets/chunks/framework.DSg0KOwT.js +20 -0
  15. package/export-template/docs/assets/chunks/retry-escalation-ladder.dark.DHipdJgZ.js +1 -0
  16. package/export-template/docs/assets/chunks/theme.Df2VAG9w.js +2 -0
  17. package/export-template/docs/assets/cold-start-timeline.DxC_Sc7w.svg +1 -0
  18. package/export-template/docs/assets/cold-start-timeline.dark.CZ17YcAG.svg +1 -0
  19. package/export-template/docs/assets/columnar-layout.PghGeOEA.svg +1 -0
  20. package/export-template/docs/assets/columnar-layout.dark.BVNlz0ff.svg +1 -0
  21. package/export-template/docs/assets/container-memory.DIO0AnIm.svg +1 -0
  22. package/export-template/docs/assets/container-memory.dark.CP-5zuCl.svg +1 -0
  23. package/export-template/docs/assets/contributor-guide_architecture_board-widgets.md.B-OsL91z.js +1 -0
  24. package/export-template/docs/assets/contributor-guide_architecture_board-widgets.md.B-OsL91z.lean.js +1 -0
  25. package/export-template/docs/assets/contributor-guide_architecture_detector-contract.md.BOeH4d1J.js +1 -0
  26. package/export-template/docs/assets/contributor-guide_architecture_detector-contract.md.BOeH4d1J.lean.js +1 -0
  27. package/export-template/docs/assets/contributor-guide_architecture_drill-down.md.BtPdlM7r.js +1 -0
  28. package/export-template/docs/assets/contributor-guide_architecture_drill-down.md.BtPdlM7r.lean.js +1 -0
  29. package/export-template/docs/assets/contributor-guide_architecture_impact-estimation.md.DYCDPgkh.js +1 -0
  30. package/export-template/docs/assets/contributor-guide_architecture_impact-estimation.md.DYCDPgkh.lean.js +1 -0
  31. package/export-template/docs/assets/contributor-guide_architecture_index.md.3TO9ic6w.js +1 -0
  32. package/export-template/docs/assets/contributor-guide_architecture_index.md.3TO9ic6w.lean.js +1 -0
  33. package/export-template/docs/assets/contributor-guide_architecture_overview.md.CehiRmGn.js +1 -0
  34. package/export-template/docs/assets/contributor-guide_architecture_overview.md.CehiRmGn.lean.js +1 -0
  35. package/export-template/docs/assets/contributor-guide_architecture_state-and-history.md.m3S3UdMk.js +1 -0
  36. package/export-template/docs/assets/contributor-guide_architecture_state-and-history.md.m3S3UdMk.lean.js +1 -0
  37. package/export-template/docs/assets/contributor-guide_architecture_widget-rendering.md.DbqPf2OT.js +1 -0
  38. package/export-template/docs/assets/contributor-guide_architecture_widget-rendering.md.DbqPf2OT.lean.js +1 -0
  39. package/export-template/docs/assets/contributor-guide_architecture_worker-protocol.md.B93qJ_tT.js +6 -0
  40. package/export-template/docs/assets/contributor-guide_architecture_worker-protocol.md.B93qJ_tT.lean.js +1 -0
  41. package/export-template/docs/assets/contributor-guide_contributing.md.CvRsdr6J.js +1 -0
  42. package/export-template/docs/assets/contributor-guide_contributing.md.CvRsdr6J.lean.js +1 -0
  43. package/export-template/docs/assets/contributor-guide_development-setup.md.DvAN_9mK.js +12 -0
  44. package/export-template/docs/assets/contributor-guide_development-setup.md.DvAN_9mK.lean.js +1 -0
  45. package/export-template/docs/assets/contributor-guide_testing.md.6rIKqSyY.js +1 -0
  46. package/export-template/docs/assets/contributor-guide_testing.md.6rIKqSyY.lean.js +1 -0
  47. package/export-template/docs/assets/dag-stages.DSz_S937.svg +1 -0
  48. package/export-template/docs/assets/dag-stages.dark.F72UzxH4.svg +1 -0
  49. package/export-template/docs/assets/driver-executor.D5pQ7YN1.svg +1 -0
  50. package/export-template/docs/assets/driver-executor.dark.BmX9cPvh.svg +1 -0
  51. package/export-template/docs/assets/duplicate-plan-subtree.B4cvN6fj.svg +1 -0
  52. package/export-template/docs/assets/duplicate-plan-subtree.dark.Dw8wS0Ag.svg +1 -0
  53. package/export-template/docs/assets/index.md.CHJVslga.js +1 -0
  54. package/export-template/docs/assets/index.md.CHJVslga.lean.js +1 -0
  55. package/export-template/docs/assets/inter-italic-cyrillic-ext.r48I6akx.woff2 +0 -0
  56. package/export-template/docs/assets/inter-italic-cyrillic.By2_1cv3.woff2 +0 -0
  57. package/export-template/docs/assets/inter-italic-greek-ext.1u6EdAuj.woff2 +0 -0
  58. package/export-template/docs/assets/inter-italic-greek.DJ8dCoTZ.woff2 +0 -0
  59. package/export-template/docs/assets/inter-italic-latin-ext.CN1xVJS-.woff2 +0 -0
  60. package/export-template/docs/assets/inter-italic-latin.C2AdPX0b.woff2 +0 -0
  61. package/export-template/docs/assets/inter-italic-vietnamese.BSbpV94h.woff2 +0 -0
  62. package/export-template/docs/assets/inter-roman-cyrillic-ext.BBPuwvHQ.woff2 +0 -0
  63. package/export-template/docs/assets/inter-roman-cyrillic.C5lxZ8CY.woff2 +0 -0
  64. package/export-template/docs/assets/inter-roman-greek-ext.CqjqNYQ-.woff2 +0 -0
  65. package/export-template/docs/assets/inter-roman-greek.BBVDIX6e.woff2 +0 -0
  66. package/export-template/docs/assets/inter-roman-latin-ext.4ZJIpNVo.woff2 +0 -0
  67. package/export-template/docs/assets/inter-roman-latin.Di8DUHzh.woff2 +0 -0
  68. package/export-template/docs/assets/inter-roman-vietnamese.BjW4sHH5.woff2 +0 -0
  69. package/export-template/docs/assets/join-strategy.C_FvrCEo.svg +1 -0
  70. package/export-template/docs/assets/join-strategy.dark.ChMLnNII.svg +1 -0
  71. package/export-template/docs/assets/memory-borrowing.BqQRJg0u.svg +1 -0
  72. package/export-template/docs/assets/memory-borrowing.dark.Yhh20O9C.svg +1 -0
  73. package/export-template/docs/assets/memory-regions.XHvO7jHG.svg +1 -0
  74. package/export-template/docs/assets/memory-regions.dark.D4TP9_08.svg +1 -0
  75. package/export-template/docs/assets/repartition-vs-coalesce.BovLRrpj.svg +1 -0
  76. package/export-template/docs/assets/repartition-vs-coalesce.dark.BhAczKZQ.svg +1 -0
  77. package/export-template/docs/assets/retry-escalation-ladder.DyTKJJmZ.svg +1 -0
  78. package/export-template/docs/assets/retry-escalation-ladder.dark.BdsabtU3.svg +1 -0
  79. package/export-template/docs/assets/shuffle-map-reduce.KuOEZVmg.svg +1 -0
  80. package/export-template/docs/assets/shuffle-map-reduce.dark.BgQZnFSb.svg +1 -0
  81. package/export-template/docs/assets/spill-classification.BU2euYDO.svg +1 -0
  82. package/export-template/docs/assets/spill-classification.dark.D7i1M40d.svg +1 -0
  83. package/export-template/docs/assets/style.DXOMCXxn.css +1 -0
  84. package/export-template/docs/assets/tuning-reference_anti-patterns.md.Df1YMIHu.js +1 -0
  85. package/export-template/docs/assets/tuning-reference_anti-patterns.md.Df1YMIHu.lean.js +1 -0
  86. package/export-template/docs/assets/tuning-reference_aqe.md.BIsCtLzm.js +1 -0
  87. package/export-template/docs/assets/tuning-reference_aqe.md.BIsCtLzm.lean.js +1 -0
  88. package/export-template/docs/assets/tuning-reference_bottleneck-broadcast-sizing.md.CEstB3Ia.js +1 -0
  89. package/export-template/docs/assets/tuning-reference_bottleneck-broadcast-sizing.md.CEstB3Ia.lean.js +1 -0
  90. package/export-template/docs/assets/tuning-reference_bottleneck-cold-start.md.CEuy-72y.js +7 -0
  91. package/export-template/docs/assets/tuning-reference_bottleneck-cold-start.md.CEuy-72y.lean.js +1 -0
  92. package/export-template/docs/assets/tuning-reference_bottleneck-duplicate-plan-subtree.md.CIohQDfn.js +1 -0
  93. package/export-template/docs/assets/tuning-reference_bottleneck-duplicate-plan-subtree.md.CIohQDfn.lean.js +1 -0
  94. package/export-template/docs/assets/tuning-reference_bottleneck-failures.md.4z5BXGJ2.js +6 -0
  95. package/export-template/docs/assets/tuning-reference_bottleneck-failures.md.4z5BXGJ2.lean.js +1 -0
  96. package/export-template/docs/assets/tuning-reference_bottleneck-gc.md.DSxzZRK7.js +6 -0
  97. package/export-template/docs/assets/tuning-reference_bottleneck-gc.md.DSxzZRK7.lean.js +1 -0
  98. package/export-template/docs/assets/tuning-reference_bottleneck-job-failure-rate.md.BaJl__1W.js +8 -0
  99. package/export-template/docs/assets/tuning-reference_bottleneck-job-failure-rate.md.BaJl__1W.lean.js +1 -0
  100. package/export-template/docs/assets/tuning-reference_bottleneck-memory-utilization.md.DbP-SJZc.js +7 -0
  101. package/export-template/docs/assets/tuning-reference_bottleneck-memory-utilization.md.DbP-SJZc.lean.js +1 -0
  102. package/export-template/docs/assets/tuning-reference_bottleneck-retry-waste.md.D5JMjVOt.js +1 -0
  103. package/export-template/docs/assets/tuning-reference_bottleneck-retry-waste.md.D5JMjVOt.lean.js +1 -0
  104. package/export-template/docs/assets/tuning-reference_bottleneck-shuffle.md.CM-nTmIH.js +12 -0
  105. package/export-template/docs/assets/tuning-reference_bottleneck-shuffle.md.CM-nTmIH.lean.js +1 -0
  106. package/export-template/docs/assets/tuning-reference_bottleneck-skew.md.BdUwiDhn.js +14 -0
  107. package/export-template/docs/assets/tuning-reference_bottleneck-skew.md.BdUwiDhn.lean.js +1 -0
  108. package/export-template/docs/assets/tuning-reference_bottleneck-slow-host.md.BlIo6UDW.js +7 -0
  109. package/export-template/docs/assets/tuning-reference_bottleneck-slow-host.md.BlIo6UDW.lean.js +1 -0
  110. package/export-template/docs/assets/tuning-reference_bottleneck-small-files.md.B8kloyx8.js +5 -0
  111. package/export-template/docs/assets/tuning-reference_bottleneck-small-files.md.B8kloyx8.lean.js +1 -0
  112. package/export-template/docs/assets/tuning-reference_bottleneck-spill.md.PNH7mITt.js +6 -0
  113. package/export-template/docs/assets/tuning-reference_bottleneck-spill.md.PNH7mITt.lean.js +1 -0
  114. package/export-template/docs/assets/tuning-reference_bottleneck-straggler.md.DY36fHN5.js +7 -0
  115. package/export-template/docs/assets/tuning-reference_bottleneck-straggler.md.DY36fHN5.lean.js +1 -0
  116. package/export-template/docs/assets/tuning-reference_bottleneck-tiny-tasks.md.QTV7O8kU.js +8 -0
  117. package/export-template/docs/assets/tuning-reference_bottleneck-tiny-tasks.md.QTV7O8kU.lean.js +1 -0
  118. package/export-template/docs/assets/tuning-reference_bottleneck-utilization.md.DTiueZC3.js +5 -0
  119. package/export-template/docs/assets/tuning-reference_bottleneck-utilization.md.DTiueZC3.lean.js +1 -0
  120. package/export-template/docs/assets/tuning-reference_caching.md.B7aQ8asB.js +1 -0
  121. package/export-template/docs/assets/tuning-reference_caching.md.B7aQ8asB.lean.js +1 -0
  122. package/export-template/docs/assets/tuning-reference_cluster-config.md.ZVmDGsQ3.js +1 -0
  123. package/export-template/docs/assets/tuning-reference_cluster-config.md.ZVmDGsQ3.lean.js +1 -0
  124. package/export-template/docs/assets/tuning-reference_config.md.UvveiWG3.js +1 -0
  125. package/export-template/docs/assets/tuning-reference_config.md.UvveiWG3.lean.js +1 -0
  126. package/export-template/docs/assets/tuning-reference_data-formats.md.bjCAWH3N.js +1 -0
  127. package/export-template/docs/assets/tuning-reference_data-formats.md.bjCAWH3N.lean.js +1 -0
  128. package/export-template/docs/assets/tuning-reference_index.md.BQ_NooMV.js +1 -0
  129. package/export-template/docs/assets/tuning-reference_index.md.BQ_NooMV.lean.js +1 -0
  130. package/export-template/docs/assets/tuning-reference_intro.md.CobD-lGB.js +1 -0
  131. package/export-template/docs/assets/tuning-reference_intro.md.CobD-lGB.lean.js +1 -0
  132. package/export-template/docs/assets/tuning-reference_joins.md.BtKs_CuW.js +1 -0
  133. package/export-template/docs/assets/tuning-reference_joins.md.BtKs_CuW.lean.js +1 -0
  134. package/export-template/docs/assets/tuning-reference_memory-model.md.DhT-n4y3.js +1 -0
  135. package/export-template/docs/assets/tuning-reference_memory-model.md.DhT-n4y3.lean.js +1 -0
  136. package/export-template/docs/assets/tuning-reference_metrics.md.mLOh7Apj.js +1 -0
  137. package/export-template/docs/assets/tuning-reference_metrics.md.mLOh7Apj.lean.js +1 -0
  138. package/export-template/docs/assets/tuning-reference_partitioning.md.q0zKF_8X.js +1 -0
  139. package/export-template/docs/assets/tuning-reference_partitioning.md.q0zKF_8X.lean.js +1 -0
  140. package/export-template/docs/assets/tuning-reference_pyspark.md.DDCfvN9t.js +6 -0
  141. package/export-template/docs/assets/tuning-reference_pyspark.md.DDCfvN9t.lean.js +1 -0
  142. package/export-template/docs/assets/tuning-reference_shuffle.md.BZZ7R4Ix.js +1 -0
  143. package/export-template/docs/assets/tuning-reference_shuffle.md.BZZ7R4Ix.lean.js +1 -0
  144. package/export-template/docs/assets/tuning-reference_spark-architecture.md.Dwzm5avO.js +1 -0
  145. package/export-template/docs/assets/tuning-reference_spark-architecture.md.Dwzm5avO.lean.js +1 -0
  146. package/export-template/docs/assets/tuning-reference_table-formats.md.D6wj-2dX.js +1 -0
  147. package/export-template/docs/assets/tuning-reference_table-formats.md.D6wj-2dX.lean.js +1 -0
  148. package/export-template/docs/assets/udf-execution-models.BUFDICuG.svg +1 -0
  149. package/export-template/docs/assets/udf-execution-models.dark.YTNS6GDq.svg +1 -0
  150. package/export-template/docs/assets/user-guide_alternative-log-retrieval.md.sU3KGarf.js +1 -0
  151. package/export-template/docs/assets/user-guide_alternative-log-retrieval.md.sU3KGarf.lean.js +1 -0
  152. package/export-template/docs/assets/user-guide_getting-started.md.DtEM37MK.js +3 -0
  153. package/export-template/docs/assets/user-guide_getting-started.md.DtEM37MK.lean.js +1 -0
  154. package/export-template/docs/assets/user-guide_mcp-tools.md.C8MiIu7F.js +125 -0
  155. package/export-template/docs/assets/user-guide_mcp-tools.md.C8MiIu7F.lean.js +1 -0
  156. package/export-template/docs/assets/user-guide_run-comparison.md.S0TWWmLY.js +1 -0
  157. package/export-template/docs/assets/user-guide_run-comparison.md.S0TWWmLY.lean.js +1 -0
  158. package/export-template/docs/assets/user-guide_understanding-findings.md.D0R_Y-R2.js +1 -0
  159. package/export-template/docs/assets/user-guide_understanding-findings.md.D0R_Y-R2.lean.js +1 -0
  160. package/export-template/docs/contributor-guide/architecture/board-widgets.html +25 -0
  161. package/export-template/docs/contributor-guide/architecture/detector-contract.html +25 -0
  162. package/export-template/docs/contributor-guide/architecture/drill-down.html +25 -0
  163. package/export-template/docs/contributor-guide/architecture/impact-estimation.html +25 -0
  164. package/export-template/docs/contributor-guide/architecture/index.html +25 -0
  165. package/export-template/docs/contributor-guide/architecture/overview.html +25 -0
  166. package/export-template/docs/contributor-guide/architecture/state-and-history.html +25 -0
  167. package/export-template/docs/contributor-guide/architecture/widget-rendering.html +25 -0
  168. package/export-template/docs/contributor-guide/architecture/worker-protocol.html +30 -0
  169. package/export-template/docs/contributor-guide/contributing.html +25 -0
  170. package/export-template/docs/contributor-guide/development-setup.html +36 -0
  171. package/export-template/docs/contributor-guide/testing.html +25 -0
  172. package/export-template/docs/favicon.svg +4 -0
  173. package/export-template/docs/hashmap.json +1 -0
  174. package/export-template/docs/index.html +25 -0
  175. package/export-template/docs/package.json +1 -0
  176. package/export-template/docs/tuning-reference/anti-patterns.html +25 -0
  177. package/export-template/docs/tuning-reference/aqe.html +25 -0
  178. package/export-template/docs/tuning-reference/bottleneck-broadcast-sizing.html +25 -0
  179. package/export-template/docs/tuning-reference/bottleneck-cold-start.html +31 -0
  180. package/export-template/docs/tuning-reference/bottleneck-duplicate-plan-subtree.html +25 -0
  181. package/export-template/docs/tuning-reference/bottleneck-failures.html +30 -0
  182. package/export-template/docs/tuning-reference/bottleneck-gc.html +30 -0
  183. package/export-template/docs/tuning-reference/bottleneck-job-failure-rate.html +32 -0
  184. package/export-template/docs/tuning-reference/bottleneck-memory-utilization.html +31 -0
  185. package/export-template/docs/tuning-reference/bottleneck-retry-waste.html +25 -0
  186. package/export-template/docs/tuning-reference/bottleneck-shuffle.html +36 -0
  187. package/export-template/docs/tuning-reference/bottleneck-skew.html +38 -0
  188. package/export-template/docs/tuning-reference/bottleneck-slow-host.html +31 -0
  189. package/export-template/docs/tuning-reference/bottleneck-small-files.html +29 -0
  190. package/export-template/docs/tuning-reference/bottleneck-spill.html +30 -0
  191. package/export-template/docs/tuning-reference/bottleneck-straggler.html +31 -0
  192. package/export-template/docs/tuning-reference/bottleneck-tiny-tasks.html +32 -0
  193. package/export-template/docs/tuning-reference/bottleneck-utilization.html +29 -0
  194. package/export-template/docs/tuning-reference/caching.html +25 -0
  195. package/export-template/docs/tuning-reference/cluster-config.html +25 -0
  196. package/export-template/docs/tuning-reference/config.html +25 -0
  197. package/export-template/docs/tuning-reference/data-formats.html +25 -0
  198. package/export-template/docs/tuning-reference/index.html +25 -0
  199. package/export-template/docs/tuning-reference/intro.html +25 -0
  200. package/export-template/docs/tuning-reference/joins.html +25 -0
  201. package/export-template/docs/tuning-reference/memory-model.html +25 -0
  202. package/export-template/docs/tuning-reference/metrics.html +25 -0
  203. package/export-template/docs/tuning-reference/partitioning.html +25 -0
  204. package/export-template/docs/tuning-reference/pyspark.html +30 -0
  205. package/export-template/docs/tuning-reference/shuffle.html +25 -0
  206. package/export-template/docs/tuning-reference/spark-architecture.html +25 -0
  207. package/export-template/docs/tuning-reference/table-formats.html +25 -0
  208. package/export-template/docs/user-guide/alternative-log-retrieval.html +25 -0
  209. package/export-template/docs/user-guide/getting-started.html +27 -0
  210. package/export-template/docs/user-guide/mcp-tools.html +149 -0
  211. package/export-template/docs/user-guide/run-comparison.html +25 -0
  212. package/export-template/docs/user-guide/understanding-findings.html +25 -0
  213. package/export-template/docs/vp-icons.css +0 -0
  214. package/export-template/favicon.svg +4 -0
  215. package/export-template/index.html +115 -0
  216. package/export-template/parser-worker-QqyEE4m9.js +64 -0
  217. package/package.json +16 -3
  218. package/vendor-core/analyzer.js +74 -74
  219. package/vendor-core/cli/budgets.js +13 -27
  220. package/vendor-core/cli/collect-run.js +43 -19
  221. package/vendor-core/core-count.js +25 -27
  222. package/vendor-core/core-locality-ratio.js +4 -11
  223. package/vendor-core/core-time-series.js +6 -12
  224. package/vendor-core/core-usage-locality.js +3 -4
  225. package/vendor-core/detectors.js +256 -375
  226. package/vendor-core/docs-config.js +69 -21
  227. package/vendor-core/docs-content/chapters/01-intro.md +32 -0
  228. package/vendor-core/docs-content/chapters/02-spark-architecture.md +76 -0
  229. package/vendor-core/docs-content/chapters/03-memory-model.md +73 -0
  230. package/vendor-core/docs-content/chapters/04-partitioning.md +65 -0
  231. package/vendor-core/docs-content/chapters/05-joins.md +62 -0
  232. package/vendor-core/docs-content/chapters/06-shuffle.md +59 -0
  233. package/vendor-core/docs-content/chapters/07-data-formats.md +81 -0
  234. package/vendor-core/docs-content/chapters/07b-table-formats.md +56 -0
  235. package/vendor-core/docs-content/chapters/08-caching.md +58 -0
  236. package/vendor-core/docs-content/chapters/09-pyspark.md +78 -0
  237. package/vendor-core/docs-content/chapters/10-aqe.md +167 -0
  238. package/vendor-core/docs-content/chapters/11-cluster-config.md +170 -0
  239. package/vendor-core/docs-content/chapters/12-anti-patterns.md +171 -0
  240. package/vendor-core/docs-content/chapters/14-metrics.md +87 -0
  241. package/vendor-core/docs-content/chapters/15-config.md +93 -0
  242. package/vendor-core/docs-content/chapters/nav-index.json +370 -0
  243. package/vendor-core/docs-content/detection/cache.md +6 -0
  244. package/vendor-core/docs-content/detection/cfg.md +15 -0
  245. package/vendor-core/docs-content/detection/chrn.md +7 -0
  246. package/vendor-core/docs-content/detection/cold.md +4 -0
  247. package/vendor-core/docs-content/detection/cstor.md +4 -0
  248. package/vendor-core/docs-content/detection/fail.md +5 -0
  249. package/vendor-core/docs-content/detection/gc.md +4 -0
  250. package/vendor-core/docs-content/detection/host.md +5 -0
  251. package/vendor-core/docs-content/detection/incmp.md +6 -0
  252. package/vendor-core/docs-content/detection/jobs.md +4 -0
  253. package/vendor-core/docs-content/detection/local.md +7 -0
  254. package/vendor-core/docs-content/detection/mem.md +10 -0
  255. package/vendor-core/docs-content/detection/part.md +5 -0
  256. package/vendor-core/docs-content/detection/plan.md +14 -0
  257. package/vendor-core/docs-content/detection/retry.md +4 -0
  258. package/vendor-core/docs-content/detection/sfail.md +5 -0
  259. package/vendor-core/docs-content/detection/shape.md +5 -0
  260. package/vendor-core/docs-content/detection/shfl.md +4 -0
  261. package/vendor-core/docs-content/detection/skew.md +6 -0
  262. package/vendor-core/docs-content/detection/slow.md +6 -0
  263. package/vendor-core/docs-content/detection/spec.md +7 -0
  264. package/vendor-core/docs-content/detection/spill.md +7 -0
  265. package/vendor-core/docs-content/detection/strag.md +5 -0
  266. package/vendor-core/docs-content/detection/tiny.md +4 -0
  267. package/vendor-core/docs-content/detection/util.md +4 -0
  268. package/vendor-core/docs-content/diagrams/aqe-loop.dark.svg +1 -0
  269. package/vendor-core/docs-content/diagrams/aqe-loop.svg +1 -0
  270. package/vendor-core/docs-content/diagrams/broadcast-vs-shuffle.dark.svg +1 -0
  271. package/vendor-core/docs-content/diagrams/broadcast-vs-shuffle.svg +1 -0
  272. package/vendor-core/docs-content/diagrams/cache-lifecycle.dark.svg +1 -0
  273. package/vendor-core/docs-content/diagrams/cache-lifecycle.svg +1 -0
  274. package/vendor-core/docs-content/diagrams/cold-start-timeline.dark.svg +1 -0
  275. package/vendor-core/docs-content/diagrams/cold-start-timeline.svg +1 -0
  276. package/vendor-core/docs-content/diagrams/columnar-layout.dark.svg +1 -0
  277. package/vendor-core/docs-content/diagrams/columnar-layout.svg +1 -0
  278. package/vendor-core/docs-content/diagrams/container-memory.dark.svg +1 -0
  279. package/vendor-core/docs-content/diagrams/container-memory.svg +1 -0
  280. package/vendor-core/docs-content/diagrams/dag-stages.dark.svg +1 -0
  281. package/vendor-core/docs-content/diagrams/dag-stages.svg +1 -0
  282. package/vendor-core/docs-content/diagrams/driver-executor.dark.svg +1 -0
  283. package/vendor-core/docs-content/diagrams/driver-executor.svg +1 -0
  284. package/vendor-core/docs-content/diagrams/duplicate-plan-subtree.dark.svg +1 -0
  285. package/vendor-core/docs-content/diagrams/duplicate-plan-subtree.svg +1 -0
  286. package/vendor-core/docs-content/diagrams/join-strategy.dark.svg +1 -0
  287. package/vendor-core/docs-content/diagrams/join-strategy.svg +1 -0
  288. package/vendor-core/docs-content/diagrams/memory-borrowing.dark.svg +1 -0
  289. package/vendor-core/docs-content/diagrams/memory-borrowing.svg +1 -0
  290. package/vendor-core/docs-content/diagrams/memory-regions.dark.svg +1 -0
  291. package/vendor-core/docs-content/diagrams/memory-regions.svg +1 -0
  292. package/vendor-core/docs-content/diagrams/repartition-vs-coalesce.dark.svg +1 -0
  293. package/vendor-core/docs-content/diagrams/repartition-vs-coalesce.svg +1 -0
  294. package/vendor-core/docs-content/diagrams/retry-escalation-ladder.dark.svg +1 -0
  295. package/vendor-core/docs-content/diagrams/retry-escalation-ladder.svg +1 -0
  296. package/vendor-core/docs-content/diagrams/shuffle-map-reduce.dark.svg +1 -0
  297. package/vendor-core/docs-content/diagrams/shuffle-map-reduce.svg +1 -0
  298. package/vendor-core/docs-content/diagrams/spill-classification.dark.svg +1 -0
  299. package/vendor-core/docs-content/diagrams/spill-classification.svg +1 -0
  300. package/vendor-core/docs-content/diagrams/udf-execution-models.dark.svg +1 -0
  301. package/vendor-core/docs-content/diagrams/udf-execution-models.svg +1 -0
  302. package/vendor-core/docs-content/tuning/broadcast-sizing.md +78 -0
  303. package/vendor-core/docs-content/tuning/cold-start.md +81 -0
  304. package/vendor-core/docs-content/tuning/duplicate-plan-subtree.md +45 -0
  305. package/vendor-core/docs-content/tuning/failures.md +124 -0
  306. package/vendor-core/docs-content/tuning/gc.md +110 -0
  307. package/vendor-core/docs-content/tuning/job-failure-rate.md +101 -0
  308. package/vendor-core/docs-content/tuning/memory-utilization.md +58 -0
  309. package/vendor-core/docs-content/tuning/retry-waste.md +90 -0
  310. package/vendor-core/docs-content/tuning/shuffle.md +154 -0
  311. package/vendor-core/docs-content/tuning/skew.md +123 -0
  312. package/vendor-core/docs-content/tuning/slow-host.md +117 -0
  313. package/vendor-core/docs-content/tuning/small-files.md +99 -0
  314. package/vendor-core/docs-content/tuning/spill.md +114 -0
  315. package/vendor-core/docs-content/tuning/straggler.md +103 -0
  316. package/vendor-core/docs-content/tuning/tiny-tasks.md +94 -0
  317. package/vendor-core/docs-content/tuning/utilization.md +90 -0
  318. package/vendor-core/docs-site-config.js +10 -17
  319. package/vendor-core/efficiency-model.js +7 -13
  320. package/vendor-core/etl-phases.js +3 -5
  321. package/vendor-core/event-handlers.js +232 -134
  322. package/vendor-core/event-schemas.js +48 -114
  323. package/vendor-core/evidence-availability.js +5 -10
  324. package/vendor-core/evidence-report.js +72 -122
  325. package/vendor-core/export-data.js +48 -0
  326. package/vendor-core/finding-action-label.js +4 -10
  327. package/vendor-core/finding-filter-predicate.js +3 -7
  328. package/vendor-core/finding-generic-recommendation.js +112 -0
  329. package/vendor-core/finding-names.js +51 -0
  330. package/vendor-core/format-utils.js +112 -38
  331. package/vendor-core/impact-band.js +18 -24
  332. package/vendor-core/impact-estimator.js +38 -74
  333. package/vendor-core/ingest.js +7 -13
  334. package/vendor-core/job-groups.js +3 -6
  335. package/vendor-core/list-runs.js +278 -0
  336. package/vendor-core/load-vendored.js +6 -12
  337. package/vendor-core/log-header-peek.js +81 -0
  338. package/vendor-core/lz4-block.js +4 -6
  339. package/vendor-core/mcp-server-factory.js +38 -8
  340. package/vendor-core/mcp-tools.js +105 -76
  341. package/vendor-core/model-assembler.js +8 -16
  342. package/vendor-core/occupancy.js +5 -9
  343. package/vendor-core/parser-worker.js +18 -27
  344. package/vendor-core/plan-dot.js +2 -5
  345. package/vendor-core/plan-duration-attribution.js +78 -29
  346. package/vendor-core/plan-graph-model.js +126 -69
  347. package/vendor-core/plan-node-detail.js +31 -17
  348. package/vendor-core/plan-summary.js +19 -8
  349. package/vendor-core/recommendation-rollup.js +35 -39
  350. package/vendor-core/redact.js +72 -16
  351. package/vendor-core/rolling-log-reassembly.js +4 -6
  352. package/vendor-core/run-comparison.js +65 -70
  353. package/vendor-core/scaling-sim.js +5 -7
  354. package/vendor-core/session-snapshot.js +1 -1
  355. package/vendor-core/shs-fetch.js +4 -6
  356. package/vendor-core/shs-load.js +9 -13
  357. package/vendor-core/shs-request.js +1 -1
  358. package/vendor-core/stage-quantiles.js +14 -0
  359. package/vendor-core/types.js +78 -18
  360. package/vendor-core/wasted-core-hours.js +7 -12
@@ -1,33 +1,23 @@
1
1
  import { pathBasename, formatBytes, nsToMs, IMPACT_BAND_ORDER } from './format-utils.js';
2
2
  import { scanRelationId } from './plan-summary.js';
3
- import { computeTotalCores } from './core-count.js';
3
+ import { computePeakConcurrentCores, computePeakConcurrentExecutorCount } from './core-count.js';
4
4
  import { walkPlanTree } from './plan-tree-walk.js';
5
5
  import { computeCoreLocalityRatio } from './core-locality-ratio.js';
6
6
  import { estimateSingleStage, } from './occupancy.js';
7
+ import { isExchangeNode, isBroadcastExchangeNode } from './plan-node-detail.js';
7
8
 
8
9
 
9
10
  const MB = 1024 * 1024;
10
11
  const GB = 1024 * MB;
11
12
  const TB = 1024 * GB;
12
13
 
13
- // ---------------------------------------------------------------------------
14
14
  // Local runtime shapes.
15
15
  //
16
- // `types.ts`'s `Stage`/`SqlExecution`/`SparkAppInfo`/`AppModel` describe the
17
- // *posted* AppModel surface at a level of detail that suits the view layer
18
- // (both carry a catch-all `[key: string]: unknown` index signature for
19
- // anything beyond their few named fields). This file needs the FULL set of
20
- // fields `finalizeStage` (src/stage-quantiles.ts) actually computes and
21
- // `event-handlers.ts`'s stage/sql/app records actually carry, accessed
22
- // directly (arithmetic, comparisons) rather than read-and-display, so an
23
- // index-signature-shaped type would force an `unknown` cast at nearly every
24
- // field access. Every field below is verified against a real access in this
25
- // file (or a helper it calls); none are speculative.
26
- //
27
- // `analyzer.js` (the only real caller of `detect()`, still plain JS) passes
28
- // whatever the parser actually produced, so these types describe reality,
29
- // not a narrowing of some existing stricter type; there is nothing unsound
30
- // about them being independent of `types.ts`'s `Stage`/`SqlExecution`.
16
+ // types.ts's Stage/SqlExecution/SparkAppInfo describe the posted AppModel surface for the view
17
+ // layer (with a catch-all index signature). This file needs the FULL set of fields finalizeStage
18
+ // computes and event-handlers.ts records carry, accessed directly (arithmetic, comparisons), so
19
+ // an index-signature type would force an `unknown` cast at nearly every access. Every field below
20
+ // is verified against a real access here.
31
21
 
32
22
 
33
23
 
@@ -35,9 +25,18 @@ const TB = 1024 * GB;
35
25
 
36
26
 
37
27
 
38
- // Per-executor snapshot from a StageExecutorMetrics event (mergeStageRddInfo/
39
- // onStageExecutorMetrics): a loose bag of Spark's ExecutorMetrics field
40
- // names, only a few of which any detector reads.
28
+
29
+
30
+
31
+
32
+
33
+
34
+
35
+
36
+
37
+
38
+ // Per-executor snapshot from a StageExecutorMetrics event: a loose bag of Spark's
39
+ // ExecutorMetrics field names, only a few of which any detector reads.
41
40
 
42
41
 
43
42
 
@@ -81,6 +80,8 @@ const TB = 1024 * GB;
81
80
 
82
81
 
83
82
 
83
+
84
+
84
85
 
85
86
 
86
87
 
@@ -116,27 +117,19 @@ const TB = 1024 * GB;
116
117
 
117
118
 
118
119
 
119
-
120
-
120
+
121
121
 
122
122
 
123
123
 
124
124
 
125
- // The full context object analyzer.js's `analyze()` builds and passes as
126
- // every stage/sql detect()'s second argument, and as the sole argument for
127
- // 'app' scope (`d.detect(ctx)`). 'config' scope gets a narrower `{ app }`
128
- // (see auditConfig in analyzer.js), typed per-entry below instead of here.
125
+ // The full context analyze() passes as every stage/sql detect()'s second arg, and as the sole
126
+ // arg for 'app' scope. 'config' scope gets a narrower `{ app }` (auditConfig), typed per-entry.
129
127
  //
130
- // `app` is non-nullable here (unlike `AppModel.app: SparkAppInfo | null`):
131
- // `analyze()` only ever runs after a full parse, by which point `app` is
132
- // always populated; the `SparkAppInfo | null` nullability models the
133
- // in-progress-parse window this file never observes. This matches every
134
- // 'app'-scope entry below: most defensively write `ctx.app?.foo` anyway
135
- // (harmless on a non-nullable value), and `coldStart` reads `app.startTime`
136
- // with no guard at all, which only type-checks if `app` is non-nullable.
137
- // `auditConfig`'s separate config-scope target type keeps `app` nullable
138
- // instead, since `auditConfig(appModel.app)` (src/analyzer.js) really can be
139
- // called with a `null` app and every config-scope entry optional-chains it.
128
+ // `app` is typed as non-nullable here (unlike AppModel.app), but incompleteRun, coldStart,
129
+ // utilization, and autoscalingChurn defensively guard against null at runtime to tolerate
130
+ // malformed/incomplete logs. Despite the type annotation, app may be null in edge cases, and
131
+ // these detectors handle it gracefully. auditConfig's config-scope target keeps `app` nullable
132
+ // instead.
140
133
 
141
134
 
142
135
 
@@ -145,17 +138,13 @@ const TB = 1024 * GB;
145
138
 
146
139
 
147
140
 
148
-
149
-
150
-
151
-
152
-
141
+
142
+
153
143
 
154
144
 
155
145
 
156
- // `auditConfig(app)` (src/analyzer.js) calls every 'config'-scope entry's
157
- // `detect({ app })` directly with whatever `appModel.app` is at call time
158
- // (`SparkAppInfo | null`), independent of the full `DetectorCtx` above.
146
+ // auditConfig(app) calls every config-scope detect({ app }) with whatever appModel.app is
147
+ // (SparkAppInfo | null), independent of DetectorCtx.
159
148
 
160
149
 
161
150
 
@@ -167,8 +156,7 @@ const TB = 1024 * GB;
167
156
 
168
157
 
169
158
 
170
- // sparkDoctor SpillPressureDetector (5a) + SpillSkewDetector (5b).
171
- // Returns { magnitude } | null. Magnitude ∈ 'severe'|'high'|'medium'.
159
+ // SpillPressureDetector (5a) + SpillSkewDetector (5b).
172
160
  function computeSpillMagnitude(
173
161
  stage ,
174
162
  t ,
@@ -195,15 +183,9 @@ function pickDominantReason(reasons )
195
183
  return [...reasons].sort((a, b) => b.count - a.count)[0].reason;
196
184
  }
197
185
 
198
- // Shared by every scope:'sql' detector below. `sql.get(executionId).stageIds`
199
- // (src/model-assembler.js) is always empty because parser-worker.js never
200
- // populates it, so stage linkage is derived the other way round, from each
201
- // stage's own (correctly populated) `sqlExecutionId`. Parameter type is
202
- // intentionally the minimal shape needed (not `DetectorStage`): callers
203
- // outside this file (Topbar.tsx, plan-node-detail.ts, plan-graph-model.ts)
204
- // pass `appModel.stages`, typed `Map<StageId, Stage>` per types.ts, which
205
- // carries `id`/`sqlExecutionId` but not this file's fuller `DetectorStage`
206
- // shape.
186
+ // Shared by every scope:'sql' detector. sql.get(id).stageIds is always empty (parser-worker
187
+ // never populates it), so stage linkage is derived from each stage's own sqlExecutionId.
188
+ // Minimal param shape (not DetectorStage): external callers pass Map<StageId, Stage>.
207
189
  export function stageIdsForSqlExec(
208
190
  executionId ,
209
191
  stages ,
@@ -213,38 +195,23 @@ export function stageIdsForSqlExec(
213
195
  return out;
214
196
  }
215
197
 
216
- // Shared by the three Plan Advisor detectors below (duplicatePlanSubtree,
217
- // smallFiles, broadcastSizing): union the given nodes' own `stageIds`, or
218
- // fall back to the whole execution's stage set when none of them have any
219
- // coverage. A finding never partially blends a narrowed set with the
220
- // execution-wide one (see docs/architecture.md's plan-node-to-stage mapping
221
- // section). `fallback` is a param, not computed here, so callers can compute
222
- // `stageIdsForSqlExec` once per `detect()` call and reuse it across every
223
- // finding in that call instead of re-walking `stages` per finding.
198
+ // Shared by the three Plan Advisor detectors: union the given nodes' own stageIds, or fall back
199
+ // to the whole execution's stage set when none have coverage (never a partial blend). `fallback`
200
+ // is a param so callers compute stageIdsForSqlExec once per detect() and reuse it per finding.
224
201
  export function unionStageIds(nodes , fallback ) {
225
202
  const union = new Set ();
226
203
  for (const node of nodes) for (const sid of node.stageIds ?? []) union.add(sid);
227
204
  return union.size > 0 ? [...union].sort((a, b) => a - b) : fallback;
228
205
  }
229
206
 
230
- // Bottom-up shape computation for duplicate-subtree detection (sparkDoctor
231
- // idea, src/detectors.js `duplicatePlanSubtree`) and for the cachingOpportunity
232
- // composite (join/union) detector's anchor-fingerprint contract. `size` is the
233
- // node count of the subtree rooted at each node; the default fingerprint
234
- // encodes operator name + sorted metric *names* (never values, per spec) +
235
- // children fingerprints in original order, so two subtrees with the same shape
236
- // but different row counts/literals still collide, which is the point (we
237
- // have no expr/plan/codegen IDs to strip in the first place, since `metrics`
238
- // never carried them).
207
+ // Bottom-up shape computation for duplicate-subtree detection and the
208
+ // cachingOpportunity composite detector. `size` is the subtree node count; the default
209
+ // fingerprint encodes operator name + sorted metric NAMES (never values, per spec) + child
210
+ // fingerprints, so two subtrees with the same shape but different values still collide.
239
211
  //
240
- // `opts.includeDetail` (default false) folds `root`'s own `detail` text
241
- // (normalized via `opts.normalizeDetail`) into ONLY `root`'s fingerprint,
242
- // never into any recursive child call's fingerprint, regardless of `opts`.
243
- // This is what lets a caller ask "does this specific node's own detail +
244
- // structural shape match another node's", without descendant filter/scan
245
- // detail (literals, paths) ever entering the comparison. The existing
246
- // `duplicatePlanSubtree` call site passes no options: identical behavior,
247
- // zero regression to its existing tests.
212
+ // opts.includeDetail (default false) folds root's own detail (via opts.normalizeDetail) into
213
+ // ONLY root's fingerprint, never a child's: lets a caller match "this node's own detail + shape"
214
+ // without descendant scan detail entering the comparison.
248
215
 
249
216
 
250
217
  export function computePlanShapes(
@@ -256,9 +223,22 @@ export function computePlanShapes(
256
223
  const allNodes = [];
257
224
  function visit(node , isRoot ) {
258
225
  allNodes.push(node);
259
- const childShapes = (node.children ?? []).map((c) => visit(c, false));
226
+ // A read half wrapping a write half (the Exchange split from
227
+ // resolvePlanTree, see event-handlers.ts) is one logical Spark operator
228
+ // for subtree-shape purposes. Without this, every real Exchange in a
229
+ // matched subtree would count twice, inflating duplicatePlanSubtree's
230
+ // reported subtreeSize and shifting its groupIndex-derived findingIds
231
+ // for an otherwise-unchanged plan. Skip straight through the write
232
+ // wrapper: size comes from its real children, and metricNames from its
233
+ // real metrics (the read half's own metrics are always empty), so
234
+ // fingerprint distinctiveness between different real Exchanges is
235
+ // preserved too.
236
+ const writeHalf = node.exchangeRole === 'read' ? node.children[0] : null;
237
+ const realChildren = writeHalf ? writeHalf.children : (node.children ?? []);
238
+ const realMetrics = writeHalf ? writeHalf.metrics : node.metrics;
239
+ const childShapes = realChildren.map((c) => visit(c, false));
260
240
  const size = 1 + childShapes.reduce((sum, c) => sum + c.size, 0);
261
- const metricNames = (node.metrics ?? []).map((m) => m.name).sort().join(',');
241
+ const metricNames = (realMetrics ?? []).map((m) => m.name).sort().join(',');
262
242
  const childFingerprints = childShapes.map((c) => c.fingerprint).join(',');
263
243
  const fingerprint = isRoot && includeDetail
264
244
  ? `${node.name}[${metricNames}]<${normalize(node.detail ?? '')}>{${childFingerprints}}`
@@ -271,16 +251,10 @@ export function computePlanShapes(
271
251
  return { shapeOf, allNodes };
272
252
  }
273
253
 
274
- // Normalizes an anchor join/union node's own `detail` text for the
275
- // cachingOpportunity composite detector's structural fingerprint
276
- // (computePlanShapes's opts.includeDetail path). Strips per-analysis
277
- // numbering noise that is never part of a join/union's logical identity:
278
- // expr ids, plan/codegen-stage ids, and AQE's runtime BuildLeft/BuildRight
279
- // broadcast-side choice (which can flip between executions of the logically
280
- // identical join based on runtime stats); it then canonicalizes commutative
281
- // equality operand order so `A.x = B.y` and `B.y = A.x` collide. Everything
282
- // else (join type, columns, literal values) is kept: that's the
283
- // semantically meaningful part a leaf-relation-set identity would miss.
254
+ // Normalizes an anchor join/union node's detail for the cachingOpportunity composite
255
+ // fingerprint. Strips per-analysis numbering (expr ids, plan/codegen ids) and AQE's runtime
256
+ // BuildLeft/BuildRight choice (can flip between runs), then canonicalizes commutative equality
257
+ // operand order so `A.x = B.y` and `B.y = A.x` collide. Join type, columns, literals are kept.
284
258
  export function normalizeDetail(detail ) {
285
259
  let s = detail
286
260
  .replace(/#\d+L?/g, '')
@@ -296,34 +270,21 @@ export function normalizeDetail(detail ) {
296
270
 
297
271
  const JOIN_NAME_RE = /Join/i;
298
272
 
299
- // Structural operator kind for cachingOpportunity's composite detection:
300
- // 'join' covers every Spark join physical operator (SortMergeJoin,
301
- // BroadcastHashJoin, ShuffledHashJoin, BroadcastNestedLoopJoin, the same set
302
- // plan-summary.js's visitJoin recognizes); 'union' is Spark's exact `Union`
303
- // node name. CartesianProduct is deliberately excluded (out of scope, same
304
- // as plan-summary.js's join handling).
273
+ // Structural operator kind for cachingOpportunity's composite detection: 'join' covers every
274
+ // Spark join physical operator; 'union' is Spark's exact `Union` node. CartesianProduct is
275
+ // deliberately excluded (out of scope, as in plan-summary.ts).
305
276
  export function planOperatorKind(name ) {
306
277
  if (JOIN_NAME_RE.test(name)) return 'join';
307
278
  if (name === 'Union') return 'union';
308
279
  return null;
309
280
  }
310
281
 
311
- // Single bottom-up pass over a resolved planTree producing one composite
312
- // candidate per join/union node, in O(n) total (see the design doc's
313
- // "Compute cost" section; this deliberately does NOT call
314
- // computePlanShapes per join/union node, since subtrees overlap and that
315
- // would be O(n·k) on multi-way star/snowflake joins). For every node it
316
- // computes the plain (detail-free) fingerprint once (identical formula to
317
- // computePlanShapes's default path) and merges each subtree's leaf-relation
318
- // byte map (via scanRelationId, same identity cachingOpportunity's existing
319
- // leaf aggregation uses) bottom-up. When a node is a join/union, its anchor
320
- // fingerprint folds only its OWN normalized detail (per computePlanShapes's
321
- // opts.includeDetail contract), computed here inline, in O(1), from the
322
- // node's own detail plus its already-computed child fingerprints, not via a
323
- // nested computePlanShapes call. `ancestorNodes` (strict ancestors, root
324
- // first) is threaded down for free via the recursion's own call stack, so
325
- // later nested-composite dedupe (cachingOpportunity.detect()) doesn't need a
326
- // separate tree walk to determine containment.
282
+ // Single bottom-up O(n) pass producing one composite candidate per join/union node. Deliberately
283
+ // does NOT call computePlanShapes per node (subtrees overlap, that would be O(n·k) on
284
+ // star/snowflake joins): computes the plain fingerprint once and merges each subtree's
285
+ // leaf-relation byte map bottom-up. A join/union's anchor fingerprint folds only its OWN
286
+ // normalized detail, inline in O(1). `ancestorNodes` threads down via the call stack so
287
+ // nested-composite dedupe needs no separate tree walk.
327
288
 
328
289
 
329
290
 
@@ -374,17 +335,10 @@ export function findCompositeCandidates(root ) {
374
335
  }
375
336
 
376
337
 
377
- // First scanned table/relation identity found in a subtree (pre-order, per
378
- // walkPlanTree's contract), or null when the subtree touches no named
379
- // relation. Surfaced on duplicatePlanSubtree findings (as `sampleRelation`,
380
- // see findDuplicateSubtrees below) so a user can tell apart two same-shaped
381
- // duplicate groups that scan different tables (real-log bug: two unrelated
382
- // "BroadcastExchange over a Project/Filter/Scan" patterns, one per dimension
383
- // table, produced byte-identical findings). This is best-effort/informational
384
- // only, not a uniqueness guarantee: it's null for scan-less subtrees (JDBC/
385
- // Kafka/LocalRelation) and can coincide when two groups share their first-
386
- // encountered leaf. See findDuplicateSubtrees's groupIndex for the actual
387
- // discriminator findingId() relies on.
338
+ // First scanned relation identity in a subtree (pre-order), or null when none. Surfaced on
339
+ // duplicatePlanSubtree findings as `sampleRelation` so a user can tell apart same-shaped groups
340
+ // that scan different tables. Best-effort only: null for scan-less subtrees, can coincide across
341
+ // groups; findDuplicateSubtrees's groupIndex is the actual discriminator findingId relies on.
388
342
  function firstLeafRelationId(node ) {
389
343
  let found = null;
390
344
  walkPlanTree(node, (n) => {
@@ -394,14 +348,10 @@ function firstLeafRelationId(node ) {
394
348
  return found;
395
349
  }
396
350
 
397
- // Groups nodes by fingerprint, keeping only groups of size >= minOccurrences
398
- // whose shared subtree size is >= minSubtreeSize. De-overlap: processes
399
- // candidate groups largest-subtree-first, and once a group is accepted every
400
- // node inside each of its matched occurrences is marked "claimed" so a
401
- // smaller, fully-nested duplicate group inside an already-accepted match is
402
- // dropped (a 5-node duplicate should not also emit findings for its 3-node
403
- // sub-subtrees); occurrences of a smaller group that fall OUTSIDE any
404
- // accepted larger match still count normally.
351
+ // Groups nodes by fingerprint, keeping groups of size >= minOccurrences whose subtree size is
352
+ // >= minSubtreeSize. De-overlap: process largest-subtree-first, and once a group is accepted mark
353
+ // every node in its matches "claimed" so a smaller fully-nested duplicate is dropped; occurrences
354
+ // of a smaller group OUTSIDE any accepted match still count.
405
355
 
406
356
 
407
357
 
@@ -417,7 +367,12 @@ export function findDuplicateSubtrees(
417
367
  { minSubtreeSize, minOccurrences } ,
418
368
  ) {
419
369
  const { shapeOf, allNodes } = computePlanShapes(root);
420
- const eligible = allNodes.filter(n => shapeOf.get(n) .size >= minSubtreeSize);
370
+ // Defensive, not load-bearing: computePlanShapes's visit() already skips
371
+ // straight through a read node to its write half's real children, so no
372
+ // write-half node is ever pushed into allNodes in the first place, this
373
+ // filter can structurally never exclude anything. Kept in case that
374
+ // invariant ever changes upstream.
375
+ const eligible = allNodes.filter(n => shapeOf.get(n) .size >= minSubtreeSize && n.exchangeRole !== 'write');
421
376
 
422
377
  const groups = new Map ();
423
378
  for (const n of eligible) {
@@ -445,14 +400,10 @@ export function findDuplicateSubtrees(
445
400
  rootName: unclaimed[0].name,
446
401
  subtreeSize: shapeOf.get(unclaimed[0]) .size,
447
402
  occurrences: unclaimed.length,
448
- isExchangeRoot: /Exchange/i.test(unclaimed[0].name),
403
+ isExchangeRoot: isExchangeNode(unclaimed[0]),
449
404
  sampleRelation: firstLeafRelationId(unclaimed[0]),
450
- // Deterministic position within this execution's group list. `sampleRelation`
451
- // is best-effort (null when the subtree touches no named catalog relation, e.g.
452
- // JDBC/Kafka/LocalRelation sources, or identical when two groups happen to share
453
- // their first-encountered leaf) and is NOT sufficient on its own to guarantee two
454
- // structurally-distinct groups get distinct finding ids; groupIndex is the actual
455
- // uniqueness guarantee findingId() relies on.
405
+ // Deterministic position within this execution's group list: the actual uniqueness
406
+ // guarantee findingId relies on, since sampleRelation is best-effort (null or coincident).
456
407
  groupIndex: results.length,
457
408
  nodes: unclaimed,
458
409
  });
@@ -460,21 +411,16 @@ export function findDuplicateSubtrees(
460
411
  return results;
461
412
  }
462
413
 
463
- // Exact metric names Spark emits: verified against a real SQLExecutionStart
464
- // event's sparkPlanInfo in examples/big-application_1777489669889_51251_1.
465
- // NOTE: the write-side byte metric is "written output", not "size of written
466
- // files" as an earlier draft of this detector's spec assumed.
414
+ // Exact metric names Spark emits, verified against a real SQLExecutionStart's sparkPlanInfo.
415
+ // The write-side byte metric is "written output", not "size of written files".
467
416
  const FILES_READ_COUNT = 'number of files read';
468
417
  const FILES_READ_BYTES = 'size of files read';
469
418
  const FILES_WRITTEN_COUNT = 'number of written files';
470
419
  const FILES_WRITTEN_BYTES = 'written output';
471
420
 
472
- // Byte size of a join-side subtree for broadcast sizing (dataflint
473
- // JoinToBroadcastAlert). Stops descending the instant a node carries a
474
- // "data size" metric: that node's value already aggregates everything
475
- // beneath it (e.g. an Exchange's "data size" already reflects everything it
476
- // shuffled), so summing further down would double-count. Only recurses into
477
- // children when the current node carries no such metric.
421
+ // Byte size of a join-side subtree for broadcast sizing. Stops
422
+ // descending at a node with a "data size" metric: that value already aggregates everything
423
+ // beneath it, so summing further would double-count. Only recurses when no such metric.
478
424
  function sumBoundarySize(node ) {
479
425
  const m = (node.metrics ?? []).find(x => x.name === 'data size');
480
426
  if (m) return m.value;
@@ -483,10 +429,8 @@ function sumBoundarySize(node ) {
483
429
  return sum;
484
430
  }
485
431
 
486
- // Nodes that actually fed sumBoundarySize's total for this subtree: mirrors
487
- // its recursion exactly (same "data size" metric check, same stop condition)
488
- // so the implicated stageIds line up with the size that was actually
489
- // compared, however deep that turns out to live.
432
+ // Nodes that fed sumBoundarySize's total: mirrors its recursion exactly so implicated stageIds
433
+ // line up with the size actually compared.
490
434
  function boundarySizeContributors(node ) {
491
435
  const m = (node.metrics ?? []).find((x) => x.name === 'data size');
492
436
  if (m) return [node];
@@ -516,10 +460,8 @@ function maxMedianRatio(
516
460
 
517
461
 
518
462
 
519
- // Machine-readable detector metadata for the evidence report: type, version,
520
- // scope, thresholds, and doc anchor per entry (no `detect` closure). Lets a
521
- // portable report record exactly which detector + threshold set produced each
522
- // finding, so evidence stays reproducible as detectors evolve.
463
+ // Machine-readable detector metadata for the evidence report (no `detect` closure), so a
464
+ // portable report records which detector + thresholds produced each finding.
523
465
  export function detectorCatalog() {
524
466
  return DETECTORS.map((d) => ({
525
467
  type: d.type,
@@ -530,21 +472,11 @@ export function detectorCatalog() {
530
472
  }));
531
473
  }
532
474
 
533
- // True task-duration skew ratio for a stage: P95/median once there are enough
534
- // tasks to trust a P95 estimate, otherwise max/median. Returns null when the
535
- // stage has no measurable median (p50 === 0). Exported so the CLI budget gate
536
- // (src/cli/budgets.ts) can recompute the same ratio the detector uses, rather
537
- // than reading the detector's own findings (which are floored at ratioWarn).
538
- //
539
- // Fields widened to optional: the `skew` detector below only ever calls this
540
- // with an already-finalized `DetectorStage` (all four always numeric by the
541
- // time `analyze()` runs), but `cli/budgets.ts`'s `checkSkew` calls it
542
- // directly against `AppModel.stages`: real `Stage` records for a stage that
543
- // never received a `StageCompleted` event (e.g. an unfinished run) genuinely
544
- // lack these fields (see `types.ts`'s `Stage`). The `as number` casts below
545
- // preserve the original behavior byte-for-byte: dividing through an absent
546
- // field still naturally produces `NaN` (as it always has for untyped JS
547
- // callers), rather than adding a new guard that would change the result.
475
+ // True task-duration skew ratio: P95/median once enough tasks to trust P95, else max/median.
476
+ // Null when no measurable median (p50 === 0). Exported so cli/budgets.ts recomputes the same
477
+ // ratio rather than reading findings floored at ratioWarn.
478
+ // Fields optional: budgets.ts calls this against raw AppModel.stages, whose stages may lack
479
+ // these fields (unfinished run). The `as number` casts keep behavior: an absent field yields NaN.
548
480
  export function computeSkewRatio(
549
481
  stage ,
550
482
  minTasksForP95 ,
@@ -556,18 +488,10 @@ export function computeSkewRatio(
556
488
  : { ratio: (max ) / (p50 ), metric: 'max/median' };
557
489
  }
558
490
 
559
- // Absolute-magnitude floor expressed as a % of total app runtime rather than
560
- // a fixed ms constant (mirrors computeSpillMagnitude's/slowHost's ratio+floor
561
- // pattern above/below): a skew/straggler/taskStageSkew ratio computed on a
562
- // handful of milliseconds is noise in a run that took hours, but the same
563
- // absolute waste is real in a run that only took seconds; a fixed-ms floor
564
- // can't scale between those. Used by the `skew` and `straggler` entries
565
- // below, each gated via `clippedWasteMs` (below) against the *same
566
- // occupancy-clipped* wall-clock figure
567
- // src/impact-estimator.ts displays as that finding's savings, not the raw
568
- // pre-clip delta, which can stay large after clipping collapses the
569
- // recoverable time to near zero (the stage's own longest task already
570
- // accounts for nearly all of its wall-clock window).
491
+ // Absolute-magnitude floor as a % of app runtime, not a fixed ms constant: a skew/straggler
492
+ // ratio on a few ms is noise in an hours-long run but real in a seconds-long one; a fixed-ms
493
+ // floor can't scale. Used by skew/straggler, gated via clippedWasteMs against the same
494
+ // occupancy-clipped figure impact-estimator.ts displays as savings.
571
495
  // NOT SOURCED: floor percentages are our own noise floor, unvalidated.
572
496
  function computeAppDurationMs(ctx ) {
573
497
  const app = ctx?.app;
@@ -576,34 +500,22 @@ function computeAppDurationMs(ctx ) {
576
500
  return durationMs > 0 ? durationMs : null;
577
501
  }
578
502
 
579
- // Unknown app timing (computeAppDurationMs returned null) never suppresses a
580
- // finding; it just skips the floor gate, preserving prior ratio-only
581
- // behavior when total runtime can't be computed.
503
+ // Unknown app timing never suppresses a finding; it just skips the floor gate.
582
504
  function meetsRuntimeFloor(wasteMs , appDurationMs , floorPct ) {
583
505
  return appDurationMs == null || wasteMs >= appDurationMs * floorPct;
584
506
  }
585
507
 
586
- // Runs a raw waste delta through the same ceiling/occupancy clip
587
- // src/impact-estimator.ts's singleStageImpact applies before display, so the
588
- // runtime floor above is checked against a stage's actual recoverable
589
- // wall-clock time rather than a delta that a physical floor (the stage's own
590
- // longest task) may leave almost entirely unrecoverable. Falls back to the
591
- // raw delta when occupancy data isn't available for this stage (ctx omitted,
592
- // or the stage was excluded from the occupancy sweep for having <= 0
593
- // duration), same as the pre-existing ratio-only behavior for unknown app
594
- // timing above.
508
+ // Runs a raw waste delta through the same occupancy clip impact-estimator.ts applies before
509
+ // display, so the runtime floor checks recoverable wall-clock, not a delta a physical floor
510
+ // leaves unrecoverable. Falls back to the raw delta when occupancy data is unavailable.
595
511
  function clippedWasteMs(wasteMs , stageId , ctx ) {
596
512
  if (!ctx) return wasteMs;
597
513
  const est = estimateSingleStage(wasteMs, stageId, ctx.stages , ctx.occupancy);
598
514
  return est ? est.wallClock.high : wasteMs;
599
515
  }
600
516
 
601
- // cacheUtilization's per-RDD copy (private to this file), following the
602
- // existing pattern of small per-detector formatting helpers (e.g.
603
- // pickDominantReason above). Both variants share the same confidence/
604
- // validationRequired text: the ratio is a point-in-time storage snapshot
605
- // from stage-submission events (src/event-handlers.js's mergeStageRddInfo),
606
- // not a runtime block-access read-count.
517
+ // Shared by cacheUtilization's two variants: the ratio is a point-in-time storage snapshot from
518
+ // stage-submission events, not a runtime block-access read-count.
607
519
  const CACHE_UTILIZATION_VALIDATION =
608
520
  "This ratio is a point-in-time storage snapshot from stage-submission events, not a runtime read-count. Confirm against the Spark UI's Storage tab before acting.";
609
521
 
@@ -638,19 +550,11 @@ function diskSpilloverFinding(rdd , diskRatio , impactBan
638
550
  };
639
551
  }
640
552
 
641
- // Entry shape for every item in DETECTORS. `TTarget` stays `unknown` at the
642
- // array level (kept as the default, never instantiated per-scope): `detect`'s
643
- // real first-argument shape varies by `scope` (a `DetectorStage` for
644
- // 'stage', a `DetectorSqlExec` for 'sql', a `DetectorCtx` for 'app', or a
645
- // narrower `{ app }` for 'config'; see analyzer.js's `analyze`/`auditConfig`,
646
- // the only real callers), and unifying those four into one `TTarget` would
647
- // need either an unsound cast or a discriminated-union-of-detectors redesign
648
- // this migration task doesn't ask for. Each entry below still gets a
649
- // precisely-typed `detect` by annotating its own `target`/`ctx` parameters
650
- // directly: object-literal methods (this `detect(target) {}` shorthand, not
651
- // an arrow function assigned to a property) are checked bivariantly against
652
- // an interface's method parameter types, so a narrower, concrete annotation
653
- // here does not conflict with `Detector`'s `unknown` declaration.
553
+ // Entry shape for every DETECTORS item. TTarget stays `unknown` at the array level: detect's
554
+ // real first-arg varies by scope (DetectorStage/DetectorSqlExec/DetectorCtx/{ app }), and
555
+ // unifying them would need an unsound cast or a discriminated-union redesign. Each entry gets a
556
+ // precise detect by annotating its own params: object-literal method params are checked
557
+ // bivariantly, so a narrower annotation here doesn't conflict with the `unknown` declaration.
654
558
 
655
559
 
656
560
 
@@ -658,14 +562,11 @@ function diskSpilloverFinding(rdd , diskRatio , impactBan
658
562
 
659
563
 
660
564
 
661
-
662
-
663
-
664
-
665
-
565
+
666
566
 
667
567
 
668
568
 
569
+
669
570
 
670
571
 
671
572
 
@@ -678,9 +579,8 @@ function diskSpilloverFinding(rdd , diskRatio , impactBan
678
579
  // *Ratio = multiplicative factor
679
580
  // *Share/*Rate/*Util = 0–1 fraction (normalized)
680
581
 
681
- // `straggler`'s own noise-floor thresholds (NOT SOURCED: unvalidated),
682
- // exported so src/impact-band.ts can reuse the same figures as its global
683
- // impact-band floor instead of hand-copying the literals.
582
+ // straggler's noise-floor thresholds (NOT SOURCED: unvalidated), exported so impact-band.ts
583
+ // reuses the same figures instead of hand-copying.
684
584
  export const STRAGGLER_FLOOR_PCT_WARN = 0.005;
685
585
  export const STRAGGLER_FLOOR_PCT_CRIT = 0.02;
686
586
 
@@ -700,9 +600,8 @@ export const DETECTORS = [
700
600
  if (result === null) return null;
701
601
  const { ratio, metric } = result;
702
602
  if (ratio <= this.thresholds.ratioWarn) return null;
703
- // Same absolute delta src/impact-estimator.ts's 'skew' case reports as
704
- // this finding's savings; clipped the same way before the floor check
705
- // so the gate agrees with what's actually displayed.
603
+ // Same absolute delta impact-estimator.ts's 'skew' case reports as savings; clipped the
604
+ // same way before the floor check so the gate agrees with what's displayed.
706
605
  const wasteMs = Math.max(0, metric === 'P95/median' ? stage.taskDurationP95 - stage.taskDurationP50 : stage.taskDurationMax - stage.taskDurationP50);
707
606
  const appDurationMs = computeAppDurationMs(ctx);
708
607
  const floorWasteMs = clippedWasteMs(wasteMs, stage.id, ctx);
@@ -712,6 +611,8 @@ export const DETECTORS = [
712
611
  type: 'skew', stageId: stage.id,
713
612
  impactBand: 'warning',
714
613
  metric, value,
614
+ confidence: 'low',
615
+ validationRequired: 'The 0.5% runtime-floor percentage that gates this finding is our own noise floor, unvalidated: no external tool publishes an equivalent metric to calibrate against. Confirm against known-good/known-bad real logs before trusting the impact-band split.',
715
616
  recommendation: `Task duration ratio (${metric}) is ${value}×: for join-driven skew, enable AQE skew-join handling (spark.sql.adaptive.skewJoin.enabled); otherwise salt the key or repartition on a better key to reduce task skew.`,
716
617
  };
717
618
  },
@@ -753,11 +654,9 @@ export const DETECTORS = [
753
654
  });
754
655
  }
755
656
  }
756
- // TaskStageSkew: straggler cost vs stage wall-clock. Skip near-zero duration.
757
- // Always info, like its lowParallelism/dataExplosion siblings above: satisfying
758
- // this trigger mathematically forces the occupancy-clipped wall-clock estimate to
759
- // exactly zero on every firing (see impact-estimator.ts's costOnly branch below),
760
- // so there is no wall-clock-backed impact-band tier left to gate on.
657
+ // TaskStageSkew: straggler cost vs stage wall-clock. Skip near-zero duration. Always info
658
+ // like its siblings: this trigger forces the occupancy-clipped estimate to exactly zero on
659
+ // every firing, so there's no wall-clock-backed tier left to gate on.
761
660
  const stageDurationMs = (stage.completedAt ?? 0) - (stage.submittedAt ?? 0);
762
661
  if (stageDurationMs > 0) {
763
662
  const ratio = stage.taskDurationMax / stageDurationMs;
@@ -808,9 +707,8 @@ export const DETECTORS = [
808
707
  const out = [];
809
708
  const { shuffleReadP50: p50, shuffleReadMax: max, shuffleReadBytes: total, taskCount } = stage;
810
709
  if (max > this.thresholds.skewRatio * p50 && max > this.thresholds.skewFloorBytes) {
811
- // p50 can be 0 (more than half the shuffle partitions empty): a
812
- // ratio against zero renders the literal string "Infinity×", so fall
813
- // back to median-free phrasing instead of dividing by p50.
710
+ // p50 can be 0 (over half the shuffle partitions empty): a ratio against zero renders
711
+ // "Infinity×", so fall back to median-free phrasing.
814
712
  const ratioText = p50 > 0
815
713
  ? `${Math.round(max / p50 * 10) / 10}× the median (${formatBytes(p50)})`
816
714
  : `far larger than the median (${formatBytes(p50)}, effectively empty)`;
@@ -828,6 +726,10 @@ export const DETECTORS = [
828
726
  });
829
727
  }
830
728
  if (max >= this.thresholds.maxPartBytes) {
729
+ // Fixed 'critical': an OOM/crash-risk safety signal, not a time-waste one. Exempted in
730
+ // impact-band.ts's deriveImpactBand from the wall-clock-based overwrite every other
731
+ // finding here gets, so a long-running job can't demote an active crash risk to 'info'
732
+ // just because the modeled time savings are a small fraction of total runtime.
831
733
  out.push({
832
734
  type: 'partitionSizing', stageId: stage.id, impactBand: 'critical',
833
735
  rule: 'maxPartitionTooBig', metric: 'shuffleReadMax', value: max,
@@ -864,18 +766,19 @@ export const DETECTORS = [
864
766
  {
865
767
  type: 'gc', scope: 'stage', order: 50, fixEffort: 'config', version: 1,
866
768
  docAnchor: '#bottleneck-gc',
769
+ confidence: 'low',
770
+ validationRequired: 'The 10-second minimum-runtime floor that gates this finding is our own noise floor, unvalidated: no external tool publishes an equivalent metric to calibrate against. Confirm against known-good/known-bad real logs before trusting the impact-band split.',
867
771
  thresholds: {
868
772
  warnPct100: 10,
869
- // Descending tier: Dr. Elephant ExecutorGcHeuristic, ported as-is.
773
+ // Descending tier: ExecutorGcHeuristic, ported as-is.
870
774
  lowInfoPct100: 5,
871
- // NOT SOURCED: our own noise floor so a stage that barely ran (gcPct
872
- // near 0 or wildly inflated by a tiny denominator) does not flag,
873
- // in either direction.
775
+ // NOT SOURCED: our own noise floor so a stage that barely ran doesn't flag either direction.
874
776
  minRunTimeMs: 10000,
875
777
  },
876
778
  detect(
877
779
 
878
780
 
781
+
879
782
 
880
783
  stage ,
881
784
  ) {
@@ -887,6 +790,7 @@ export const DETECTORS = [
887
790
  type: 'gc', stageId: stage.id,
888
791
  impactBand: 'warning',
889
792
  metric: 'gcPct', value,
793
+ confidence: this.confidence, validationRequired: this.validationRequired,
890
794
  recommendation: `GC consumed ${value}% of executor run time: reduce object creation, use primitive types, avoid UDFs, increase executor memory.`,
891
795
  };
892
796
  }
@@ -898,6 +802,7 @@ export const DETECTORS = [
898
802
  type: 'gc', stageId: stage.id, direction: 'low',
899
803
  impactBand: 'info',
900
804
  metric: 'gcPct', value,
805
+ confidence: this.confidence, validationRequired: this.validationRequired,
901
806
  recommendation: `GC consumed only ${value}% of executor run time: memory may be over-provisioned; consider reducing spark.executor.memory for cost savings.`,
902
807
  };
903
808
  }
@@ -909,12 +814,9 @@ export const DETECTORS = [
909
814
  docAnchor: '#bottleneck-slow-host',
910
815
  thresholds: {
911
816
  minHosts: 3, minTasks: 15, ratioWarn: 2.0, minShare: 0.20, shareWarn: 0.75, taskShareWarn: 0.50, ratioTiers: [1.33, 1.78, 3.16, 10],
912
- // Absolute-magnitude floors (mirrors computeSpillMagnitude's ratio+floor
913
- // pattern): on short stages, sub-second/sub-64MB differences between
914
- // hosts or executors produce huge ratios that are pure noise, not a
915
- // real slow-host problem. 1s is well above typical per-task scheduling
916
- // jitter but well below the tens-of-seconds+ means genuine slow-host
917
- // stages exhibit; 64MB mirrors the spill detector's disk-skew floor.
817
+ // Absolute-magnitude floors (mirrors computeSpillMagnitude's ratio+floor pattern): on short
818
+ // stages, sub-second/sub-64MB host differences produce huge noise ratios. 1s is above
819
+ // per-task jitter but below genuine slow-host stages; 64MB mirrors the spill disk-skew floor.
918
820
  floorMs: 1000, floorBytes: 64 * MB,
919
821
  },
920
822
  detect(
@@ -946,7 +848,7 @@ export const DETECTORS = [
946
848
  // `value` is a ratio; the estimator needs the absolute per-host mean.
947
849
  hostMeanMs: h.mean,
948
850
  host: h.host, hostTaskShare: Math.round(share * 100) / 100,
949
- recommendation: `Check executor logs for ${h.host}: possible bad node, disk pressure, or NUMA misalignment.`,
851
+ recommendation: `${h.host} may just hold data locality for its tasks or carry one heavy stage, not necessarily a hardware fault: check what it was running, and consider enabling spark.speculation to relaunch a lagging task automatically.`,
950
852
  });
951
853
  }
952
854
  }
@@ -1012,10 +914,8 @@ export const DETECTORS = [
1012
914
  type: 'stageSlowness', scope: 'stage', order: 65, fixEffort: 'code', version: 2,
1013
915
  docAnchor: '#bottleneck-stage-slowness',
1014
916
  thresholds: { infoMin: 15 },
1015
- // Cross-detector suppression (new pattern; see "Detector contract" in
1016
- // docs-site/contributor-guide/architecture/detector-contract.md).
1017
- // Requires this entry to be declared AFTER slowHost in DETECTORS so
1018
- // slowHost findings are already in `out`.
917
+ // Cross-detector suppression (see "Detector contract" in detector-contract.md). Requires this
918
+ // entry to be declared AFTER slowHost in DETECTORS so slowHost findings are already in `out`.
1019
919
  suppressWhen(finding, out) {
1020
920
  return out.some(o => o.type === 'slowHost' && o.stageId === finding.stageId);
1021
921
  },
@@ -1023,9 +923,8 @@ export const DETECTORS = [
1023
923
 
1024
924
  stage ,
1025
925
  ) {
1026
- // Basis is real wall-clock stage duration, not per-executor average
1027
- // (Decision 7); Task 11's impact-estimator formula reuses this exact
1028
- // stageDurationMs computation.
926
+ // Basis is real wall-clock stage duration, not per-executor average; the impact-estimator
927
+ // formula reuses this exact stageDurationMs computation.
1029
928
  const stageDurationMs = (stage.completedAt ?? 0) - (stage.submittedAt ?? 0);
1030
929
  if (!(stageDurationMs > 0)) return null;
1031
930
  const durationMinutes = stageDurationMs / 60000;
@@ -1036,7 +935,7 @@ export const DETECTORS = [
1036
935
  return {
1037
936
  type: 'stageSlowness', stageId: stage.id, impactBand,
1038
937
  metric: 'stageDurationMinutes', value,
1039
- recommendation: `This stage ran ${value} minutes with no more specific cause flagged: profile its query plan and check for wide shuffles or expensive UDFs.`,
938
+ recommendation: `This stage ran ${value} minutes with no more specific cause flagged: often a partition-count problem, raise parallelism via spark.sql.shuffle.partitions or spark.default.parallelism, or check for a large per-task data volume driving heavy shuffle and spill.`,
1040
939
  };
1041
940
  },
1042
941
  },
@@ -1050,6 +949,9 @@ export const DETECTORS = [
1050
949
  type: 'stageFailed', stageId: stage.id, impactBand: 'critical',
1051
950
  variant: 'stageFailure',
1052
951
  metric: 'stageFailureReason', value: stage.stageFailureReason,
952
+ numTasks: stage.taskCount,
953
+ memoryBytesSpilled: stage.memoryBytesSpilled,
954
+ failedTaskDetails: stage.failedTaskSamples ?? [],
1053
955
  recommendation: `This stage attempt failed outright. Inspect the driver log for the failure reason and the job that triggered it.`,
1054
956
  };
1055
957
  },
@@ -1081,9 +983,8 @@ export const DETECTORS = [
1081
983
  {
1082
984
  type: 'straggler', scope: 'stage', order: 70, fixEffort: 'code', version: 1,
1083
985
  docAnchor: '#bottleneck-straggler',
1084
- // floorPctWarn/floorPctCrit are re-exported as STRAGGLER_FLOOR_PCT_WARN/CRIT
1085
- // below and reused as src/impact-band.ts's global noise floor: keep the two
1086
- // in sync, don't hand-edit one without the other.
986
+ // floorPctWarn/floorPctCrit are re-exported as STRAGGLER_FLOOR_PCT_WARN/CRIT and reused as
987
+ // impact-band.ts's global noise floor: keep the two in sync.
1087
988
  thresholds: { minTasks: 10, shareWarn: 0.05, warnPct: 0.10, critPct: 0.20, floorPctWarn: STRAGGLER_FLOOR_PCT_WARN, floorPctCrit: STRAGGLER_FLOOR_PCT_CRIT },
1088
989
  detect(
1089
990
 
@@ -1097,12 +998,9 @@ export const DETECTORS = [
1097
998
  if ((stage.speculativeTasks ?? 0) === 0 && stragglerShare <= this.thresholds.shareWarn) return null;
1098
999
  const useSpeculative = (stage.speculativeTasks ?? 0) > 0;
1099
1000
  const speculativeShare = useSpeculative ? stage.speculativeTasks / stage.taskCount : 0;
1100
- // Same absolute delta src/impact-estimator.ts's shared straggler/
1101
- // stageShape case reports as this finding's savings: a high share of
1102
- // stragglers/speculative retries on a stage whose tasks barely vary in
1103
- // duration models near-zero savings, so it must not outrank 'info'.
1104
- // Clipped the same way before the floor check so the gate agrees with
1105
- // what's actually displayed.
1001
+ // Same absolute delta impact-estimator.ts's straggler/stageShape case reports as savings: a
1002
+ // high straggler/speculative share on a stage whose tasks barely vary models near-zero
1003
+ // savings, so it must not outrank 'info'. Clipped the same way before the floor check.
1106
1004
  const wasteMs = Math.max(0, stage.taskDurationMax - stage.taskDurationP50);
1107
1005
  const appDurationMs = computeAppDurationMs(ctx);
1108
1006
  const floorWasteMs = clippedWasteMs(wasteMs, stage.id, ctx);
@@ -1110,21 +1008,14 @@ export const DETECTORS = [
1110
1008
  const meetsCritFloor = meetsRuntimeFloor(floorWasteMs, appDurationMs, this.thresholds.floorPctCrit);
1111
1009
  const speculativeTier = speculativeShare >= this.thresholds.critPct && meetsCritFloor ? 'critical'
1112
1010
  : speculativeShare >= this.thresholds.warnPct && meetsWarnFloor ? 'warning' : 'info';
1113
- // Straggler share has no dedicated critical tier per
1114
- // docs-site/contributor-guide/architecture/detector-contract.md; it can only push to warning.
1011
+ // Straggler share has no dedicated critical tier per detector-contract.md; only warning.
1115
1012
  const stragglerTier = stragglerShare > this.thresholds.shareWarn && meetsWarnFloor ? 'warning' : 'info';
1116
- // Fixed fallback: overwritten by deriveImpactBand (src/impact-band.ts) whenever
1117
- // this finding gets a real wallClock estimate, which is the common case. Only
1118
- // surfaces on the rare miss (stage excluded from the occupancy sweep).
1013
+ // Fixed fallback: overwritten by deriveImpactBand when this finding gets a real wallClock
1014
+ // estimate (the common case). Only surfaces on the rare occupancy-sweep miss.
1119
1015
  const impactBand = 'info';
1120
- // Report whichever signal actually drove the finding, not just whether
1121
- // speculative execution happened to be on: a high stragglerShare with
1122
- // few/no speculative retries must not be reported as a low-value
1123
- // speculativeTasks count (real-log bug: a 50%-straggler-share stage
1124
- // with 1 speculative task reported an impact band of 'warning' but metric
1125
- // 'speculativeTasks: 1', hiding the actual cause). Ties keep the prior
1126
- // default (speculative-driven) so existing speculative-only findings
1127
- // are unaffected.
1016
+ // Report whichever signal actually drove the finding, not just whether speculation was on:
1017
+ // a high stragglerShare with few speculative retries must not be reported as a low-value
1018
+ // speculativeTasks count. Ties keep the speculative-driven default.
1128
1019
  const useSpeculativeMetric = useSpeculative && !(IMPACT_BAND_ORDER[stragglerTier] < IMPACT_BAND_ORDER[speculativeTier]);
1129
1020
  const value = useSpeculativeMetric ? stage.speculativeTasks : Math.round(stragglerShare * 100);
1130
1021
  const detail = useSpeculativeMetric
@@ -1137,7 +1028,9 @@ export const DETECTORS = [
1137
1028
  unit: useSpeculativeMetric ? 'count' : 'pct',
1138
1029
  speculativeTasks: stage.speculativeTasks ?? 0,
1139
1030
  stragglerCount: stage.stragglerCount ?? 0,
1140
- recommendation: `${detail}: investigate stragglers, likely candidates for AQE skewJoin or data locality issues.`,
1031
+ confidence: 'low',
1032
+ validationRequired: 'The 0.5%/2% runtime-floor percentages that gate this finding are our own noise floor, unvalidated: no external tool publishes an equivalent metric to calibrate against. Confirm against known-good/known-bad real logs before trusting the impact-band split.',
1033
+ recommendation: `${detail}: rule out a GC pause or a slow shuffle fetch before assuming a hardware issue; if a skewed key is the real cause, that's a candidate for AQE's skew-join handling.`,
1141
1034
  };
1142
1035
  },
1143
1036
  },
@@ -1179,6 +1072,9 @@ export const DETECTORS = [
1179
1072
  type: 'retryWaste', stageId: stage.id,
1180
1073
  impactBand: 'warning',
1181
1074
  metric: 'retryWasteMs', value: wastedMs,
1075
+ numTasks: stage.taskCount,
1076
+ memoryBytesSpilled: stage.memoryBytesSpilled,
1077
+ retriedTaskDetails: stage.retryTaskSamples ?? [],
1182
1078
  recommendation: `Retried task attempts wasted ${Math.round(wastedMs / 1000)}s of executor time (${wasted} attempt${wasted === 1 ? '' : 's'}) even though the stage completed: investigate executor loss or fetch failures.`,
1183
1079
  extended: `${wasted} task attempts were superseded by a later retry, wasting ${Math.round(wastedMs / 1000)}s of executor time. Common causes: executor loss (OOM-kill, node death) or shuffle FetchFailed forcing a stage-map recompute. Check driver logs for the dominant reason (see the Failures widget) even if the final failure rate looks low; retries hide the true cost.`,
1184
1080
  };
@@ -1206,14 +1102,13 @@ export const DETECTORS = [
1206
1102
  },
1207
1103
  },
1208
1104
  {
1209
- // No docAnchor set (documented deviation, like autoscalingChurn above):
1210
- // the upstream `shuffle-works/spark-tuning-reference` docs repo has no
1211
- // section for this tool-specific "capture stopped early" signal.
1105
+ // No docAnchor: the upstream spark-tuning-reference docs have no section for this
1106
+ // tool-specific "capture stopped early" signal.
1212
1107
  type: 'incompleteRun', scope: 'app', order: 5, fixEffort: 'code', version: 1,
1213
1108
  thresholds: {},
1214
1109
  recommendation: 'This event log never recorded an ApplicationEnd event: the capture stopped before the run finished (an in-flight job, a rotated-away log, or a cut-short capture). Findings and metrics elsewhere on this board reflect only what was captured up to that point, not the full run.',
1215
1110
  detect( ctx ) {
1216
- if (ctx.app.startTime == null || ctx.app.endTime != null) return null;
1111
+ if (!ctx.app || ctx.app.startTime == null || ctx.app.endTime != null) return null;
1217
1112
  return {
1218
1113
  type: 'incompleteRun', stageId: null, impactBand: 'warning',
1219
1114
  metric: 'applicationEnd', value: 'missing',
@@ -1230,18 +1125,22 @@ export const DETECTORS = [
1230
1125
  ctx ,
1231
1126
  ) {
1232
1127
  const { app, stages } = ctx;
1233
- if (!app.startTime || stages.size === 0) return null;
1128
+ // Nullish (not falsy) check: a literal startTime:0 must not be treated as "missing".
1129
+ if (!app || app.startTime == null || stages.size === 0) return null;
1234
1130
  let firstTaskLaunch = Infinity;
1235
1131
  for (const stage of stages.values()) {
1236
1132
  if (stage.submittedAt > 0 && stage.submittedAt < firstTaskLaunch) firstTaskLaunch = stage.submittedAt;
1237
1133
  }
1134
+ // No stage ever recorded a submission timestamp: no basis to measure a startup gap against.
1135
+ // Exposed now that a literal app.startTime:0 no longer short-circuits this detector entirely.
1136
+ if (!Number.isFinite(firstTaskLaunch)) return null;
1238
1137
  const gapSeconds = (firstTaskLaunch - app.startTime) / 1000;
1239
1138
  if (gapSeconds <= this.thresholds.gapSeconds) return null;
1240
1139
  const value = Math.round(gapSeconds);
1241
1140
  return {
1242
1141
  type: 'coldStart', stageId: null, impactBand: 'warning',
1243
1142
  metric: 'startupGapSeconds', value,
1244
- recommendation: `Executor startup took ${value}s: consider pre-warming the cluster or using dynamic allocation.`,
1143
+ recommendation: `The first task waited ${value}s for executors to become available: keep a warm pool of idle executors, or if using dynamic allocation, raise the minimum/initial executor count so it doesn't scale up from zero.`,
1245
1144
  };
1246
1145
  },
1247
1146
  },
@@ -1253,25 +1152,27 @@ export const DETECTORS = [
1253
1152
 
1254
1153
  ctx ,
1255
1154
  ) {
1256
- const { app, executorsAdded, executorsRemoved } = ctx;
1257
- if (executorsAdded.length === 0 || !app.startTime || !app.endTime) return null;
1155
+ const { app, executorsAdded, executorsRemoved, runAggregates } = ctx;
1156
+ // Nullish (not falsy) check: a literal startTime:0 must not be treated as "missing".
1157
+ if (!app || executorsAdded.length === 0 || app.startTime == null || app.endTime == null) return null;
1258
1158
  const appDuration = app.endTime - app.startTime;
1259
1159
  if (appDuration <= 0) return null;
1260
- const peakExecutors = executorsAdded.length;
1261
- let totalActiveMs = 0;
1262
- const removed = new Map ();
1263
- for (const ev of executorsRemoved) removed.set(ev.executorId, ev.timestamp);
1264
- for (const ev of executorsAdded) {
1265
- const addedAt = Math.max(ev.timestamp, app.startTime);
1266
- const removedAt = removed.has(ev.executorId) ? removed.get(ev.executorId) : app.endTime;
1267
- totalActiveMs += Math.max(0, removedAt - addedAt);
1268
- }
1269
- const utilization = (totalActiveMs / appDuration) / peakExecutors;
1160
+ // computePeakConcurrentCores (not executorsAdded.length/computeTotalCores): real concurrent
1161
+ // capacity, not a cumulative sum that double-counts a churned-through executor against its
1162
+ // replacement's (spot preemption, dynamicAllocation replacement).
1163
+ const totalCores = computePeakConcurrentCores(app, executorsAdded, executorsRemoved);
1164
+ if (totalCores <= 0) return null;
1165
+ const capacityCoreMs = totalCores * appDuration;
1166
+ // Busy core-time (from the whole-run core-time-series, same signal memoryUtilization's
1167
+ // idleCores variant already uses), not executor lifetime: an executor that exists for the
1168
+ // whole run but sits fully idle must not score as 100% used. Missing runAggregates (older
1169
+ // callers, synthetic fixtures) reads as 0 busy time rather than falling back to the
1170
+ // lifetime-based measure this replaces.
1171
+ const busyCoreMs = runAggregates?.busyCoreMs ?? 0;
1172
+ const utilization = busyCoreMs / capacityCoreMs;
1270
1173
  if (utilization >= this.thresholds.minUtil) return null;
1271
1174
 
1272
1175
  // CPU-time-based utilization (sparkMeasure): metric only, no threshold.
1273
- // Total cores: prefer executor-added Total Cores (real), else config cores.
1274
- const totalCores = computeTotalCores(app, executorsAdded);
1275
1176
  let cpuUtilizationPct = null;
1276
1177
  if (totalCores > 0) {
1277
1178
  let cpuMs = 0;
@@ -1296,10 +1197,10 @@ export const DETECTORS = [
1296
1197
  type: 'memoryUtilization', scope: 'app', order: 102, fixEffort: 'config', version: 1,
1297
1198
  docAnchor: '#bottleneck-memory-utilization',
1298
1199
  thresholds: {
1299
- idleCoreWarn: 0.50, // dataflint WastedCoresAlertsReducer
1300
- bandTooSmall: 0.95, // dataflint MemoryAlertsReducer: used/allocated
1200
+ idleCoreWarn: 0.50, // WastedCoresAlertsReducer
1201
+ bandTooSmall: 0.95, // MemoryAlertsReducer: used/allocated
1301
1202
  bandTooHigh: 0.70, // below this => over-provisioned (cost signal)
1302
- wasteBufferMultiplier: 1.5, // Dr. Elephant SparkMetricsAggregator: UNVERIFIED
1203
+ wasteBufferMultiplier: 1.5, // UNVERIFIED
1303
1204
  },
1304
1205
  detect(
1305
1206
 
@@ -1309,19 +1210,20 @@ export const DETECTORS = [
1309
1210
 
1310
1211
  ctx ,
1311
1212
  ) {
1312
- const { app, executorsAdded, runAggregates, stages } = ctx;
1213
+ const { app, executorsAdded, executorsRemoved, runAggregates, stages } = ctx;
1313
1214
  const out = [];
1314
- // Nullish (not falsy) check: unlike the older coldStart/utilization
1315
- // detectors, a literal startTime:0 must not be treated as "missing".
1215
+ // Nullish (not falsy) check: a literal startTime:0 must not be treated as "missing".
1316
1216
  if (app?.startTime == null || app?.endTime == null) return out;
1317
1217
  const appDurationMs = app.endTime - app.startTime;
1318
1218
  if (appDurationMs <= 0) return out;
1319
1219
 
1320
- // Total cores: prefer real Executor-Added Total Cores, else config.
1321
- const peakExecutors = executorsAdded.length;
1322
- const totalCores = computeTotalCores(app, executorsAdded);
1323
- // Hoisted above 1a (it is also 1b/1c's input) so the idle-cores finding can carry
1324
- // the allocated memory its MB-seconds impact estimate needs.
1220
+ // Peak concurrent executors/cores (not executorsAdded.length/computeTotalCores): a
1221
+ // cumulative sum or count double-counts a churned-through executor against its replacement's
1222
+ // (spot preemption, dynamicAllocation replacement), inflating idle-rate and waste-model figures.
1223
+ const peakExecutors = computePeakConcurrentExecutorCount(executorsAdded, executorsRemoved);
1224
+ const totalCores = computePeakConcurrentCores(app, executorsAdded, executorsRemoved);
1225
+ // Hoisted above 1a (also 1b/1c's input) so the idle-cores finding carries the allocated
1226
+ // memory its MB-seconds estimate needs.
1325
1227
  const allocatedMB = app.resources?.executor?.memoryMB ?? null;
1326
1228
 
1327
1229
  // ── 1a idle-cores rate ────────────────────────────────────────────────
@@ -1333,8 +1235,7 @@ export const DETECTORS = [
1333
1235
  out.push({
1334
1236
  type: 'memoryUtilization', variant: 'idleCores', stageId: null,
1335
1237
  impactBand: 'warning', metric: 'idleCoreRate', value,
1336
- // Raw (unrounded) rate plus the sizing inputs, for the impact estimator's
1337
- // wasted-MB-seconds model: `value` above is a rounded percentage.
1238
+ // Raw (unrounded) rate plus sizing inputs for the impact estimator: `value` is rounded pct.
1338
1239
  idleRateFraction: idleRate, allocatedMB, peakExecutors, appDurationMs,
1339
1240
  recommendation: `${value}% of allocated core-time ran no task: reduce cluster size or enable dynamic allocation.`,
1340
1241
  });
@@ -1362,10 +1263,8 @@ export const DETECTORS = [
1362
1263
  const allocatedBytes = allocatedMB * 1024 * 1024;
1363
1264
  for (const [execId, heap] of peakHeapByExec) {
1364
1265
  const ratio = heap / allocatedBytes;
1365
- // The two bands are opposite signals, not two degrees of one: an explicit
1366
- // `rule` discriminator (same pattern as stageShape) lets consumers tell the
1367
- // OOM-risk case from the over-provisioning waste case without re-deriving
1368
- // the ratio against the thresholds.
1266
+ // The two bands are opposite signals: an explicit `rule` discriminator lets consumers
1267
+ // tell OOM-risk from over-provisioning without re-deriving the ratio.
1369
1268
  if (ratio > this.thresholds.bandTooSmall) {
1370
1269
  out.push({
1371
1270
  type: 'memoryUtilization', variant: 'memoryBand', rule: 'heapNearCapacity',
@@ -1387,7 +1286,7 @@ export const DETECTORS = [
1387
1286
  }
1388
1287
  }
1389
1288
 
1390
- // ── 1c Spark Memory Limit waste model (Dr. Elephant, UNVERIFIED buffer) ─
1289
+ // ── 1c Spark Memory Limit waste model (UNVERIFIED buffer) ─
1391
1290
  if (allocatedMB != null && peakExecutors > 0) {
1392
1291
  const allocatedMBSeconds = peakExecutors * allocatedMB * (appDurationMs / 1000);
1393
1292
  let usedRunTimeMs = 0;
@@ -1400,7 +1299,7 @@ export const DETECTORS = [
1400
1299
  type: 'memoryUtilization', variant: 'wasteModel', stageId: null,
1401
1300
  impactBand: 'info', metric: 'wastedMBSeconds', value,
1402
1301
  confidence: 'low',
1403
- validationRequired: 'Memory-waste estimate uses allocated-vs-used memory-time and an unverified 1.5x buffer ported from Dr. Elephant: confirm against the Spark UI before acting.',
1302
+ validationRequired: 'Memory-waste estimate uses allocated-vs-used memory-time and an unverified 1.5x buffer: confirm against the Spark UI before acting.',
1404
1303
  recommendation: `Allocated executor memory sat largely idle over the run (~${value.toLocaleString('en-US')} MB-seconds wasted): review spark.executor.memory and executor count.`,
1405
1304
  });
1406
1305
  }
@@ -1410,13 +1309,9 @@ export const DETECTORS = [
1410
1309
  },
1411
1310
  },
1412
1311
  {
1413
- // Per-RDD cache-utilization proxies (this repo's own design: Spark
1414
- // event logs carry no block-access/read-count events, so a literal
1415
- // cache hit rate isn't derivable; see the design doc for GitHub issue
1416
- // #85). Two independent, per-RDD tiered checks over `ctx.app.rddInfo`
1417
- // storage snapshots: partial caching (numCachedPartitions < numPartitions)
1418
- // and disk spillover (diskSize share of a MEMORY_AND_DISK*-requesting
1419
- // RDD's cached footprint). An RDD can produce both findings in one pass.
1312
+ // Per-RDD cache-utilization proxies (this repo's own design: Spark event logs carry no
1313
+ // block-access events, so a literal cache hit rate isn't derivable). Two per-RDD tiered
1314
+ // checks over rddInfo snapshots: partial caching and disk spillover. An RDD can produce both.
1420
1315
  type: 'cacheUtilization', scope: 'app', order: 103, fixEffort: 'code', version: 1,
1421
1316
  docAnchor: '#memory-model',
1422
1317
  thresholds: {
@@ -1458,11 +1353,9 @@ export const DETECTORS = [
1458
1353
  },
1459
1354
  },
1460
1355
  {
1461
- // Non-local task ratio across stage.localityStats (RACK_LOCAL + ANY vs.
1462
- // all tasks), the other half of dataflint's "Wasted Cores Ratio" alert;
1463
- // the idle-core half already lives in memoryUtilization's idleCores
1464
- // variant above. NO_PREF stays in the denominator only: it's what
1465
- // shuffle-read stages legitimately report with no locality problem.
1356
+ // Non-local task ratio across stage.localityStats (RACK_LOCAL + ANY vs all tasks), the other
1357
+ // half of the "Wasted Cores Ratio" (idle-core half is memoryUtilization's idleCores).
1358
+ // NO_PREF stays in the denominator only: shuffle-read stages legitimately report it.
1466
1359
  type: 'coreLocality', scope: 'app', order: 103, fixEffort: 'config', version: 1,
1467
1360
  docAnchor: '#bottleneck-utilization',
1468
1361
  thresholds: { minTasks: 50, warnRatio: 0.15, critRatio: 0.35 },
@@ -1472,11 +1365,8 @@ export const DETECTORS = [
1472
1365
  ) {
1473
1366
  const { totalTasks, nonLocalTasks, ratio } = computeCoreLocalityRatio([...ctx.stages.values()]);
1474
1367
  if (totalTasks == null || totalTasks < this.thresholds.minTasks) return null;
1475
- // computeCoreLocalityRatio (src/core-locality-ratio.ts) only ever returns
1476
- // `ratio: null` together with `totalTasks: null` (its EMPTY sentinel sets
1477
- // both at once); the `totalTasks == null` guard above already rules that
1478
- // out, so `ratio` is guaranteed non-null here even though the function's
1479
- // declared return type keeps the two nullable independently.
1368
+ // computeCoreLocalityRatio only returns ratio:null together with totalTasks:null (shared
1369
+ // EMPTY sentinel); the totalTasks guard above rules that out, so ratio is non-null here.
1480
1370
  if (ratio < this.thresholds.warnRatio) return null;
1481
1371
 
1482
1372
  const value = Math.round(ratio * 100);
@@ -1484,8 +1374,7 @@ export const DETECTORS = [
1484
1374
  type: 'coreLocality', stageId: null,
1485
1375
  impactBand: ratio >= this.thresholds.critRatio ? 'critical' : 'warning',
1486
1376
  metric: 'nonLocalRatio', value,
1487
- // Raw count behind the ratio, for the impact estimator's network-fetch-penalty
1488
- // figure. Non-null whenever totalTasks is (both come from the same EMPTY sentinel).
1377
+ // Raw count behind the ratio, for the impact estimator. Non-null whenever totalTasks is.
1489
1378
  nonLocalTaskCount: nonLocalTasks ,
1490
1379
  confidence: 'low',
1491
1380
  validationRequired: 'The 15%/35% non-local-ratio thresholds (and the 50-task minimum) are unvalidated design-spike values: no external tool publishes an equivalent metric to calibrate against. Confirm against known-good/known-bad real logs before trusting the impact-band split.',
@@ -1494,11 +1383,9 @@ export const DETECTORS = [
1494
1383
  },
1495
1384
  },
1496
1385
  {
1497
- // Short-lived executors: an executor stood up and torn down before it
1498
- // could do useful work: wasteful re-provisioning, not normal scale-down.
1499
- // Reuses the same executorsAdded/executorsRemoved matching logic as
1500
- // `utilization` above (no new data extraction), but measures lifetime
1501
- // against a threshold instead of aggregate active-time.
1386
+ // Short-lived executors: stood up and torn down before doing useful work (wasteful
1387
+ // re-provisioning, not normal scale-down). Reuses utilization's add/remove matching, but
1388
+ // measures lifetime against a threshold instead of aggregate active-time.
1502
1389
  type: 'autoscalingChurn', scope: 'app', order: 103, fixEffort: 'config', version: 1,
1503
1390
  confidence: 'low',
1504
1391
  thresholds: { shortLivedMs: 120_000, warningPct: 0.30, criticalPct: 0.60, minExecutors: 5 },
@@ -1510,7 +1397,7 @@ export const DETECTORS = [
1510
1397
  ctx ,
1511
1398
  ) {
1512
1399
  const { app, executorsAdded, executorsRemoved } = ctx;
1513
- if (executorsAdded.length === 0 || app.endTime == null) return null;
1400
+ if (!app || executorsAdded.length === 0 || app.endTime == null) return null;
1514
1401
  if (executorsAdded.length < this.thresholds.minExecutors) return null;
1515
1402
 
1516
1403
  const removedAt = new Map ();
@@ -1540,10 +1427,8 @@ export const DETECTORS = [
1540
1427
  },
1541
1428
  },
1542
1429
  {
1543
- // Cross-execution relation reuse (repurposed from the old RDD-lineage
1544
- // heuristic, which surfaced only internal query-engine RDDs on DataFrame/
1545
- // SQL workloads; see docs/adr/0009-caching-opportunity-relation-reuse.md).
1546
- // Flags an input relation scanned by two or more SQL executions in one run.
1430
+ // Cross-execution relation reuse: flags an input relation scanned by two or more SQL
1431
+ // executions in one run, firing on real relation names (parquet:..., jdbc:...).
1547
1432
  type: 'cachingOpportunity', scope: 'app', order: 105, fixEffort: 'code', version: 1,
1548
1433
  docAnchor: '#bottleneck-utilization',
1549
1434
  thresholds: { minExecutions: 2 },
@@ -1566,12 +1451,10 @@ export const DETECTORS = [
1566
1451
 
1567
1452
 
1568
1453
 
1569
- // relationId -> { format, relation, executionIds:Set, executionBytes: Map<execId, bytes> }
1570
1454
  const byRelation = new Map ();
1571
1455
  for (const exec of sql.values()) {
1572
1456
  if (!exec.planTree) continue;
1573
- // Dedupe relations within one execution (self-joins count once), summing
1574
- // this execution's read bytes per relation across its scan nodes.
1457
+ // Dedupe relations within one execution (self-joins count once), summing read bytes per relation.
1575
1458
  const perExec = new Map ();
1576
1459
  walkPlanTree(exec.planTree, (node) => {
1577
1460
  const rid = scanRelationId(node.name ?? '', node.detail ?? '');
@@ -1591,15 +1474,13 @@ export const DETECTORS = [
1591
1474
  }
1592
1475
  }
1593
1476
 
1594
- // fingerprint -> { operator, exampleNode, executionIds:Set, executionBytes:Map<execId,bytes>, ancestorFingerprints:Set<fingerprint>, leafRelationRids:Set<rid> }
1595
1477
  const byComposite = new Map ();
1596
1478
  for (const exec of sql.values()) {
1597
1479
  if (!exec.planTree) continue;
1598
1480
  const candidates = findCompositeCandidates(exec.planTree);
1599
1481
  const fingerprintByNode = new Map (candidates.map((c) => [c.node, c.fingerprint]));
1600
1482
 
1601
- // Dedupe identical fingerprints within this execution (repeated
1602
- // identical composite counts once, mirroring the leaf perExec dedupe).
1483
+ // Dedupe identical fingerprints within this execution (repeated composite counts once).
1603
1484
  const perExecComposite = new Map ();
1604
1485
  for (const c of candidates) {
1605
1486
  let agg = perExecComposite.get(c.fingerprint);
@@ -1636,12 +1517,10 @@ export const DETECTORS = [
1636
1517
  }
1637
1518
  }
1638
1519
 
1639
- // Qualifying = has enough distinct executions on its own. Nested-dedupe:
1640
- // if a qualifying composite has a qualifying ANCESTOR composite, it is
1641
- // subsumed: fully (equal execution sets) or partially (residual).
1520
+ // Qualifying = enough distinct executions on its own. Nested-dedupe: a qualifying composite
1521
+ // with a qualifying ANCESTOR is subsumed, fully (equal sets) or partially (residual).
1642
1522
  const isQualifying = (fp ) =>
1643
1523
  byComposite.has(fp) && byComposite.get(fp) .executionIds.size >= this.thresholds.minExecutions;
1644
- // fingerprint -> { finalExecutionIds:Set, suppressed:boolean }
1645
1524
  const compositeResolutions = new Map ();
1646
1525
  for (const [fingerprint, agg] of byComposite) {
1647
1526
  if (!isQualifying(fingerprint)) { compositeResolutions.set(fingerprint, { finalExecutionIds: agg.executionIds, suppressed: true }); continue; }
@@ -1658,7 +1537,7 @@ export const DETECTORS = [
1658
1537
  const compositeVerb = { join: ['joined', 'join'], union: ['unioned', 'union'] };
1659
1538
 
1660
1539
  const out = [];
1661
- // rid -> Set<execId> covered by an emitted composite (for leaf suppression, Task 6)
1540
+ // rid -> Set<execId> covered by an emitted composite, for leaf suppression.
1662
1541
  const coveredExecutionsByRid = new Map ();
1663
1542
  for (const [fingerprint, agg] of byComposite) {
1664
1543
  const resolution = compositeResolutions.get(fingerprint) ;
@@ -1744,10 +1623,8 @@ export const DETECTORS = [
1744
1623
  let totalTasks = 0, failedTasks = 0;
1745
1624
  for (const s of stages.values()) { totalTasks += s.taskCount ?? 0; failedTasks += s.failedTasks ?? 0; }
1746
1625
  const taskFailureRate = totalTasks > 0 ? failedTasks / totalTasks : 0;
1747
- // Average wall-clock duration of the failed jobs, for the impact estimator's
1748
- // cost-only "core-hours burned on work that was thrown away" figure. Jobs
1749
- // missing either timestamp are excluded rather than counted as zero-length;
1750
- // with no timed failed job at all the average is 0 (never NaN).
1626
+ // Average wall-clock of failed jobs, for the impact estimator's cost-only figure. Jobs
1627
+ // missing either timestamp are excluded (not counted as zero); with none timed the average is 0.
1751
1628
  const timedFailedJobs = failedJobList.filter(j => j.submissionTime != null && j.completionTime != null);
1752
1629
  const avgJobDurationMs = timedFailedJobs.length > 0
1753
1630
  ? timedFailedJobs.reduce((s, j) => s + (j.completionTime - j.submissionTime ), 0) / timedFailedJobs.length
@@ -1858,12 +1735,13 @@ export const DETECTORS = [
1858
1735
  const nodes = [];
1859
1736
  for (const n of g.nodes) walkPlanTree(n, (node) => nodes.push(node));
1860
1737
  const stageIds = unionStageIds(nodes, fallbackStageIds);
1738
+ // resolvePlanTree always sets id; safe downstream of it.
1739
+ const planNodeIds = nodes.map((n) => n.id ).filter(Boolean);
1861
1740
  const touching = g.sampleRelation ? ` (touching ${g.sampleRelation})` : '';
1862
1741
  return {
1863
- type: 'duplicatePlanSubtree', executionId: sqlExec.id, stageIds,
1864
- // Fixed fallback: overwritten by deriveImpactBand whenever this finding
1865
- // gets a real wallClock estimate, which is the common case. Only
1866
- // surfaces on the rare miss (stage excluded from the occupancy sweep).
1742
+ type: 'duplicatePlanSubtree', executionId: sqlExec.id, stageIds, planNodeIds,
1743
+ // Fixed fallback: overwritten by deriveImpactBand when this finding gets a real
1744
+ // wallClock estimate (the common case). Only surfaces on the rare occupancy-sweep miss.
1867
1745
  impactBand: 'warning', metric: 'subtreeOccurrences', value: g.occurrences,
1868
1746
  rootName: g.rootName, subtreeSize: g.subtreeSize, sampleRelation: g.sampleRelation,
1869
1747
  groupIndex: g.groupIndex, confidence: 'medium',
@@ -1908,8 +1786,9 @@ export const DETECTORS = [
1908
1786
  const fallbackStageIds = stageIdsForSqlExec(sqlExec.id, ctx.stages);
1909
1787
  return hits.map(h => {
1910
1788
  const stageIds = unionStageIds([h.node], fallbackStageIds);
1789
+ const planNodeIds = h.node.id ? [h.node.id] : [];
1911
1790
  return {
1912
- type: 'smallFiles', executionId: sqlExec.id, stageIds,
1791
+ type: 'smallFiles', executionId: sqlExec.id, stageIds, planNodeIds,
1913
1792
  impactBand: 'warning',
1914
1793
  metric: 'avgFileSizeBytes', value: Math.round(h.avgBytes),
1915
1794
  fileCount: h.fileCount, direction: h.direction, nodeName: h.nodeName,
@@ -1921,10 +1800,9 @@ export const DETECTORS = [
1921
1800
  },
1922
1801
  },
1923
1802
  {
1924
- // Entry-level type is an identifier only; it never appears on an
1925
- // emitted finding. Findings carry their own type ('underBroadcast' or
1926
- // 'overBroadcast') since one shared plan-walk covers both opposite-
1927
- // direction rules (dataflint JoinToBroadcastAlert / BroadcastTooLargeAlert).
1803
+ // Entry-level type is an identifier only; it never appears on an emitted finding. Findings
1804
+ // carry 'underBroadcast'/'overBroadcast' since one shared plan-walk covers both
1805
+ // opposite-direction rules (JoinToBroadcastAlert / BroadcastTooLargeAlert).
1928
1806
  type: 'broadcastSizing', scope: 'sql', order: 132, fixEffort: 'config', version: 2,
1929
1807
  docAnchor: '#bottleneck-broadcast-sizing',
1930
1808
  thresholds: {
@@ -1959,6 +1837,8 @@ export const DETECTORS = [
1959
1837
  const contributors = [...boundarySizeContributors(childA), ...boundarySizeContributors(childB)];
1960
1838
  out.push({
1961
1839
  type: 'underBroadcast', executionId: sqlExec.id, stageIds: unionStageIds(contributors, fallbackStageIds),
1840
+ // resolvePlanTree always sets id; safe downstream of it.
1841
+ planNodeIds: contributors.map((n) => n.id ).filter(Boolean),
1962
1842
  impactBand: 'info', metric: 'smallerSideBytes', value: smaller,
1963
1843
  largerSideBytes: larger,
1964
1844
  recommendation: `The smaller input to this Sort Merge Join (${formatBytes(smaller)}) is well under the broadcast threshold relative to the larger side (${formatBytes(larger)}): this could have been a broadcast join. Consider a broadcast() hint or raising spark.sql.autoBroadcastJoinThreshold.`,
@@ -1966,17 +1846,18 @@ export const DETECTORS = [
1966
1846
  }
1967
1847
  }
1968
1848
  }
1969
- if (node.name === 'BroadcastExchange') {
1849
+ if (isBroadcastExchangeNode(node.name)) {
1970
1850
  const m = (node.metrics ?? []).find(x => x.name === 'data size');
1971
1851
  if (m && m.value > overBroadcastBytes) {
1972
- // The BroadcastExchange node's own metrics are computed on the
1973
- // driver and never appear on any TaskEnd (node.stageIds is
1974
- // always empty in real data); its child carries the
1975
- // executor-side metrics, so only the child is unioned in.
1852
+ // BroadcastExchange's own metrics are driver-computed and never on any TaskEnd
1853
+ // (node.stageIds always empty in real data); its child carries the executor-side
1854
+ // metrics, so only the child unions in.
1976
1855
  const child = (node.children ?? [])[0];
1977
1856
  out.push({
1978
1857
  type: 'overBroadcast', executionId: sqlExec.id,
1979
1858
  stageIds: unionStageIds(child ? [child] : [], fallbackStageIds),
1859
+ // resolvePlanTree always sets id; safe downstream of it.
1860
+ planNodeIds: [node.id ].filter(Boolean),
1980
1861
  impactBand: 'warning', metric: 'broadcastBytes', value: m.value,
1981
1862
  recommendation: `This broadcast (${formatBytes(m.value)}) exceeds the 1 GB threshold: check for a misapplied broadcast hint or a misconfigured spark.sql.autoBroadcastJoinThreshold.`,
1982
1863
  });