sparkforensics-cli 0.1.0 → 0.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (360) hide show
  1. package/README.md +6 -0
  2. package/bin/sparkforensics-analyze.mjs +113 -48
  3. package/export-template/docs/404.html +25 -0
  4. package/export-template/docs/assets/app.CndaAS6v.js +1 -0
  5. package/export-template/docs/assets/aqe-loop.IwQSATHw.svg +1 -0
  6. package/export-template/docs/assets/aqe-loop.dark.DGbaxqJE.svg +1 -0
  7. package/export-template/docs/assets/broadcast-vs-shuffle.Db4WY1XK.svg +1 -0
  8. package/export-template/docs/assets/broadcast-vs-shuffle.dark.C7Bxs0mG.svg +1 -0
  9. package/export-template/docs/assets/cache-lifecycle.dark.B-hS7AgU.svg +1 -0
  10. package/export-template/docs/assets/cache-lifecycle.rEOVYQNU.svg +1 -0
  11. package/export-template/docs/assets/chunks/@localSearchIndexroot.DNY8bVcl.js +1 -0
  12. package/export-template/docs/assets/chunks/VPLocalSearchBox.yJbZbsEo.js +9 -0
  13. package/export-template/docs/assets/chunks/duplicate-plan-subtree.dark.Cdp70QhV.js +1 -0
  14. package/export-template/docs/assets/chunks/framework.DSg0KOwT.js +20 -0
  15. package/export-template/docs/assets/chunks/retry-escalation-ladder.dark.DHipdJgZ.js +1 -0
  16. package/export-template/docs/assets/chunks/theme.Df2VAG9w.js +2 -0
  17. package/export-template/docs/assets/cold-start-timeline.DxC_Sc7w.svg +1 -0
  18. package/export-template/docs/assets/cold-start-timeline.dark.CZ17YcAG.svg +1 -0
  19. package/export-template/docs/assets/columnar-layout.PghGeOEA.svg +1 -0
  20. package/export-template/docs/assets/columnar-layout.dark.BVNlz0ff.svg +1 -0
  21. package/export-template/docs/assets/container-memory.DIO0AnIm.svg +1 -0
  22. package/export-template/docs/assets/container-memory.dark.CP-5zuCl.svg +1 -0
  23. package/export-template/docs/assets/contributor-guide_architecture_board-widgets.md.B-OsL91z.js +1 -0
  24. package/export-template/docs/assets/contributor-guide_architecture_board-widgets.md.B-OsL91z.lean.js +1 -0
  25. package/export-template/docs/assets/contributor-guide_architecture_detector-contract.md.BOeH4d1J.js +1 -0
  26. package/export-template/docs/assets/contributor-guide_architecture_detector-contract.md.BOeH4d1J.lean.js +1 -0
  27. package/export-template/docs/assets/contributor-guide_architecture_drill-down.md.BtPdlM7r.js +1 -0
  28. package/export-template/docs/assets/contributor-guide_architecture_drill-down.md.BtPdlM7r.lean.js +1 -0
  29. package/export-template/docs/assets/contributor-guide_architecture_impact-estimation.md.DYCDPgkh.js +1 -0
  30. package/export-template/docs/assets/contributor-guide_architecture_impact-estimation.md.DYCDPgkh.lean.js +1 -0
  31. package/export-template/docs/assets/contributor-guide_architecture_index.md.3TO9ic6w.js +1 -0
  32. package/export-template/docs/assets/contributor-guide_architecture_index.md.3TO9ic6w.lean.js +1 -0
  33. package/export-template/docs/assets/contributor-guide_architecture_overview.md.CehiRmGn.js +1 -0
  34. package/export-template/docs/assets/contributor-guide_architecture_overview.md.CehiRmGn.lean.js +1 -0
  35. package/export-template/docs/assets/contributor-guide_architecture_state-and-history.md.m3S3UdMk.js +1 -0
  36. package/export-template/docs/assets/contributor-guide_architecture_state-and-history.md.m3S3UdMk.lean.js +1 -0
  37. package/export-template/docs/assets/contributor-guide_architecture_widget-rendering.md.DbqPf2OT.js +1 -0
  38. package/export-template/docs/assets/contributor-guide_architecture_widget-rendering.md.DbqPf2OT.lean.js +1 -0
  39. package/export-template/docs/assets/contributor-guide_architecture_worker-protocol.md.B93qJ_tT.js +6 -0
  40. package/export-template/docs/assets/contributor-guide_architecture_worker-protocol.md.B93qJ_tT.lean.js +1 -0
  41. package/export-template/docs/assets/contributor-guide_contributing.md.CvRsdr6J.js +1 -0
  42. package/export-template/docs/assets/contributor-guide_contributing.md.CvRsdr6J.lean.js +1 -0
  43. package/export-template/docs/assets/contributor-guide_development-setup.md.DvAN_9mK.js +12 -0
  44. package/export-template/docs/assets/contributor-guide_development-setup.md.DvAN_9mK.lean.js +1 -0
  45. package/export-template/docs/assets/contributor-guide_testing.md.6rIKqSyY.js +1 -0
  46. package/export-template/docs/assets/contributor-guide_testing.md.6rIKqSyY.lean.js +1 -0
  47. package/export-template/docs/assets/dag-stages.DSz_S937.svg +1 -0
  48. package/export-template/docs/assets/dag-stages.dark.F72UzxH4.svg +1 -0
  49. package/export-template/docs/assets/driver-executor.D5pQ7YN1.svg +1 -0
  50. package/export-template/docs/assets/driver-executor.dark.BmX9cPvh.svg +1 -0
  51. package/export-template/docs/assets/duplicate-plan-subtree.B4cvN6fj.svg +1 -0
  52. package/export-template/docs/assets/duplicate-plan-subtree.dark.Dw8wS0Ag.svg +1 -0
  53. package/export-template/docs/assets/index.md.CHJVslga.js +1 -0
  54. package/export-template/docs/assets/index.md.CHJVslga.lean.js +1 -0
  55. package/export-template/docs/assets/inter-italic-cyrillic-ext.r48I6akx.woff2 +0 -0
  56. package/export-template/docs/assets/inter-italic-cyrillic.By2_1cv3.woff2 +0 -0
  57. package/export-template/docs/assets/inter-italic-greek-ext.1u6EdAuj.woff2 +0 -0
  58. package/export-template/docs/assets/inter-italic-greek.DJ8dCoTZ.woff2 +0 -0
  59. package/export-template/docs/assets/inter-italic-latin-ext.CN1xVJS-.woff2 +0 -0
  60. package/export-template/docs/assets/inter-italic-latin.C2AdPX0b.woff2 +0 -0
  61. package/export-template/docs/assets/inter-italic-vietnamese.BSbpV94h.woff2 +0 -0
  62. package/export-template/docs/assets/inter-roman-cyrillic-ext.BBPuwvHQ.woff2 +0 -0
  63. package/export-template/docs/assets/inter-roman-cyrillic.C5lxZ8CY.woff2 +0 -0
  64. package/export-template/docs/assets/inter-roman-greek-ext.CqjqNYQ-.woff2 +0 -0
  65. package/export-template/docs/assets/inter-roman-greek.BBVDIX6e.woff2 +0 -0
  66. package/export-template/docs/assets/inter-roman-latin-ext.4ZJIpNVo.woff2 +0 -0
  67. package/export-template/docs/assets/inter-roman-latin.Di8DUHzh.woff2 +0 -0
  68. package/export-template/docs/assets/inter-roman-vietnamese.BjW4sHH5.woff2 +0 -0
  69. package/export-template/docs/assets/join-strategy.C_FvrCEo.svg +1 -0
  70. package/export-template/docs/assets/join-strategy.dark.ChMLnNII.svg +1 -0
  71. package/export-template/docs/assets/memory-borrowing.BqQRJg0u.svg +1 -0
  72. package/export-template/docs/assets/memory-borrowing.dark.Yhh20O9C.svg +1 -0
  73. package/export-template/docs/assets/memory-regions.XHvO7jHG.svg +1 -0
  74. package/export-template/docs/assets/memory-regions.dark.D4TP9_08.svg +1 -0
  75. package/export-template/docs/assets/repartition-vs-coalesce.BovLRrpj.svg +1 -0
  76. package/export-template/docs/assets/repartition-vs-coalesce.dark.BhAczKZQ.svg +1 -0
  77. package/export-template/docs/assets/retry-escalation-ladder.DyTKJJmZ.svg +1 -0
  78. package/export-template/docs/assets/retry-escalation-ladder.dark.BdsabtU3.svg +1 -0
  79. package/export-template/docs/assets/shuffle-map-reduce.KuOEZVmg.svg +1 -0
  80. package/export-template/docs/assets/shuffle-map-reduce.dark.BgQZnFSb.svg +1 -0
  81. package/export-template/docs/assets/spill-classification.BU2euYDO.svg +1 -0
  82. package/export-template/docs/assets/spill-classification.dark.D7i1M40d.svg +1 -0
  83. package/export-template/docs/assets/style.DXOMCXxn.css +1 -0
  84. package/export-template/docs/assets/tuning-reference_anti-patterns.md.Df1YMIHu.js +1 -0
  85. package/export-template/docs/assets/tuning-reference_anti-patterns.md.Df1YMIHu.lean.js +1 -0
  86. package/export-template/docs/assets/tuning-reference_aqe.md.BIsCtLzm.js +1 -0
  87. package/export-template/docs/assets/tuning-reference_aqe.md.BIsCtLzm.lean.js +1 -0
  88. package/export-template/docs/assets/tuning-reference_bottleneck-broadcast-sizing.md.CEstB3Ia.js +1 -0
  89. package/export-template/docs/assets/tuning-reference_bottleneck-broadcast-sizing.md.CEstB3Ia.lean.js +1 -0
  90. package/export-template/docs/assets/tuning-reference_bottleneck-cold-start.md.CEuy-72y.js +7 -0
  91. package/export-template/docs/assets/tuning-reference_bottleneck-cold-start.md.CEuy-72y.lean.js +1 -0
  92. package/export-template/docs/assets/tuning-reference_bottleneck-duplicate-plan-subtree.md.CIohQDfn.js +1 -0
  93. package/export-template/docs/assets/tuning-reference_bottleneck-duplicate-plan-subtree.md.CIohQDfn.lean.js +1 -0
  94. package/export-template/docs/assets/tuning-reference_bottleneck-failures.md.4z5BXGJ2.js +6 -0
  95. package/export-template/docs/assets/tuning-reference_bottleneck-failures.md.4z5BXGJ2.lean.js +1 -0
  96. package/export-template/docs/assets/tuning-reference_bottleneck-gc.md.DSxzZRK7.js +6 -0
  97. package/export-template/docs/assets/tuning-reference_bottleneck-gc.md.DSxzZRK7.lean.js +1 -0
  98. package/export-template/docs/assets/tuning-reference_bottleneck-job-failure-rate.md.BaJl__1W.js +8 -0
  99. package/export-template/docs/assets/tuning-reference_bottleneck-job-failure-rate.md.BaJl__1W.lean.js +1 -0
  100. package/export-template/docs/assets/tuning-reference_bottleneck-memory-utilization.md.DbP-SJZc.js +7 -0
  101. package/export-template/docs/assets/tuning-reference_bottleneck-memory-utilization.md.DbP-SJZc.lean.js +1 -0
  102. package/export-template/docs/assets/tuning-reference_bottleneck-retry-waste.md.D5JMjVOt.js +1 -0
  103. package/export-template/docs/assets/tuning-reference_bottleneck-retry-waste.md.D5JMjVOt.lean.js +1 -0
  104. package/export-template/docs/assets/tuning-reference_bottleneck-shuffle.md.CM-nTmIH.js +12 -0
  105. package/export-template/docs/assets/tuning-reference_bottleneck-shuffle.md.CM-nTmIH.lean.js +1 -0
  106. package/export-template/docs/assets/tuning-reference_bottleneck-skew.md.BdUwiDhn.js +14 -0
  107. package/export-template/docs/assets/tuning-reference_bottleneck-skew.md.BdUwiDhn.lean.js +1 -0
  108. package/export-template/docs/assets/tuning-reference_bottleneck-slow-host.md.BlIo6UDW.js +7 -0
  109. package/export-template/docs/assets/tuning-reference_bottleneck-slow-host.md.BlIo6UDW.lean.js +1 -0
  110. package/export-template/docs/assets/tuning-reference_bottleneck-small-files.md.B8kloyx8.js +5 -0
  111. package/export-template/docs/assets/tuning-reference_bottleneck-small-files.md.B8kloyx8.lean.js +1 -0
  112. package/export-template/docs/assets/tuning-reference_bottleneck-spill.md.PNH7mITt.js +6 -0
  113. package/export-template/docs/assets/tuning-reference_bottleneck-spill.md.PNH7mITt.lean.js +1 -0
  114. package/export-template/docs/assets/tuning-reference_bottleneck-straggler.md.DY36fHN5.js +7 -0
  115. package/export-template/docs/assets/tuning-reference_bottleneck-straggler.md.DY36fHN5.lean.js +1 -0
  116. package/export-template/docs/assets/tuning-reference_bottleneck-tiny-tasks.md.QTV7O8kU.js +8 -0
  117. package/export-template/docs/assets/tuning-reference_bottleneck-tiny-tasks.md.QTV7O8kU.lean.js +1 -0
  118. package/export-template/docs/assets/tuning-reference_bottleneck-utilization.md.DTiueZC3.js +5 -0
  119. package/export-template/docs/assets/tuning-reference_bottleneck-utilization.md.DTiueZC3.lean.js +1 -0
  120. package/export-template/docs/assets/tuning-reference_caching.md.B7aQ8asB.js +1 -0
  121. package/export-template/docs/assets/tuning-reference_caching.md.B7aQ8asB.lean.js +1 -0
  122. package/export-template/docs/assets/tuning-reference_cluster-config.md.ZVmDGsQ3.js +1 -0
  123. package/export-template/docs/assets/tuning-reference_cluster-config.md.ZVmDGsQ3.lean.js +1 -0
  124. package/export-template/docs/assets/tuning-reference_config.md.UvveiWG3.js +1 -0
  125. package/export-template/docs/assets/tuning-reference_config.md.UvveiWG3.lean.js +1 -0
  126. package/export-template/docs/assets/tuning-reference_data-formats.md.bjCAWH3N.js +1 -0
  127. package/export-template/docs/assets/tuning-reference_data-formats.md.bjCAWH3N.lean.js +1 -0
  128. package/export-template/docs/assets/tuning-reference_index.md.BQ_NooMV.js +1 -0
  129. package/export-template/docs/assets/tuning-reference_index.md.BQ_NooMV.lean.js +1 -0
  130. package/export-template/docs/assets/tuning-reference_intro.md.CobD-lGB.js +1 -0
  131. package/export-template/docs/assets/tuning-reference_intro.md.CobD-lGB.lean.js +1 -0
  132. package/export-template/docs/assets/tuning-reference_joins.md.BtKs_CuW.js +1 -0
  133. package/export-template/docs/assets/tuning-reference_joins.md.BtKs_CuW.lean.js +1 -0
  134. package/export-template/docs/assets/tuning-reference_memory-model.md.DhT-n4y3.js +1 -0
  135. package/export-template/docs/assets/tuning-reference_memory-model.md.DhT-n4y3.lean.js +1 -0
  136. package/export-template/docs/assets/tuning-reference_metrics.md.mLOh7Apj.js +1 -0
  137. package/export-template/docs/assets/tuning-reference_metrics.md.mLOh7Apj.lean.js +1 -0
  138. package/export-template/docs/assets/tuning-reference_partitioning.md.q0zKF_8X.js +1 -0
  139. package/export-template/docs/assets/tuning-reference_partitioning.md.q0zKF_8X.lean.js +1 -0
  140. package/export-template/docs/assets/tuning-reference_pyspark.md.DDCfvN9t.js +6 -0
  141. package/export-template/docs/assets/tuning-reference_pyspark.md.DDCfvN9t.lean.js +1 -0
  142. package/export-template/docs/assets/tuning-reference_shuffle.md.BZZ7R4Ix.js +1 -0
  143. package/export-template/docs/assets/tuning-reference_shuffle.md.BZZ7R4Ix.lean.js +1 -0
  144. package/export-template/docs/assets/tuning-reference_spark-architecture.md.Dwzm5avO.js +1 -0
  145. package/export-template/docs/assets/tuning-reference_spark-architecture.md.Dwzm5avO.lean.js +1 -0
  146. package/export-template/docs/assets/tuning-reference_table-formats.md.D6wj-2dX.js +1 -0
  147. package/export-template/docs/assets/tuning-reference_table-formats.md.D6wj-2dX.lean.js +1 -0
  148. package/export-template/docs/assets/udf-execution-models.BUFDICuG.svg +1 -0
  149. package/export-template/docs/assets/udf-execution-models.dark.YTNS6GDq.svg +1 -0
  150. package/export-template/docs/assets/user-guide_alternative-log-retrieval.md.sU3KGarf.js +1 -0
  151. package/export-template/docs/assets/user-guide_alternative-log-retrieval.md.sU3KGarf.lean.js +1 -0
  152. package/export-template/docs/assets/user-guide_getting-started.md.DtEM37MK.js +3 -0
  153. package/export-template/docs/assets/user-guide_getting-started.md.DtEM37MK.lean.js +1 -0
  154. package/export-template/docs/assets/user-guide_mcp-tools.md.C8MiIu7F.js +125 -0
  155. package/export-template/docs/assets/user-guide_mcp-tools.md.C8MiIu7F.lean.js +1 -0
  156. package/export-template/docs/assets/user-guide_run-comparison.md.S0TWWmLY.js +1 -0
  157. package/export-template/docs/assets/user-guide_run-comparison.md.S0TWWmLY.lean.js +1 -0
  158. package/export-template/docs/assets/user-guide_understanding-findings.md.D0R_Y-R2.js +1 -0
  159. package/export-template/docs/assets/user-guide_understanding-findings.md.D0R_Y-R2.lean.js +1 -0
  160. package/export-template/docs/contributor-guide/architecture/board-widgets.html +25 -0
  161. package/export-template/docs/contributor-guide/architecture/detector-contract.html +25 -0
  162. package/export-template/docs/contributor-guide/architecture/drill-down.html +25 -0
  163. package/export-template/docs/contributor-guide/architecture/impact-estimation.html +25 -0
  164. package/export-template/docs/contributor-guide/architecture/index.html +25 -0
  165. package/export-template/docs/contributor-guide/architecture/overview.html +25 -0
  166. package/export-template/docs/contributor-guide/architecture/state-and-history.html +25 -0
  167. package/export-template/docs/contributor-guide/architecture/widget-rendering.html +25 -0
  168. package/export-template/docs/contributor-guide/architecture/worker-protocol.html +30 -0
  169. package/export-template/docs/contributor-guide/contributing.html +25 -0
  170. package/export-template/docs/contributor-guide/development-setup.html +36 -0
  171. package/export-template/docs/contributor-guide/testing.html +25 -0
  172. package/export-template/docs/favicon.svg +4 -0
  173. package/export-template/docs/hashmap.json +1 -0
  174. package/export-template/docs/index.html +25 -0
  175. package/export-template/docs/package.json +1 -0
  176. package/export-template/docs/tuning-reference/anti-patterns.html +25 -0
  177. package/export-template/docs/tuning-reference/aqe.html +25 -0
  178. package/export-template/docs/tuning-reference/bottleneck-broadcast-sizing.html +25 -0
  179. package/export-template/docs/tuning-reference/bottleneck-cold-start.html +31 -0
  180. package/export-template/docs/tuning-reference/bottleneck-duplicate-plan-subtree.html +25 -0
  181. package/export-template/docs/tuning-reference/bottleneck-failures.html +30 -0
  182. package/export-template/docs/tuning-reference/bottleneck-gc.html +30 -0
  183. package/export-template/docs/tuning-reference/bottleneck-job-failure-rate.html +32 -0
  184. package/export-template/docs/tuning-reference/bottleneck-memory-utilization.html +31 -0
  185. package/export-template/docs/tuning-reference/bottleneck-retry-waste.html +25 -0
  186. package/export-template/docs/tuning-reference/bottleneck-shuffle.html +36 -0
  187. package/export-template/docs/tuning-reference/bottleneck-skew.html +38 -0
  188. package/export-template/docs/tuning-reference/bottleneck-slow-host.html +31 -0
  189. package/export-template/docs/tuning-reference/bottleneck-small-files.html +29 -0
  190. package/export-template/docs/tuning-reference/bottleneck-spill.html +30 -0
  191. package/export-template/docs/tuning-reference/bottleneck-straggler.html +31 -0
  192. package/export-template/docs/tuning-reference/bottleneck-tiny-tasks.html +32 -0
  193. package/export-template/docs/tuning-reference/bottleneck-utilization.html +29 -0
  194. package/export-template/docs/tuning-reference/caching.html +25 -0
  195. package/export-template/docs/tuning-reference/cluster-config.html +25 -0
  196. package/export-template/docs/tuning-reference/config.html +25 -0
  197. package/export-template/docs/tuning-reference/data-formats.html +25 -0
  198. package/export-template/docs/tuning-reference/index.html +25 -0
  199. package/export-template/docs/tuning-reference/intro.html +25 -0
  200. package/export-template/docs/tuning-reference/joins.html +25 -0
  201. package/export-template/docs/tuning-reference/memory-model.html +25 -0
  202. package/export-template/docs/tuning-reference/metrics.html +25 -0
  203. package/export-template/docs/tuning-reference/partitioning.html +25 -0
  204. package/export-template/docs/tuning-reference/pyspark.html +30 -0
  205. package/export-template/docs/tuning-reference/shuffle.html +25 -0
  206. package/export-template/docs/tuning-reference/spark-architecture.html +25 -0
  207. package/export-template/docs/tuning-reference/table-formats.html +25 -0
  208. package/export-template/docs/user-guide/alternative-log-retrieval.html +25 -0
  209. package/export-template/docs/user-guide/getting-started.html +27 -0
  210. package/export-template/docs/user-guide/mcp-tools.html +149 -0
  211. package/export-template/docs/user-guide/run-comparison.html +25 -0
  212. package/export-template/docs/user-guide/understanding-findings.html +25 -0
  213. package/export-template/docs/vp-icons.css +0 -0
  214. package/export-template/favicon.svg +4 -0
  215. package/export-template/index.html +115 -0
  216. package/export-template/parser-worker-QqyEE4m9.js +64 -0
  217. package/package.json +16 -3
  218. package/vendor-core/analyzer.js +74 -74
  219. package/vendor-core/cli/budgets.js +13 -27
  220. package/vendor-core/cli/collect-run.js +43 -19
  221. package/vendor-core/core-count.js +25 -27
  222. package/vendor-core/core-locality-ratio.js +4 -11
  223. package/vendor-core/core-time-series.js +6 -12
  224. package/vendor-core/core-usage-locality.js +3 -4
  225. package/vendor-core/detectors.js +256 -375
  226. package/vendor-core/docs-config.js +69 -21
  227. package/vendor-core/docs-content/chapters/01-intro.md +32 -0
  228. package/vendor-core/docs-content/chapters/02-spark-architecture.md +76 -0
  229. package/vendor-core/docs-content/chapters/03-memory-model.md +73 -0
  230. package/vendor-core/docs-content/chapters/04-partitioning.md +65 -0
  231. package/vendor-core/docs-content/chapters/05-joins.md +62 -0
  232. package/vendor-core/docs-content/chapters/06-shuffle.md +59 -0
  233. package/vendor-core/docs-content/chapters/07-data-formats.md +81 -0
  234. package/vendor-core/docs-content/chapters/07b-table-formats.md +56 -0
  235. package/vendor-core/docs-content/chapters/08-caching.md +58 -0
  236. package/vendor-core/docs-content/chapters/09-pyspark.md +78 -0
  237. package/vendor-core/docs-content/chapters/10-aqe.md +167 -0
  238. package/vendor-core/docs-content/chapters/11-cluster-config.md +170 -0
  239. package/vendor-core/docs-content/chapters/12-anti-patterns.md +171 -0
  240. package/vendor-core/docs-content/chapters/14-metrics.md +87 -0
  241. package/vendor-core/docs-content/chapters/15-config.md +93 -0
  242. package/vendor-core/docs-content/chapters/nav-index.json +370 -0
  243. package/vendor-core/docs-content/detection/cache.md +6 -0
  244. package/vendor-core/docs-content/detection/cfg.md +15 -0
  245. package/vendor-core/docs-content/detection/chrn.md +7 -0
  246. package/vendor-core/docs-content/detection/cold.md +4 -0
  247. package/vendor-core/docs-content/detection/cstor.md +4 -0
  248. package/vendor-core/docs-content/detection/fail.md +5 -0
  249. package/vendor-core/docs-content/detection/gc.md +4 -0
  250. package/vendor-core/docs-content/detection/host.md +5 -0
  251. package/vendor-core/docs-content/detection/incmp.md +6 -0
  252. package/vendor-core/docs-content/detection/jobs.md +4 -0
  253. package/vendor-core/docs-content/detection/local.md +7 -0
  254. package/vendor-core/docs-content/detection/mem.md +10 -0
  255. package/vendor-core/docs-content/detection/part.md +5 -0
  256. package/vendor-core/docs-content/detection/plan.md +14 -0
  257. package/vendor-core/docs-content/detection/retry.md +4 -0
  258. package/vendor-core/docs-content/detection/sfail.md +5 -0
  259. package/vendor-core/docs-content/detection/shape.md +5 -0
  260. package/vendor-core/docs-content/detection/shfl.md +4 -0
  261. package/vendor-core/docs-content/detection/skew.md +6 -0
  262. package/vendor-core/docs-content/detection/slow.md +6 -0
  263. package/vendor-core/docs-content/detection/spec.md +7 -0
  264. package/vendor-core/docs-content/detection/spill.md +7 -0
  265. package/vendor-core/docs-content/detection/strag.md +5 -0
  266. package/vendor-core/docs-content/detection/tiny.md +4 -0
  267. package/vendor-core/docs-content/detection/util.md +4 -0
  268. package/vendor-core/docs-content/diagrams/aqe-loop.dark.svg +1 -0
  269. package/vendor-core/docs-content/diagrams/aqe-loop.svg +1 -0
  270. package/vendor-core/docs-content/diagrams/broadcast-vs-shuffle.dark.svg +1 -0
  271. package/vendor-core/docs-content/diagrams/broadcast-vs-shuffle.svg +1 -0
  272. package/vendor-core/docs-content/diagrams/cache-lifecycle.dark.svg +1 -0
  273. package/vendor-core/docs-content/diagrams/cache-lifecycle.svg +1 -0
  274. package/vendor-core/docs-content/diagrams/cold-start-timeline.dark.svg +1 -0
  275. package/vendor-core/docs-content/diagrams/cold-start-timeline.svg +1 -0
  276. package/vendor-core/docs-content/diagrams/columnar-layout.dark.svg +1 -0
  277. package/vendor-core/docs-content/diagrams/columnar-layout.svg +1 -0
  278. package/vendor-core/docs-content/diagrams/container-memory.dark.svg +1 -0
  279. package/vendor-core/docs-content/diagrams/container-memory.svg +1 -0
  280. package/vendor-core/docs-content/diagrams/dag-stages.dark.svg +1 -0
  281. package/vendor-core/docs-content/diagrams/dag-stages.svg +1 -0
  282. package/vendor-core/docs-content/diagrams/driver-executor.dark.svg +1 -0
  283. package/vendor-core/docs-content/diagrams/driver-executor.svg +1 -0
  284. package/vendor-core/docs-content/diagrams/duplicate-plan-subtree.dark.svg +1 -0
  285. package/vendor-core/docs-content/diagrams/duplicate-plan-subtree.svg +1 -0
  286. package/vendor-core/docs-content/diagrams/join-strategy.dark.svg +1 -0
  287. package/vendor-core/docs-content/diagrams/join-strategy.svg +1 -0
  288. package/vendor-core/docs-content/diagrams/memory-borrowing.dark.svg +1 -0
  289. package/vendor-core/docs-content/diagrams/memory-borrowing.svg +1 -0
  290. package/vendor-core/docs-content/diagrams/memory-regions.dark.svg +1 -0
  291. package/vendor-core/docs-content/diagrams/memory-regions.svg +1 -0
  292. package/vendor-core/docs-content/diagrams/repartition-vs-coalesce.dark.svg +1 -0
  293. package/vendor-core/docs-content/diagrams/repartition-vs-coalesce.svg +1 -0
  294. package/vendor-core/docs-content/diagrams/retry-escalation-ladder.dark.svg +1 -0
  295. package/vendor-core/docs-content/diagrams/retry-escalation-ladder.svg +1 -0
  296. package/vendor-core/docs-content/diagrams/shuffle-map-reduce.dark.svg +1 -0
  297. package/vendor-core/docs-content/diagrams/shuffle-map-reduce.svg +1 -0
  298. package/vendor-core/docs-content/diagrams/spill-classification.dark.svg +1 -0
  299. package/vendor-core/docs-content/diagrams/spill-classification.svg +1 -0
  300. package/vendor-core/docs-content/diagrams/udf-execution-models.dark.svg +1 -0
  301. package/vendor-core/docs-content/diagrams/udf-execution-models.svg +1 -0
  302. package/vendor-core/docs-content/tuning/broadcast-sizing.md +78 -0
  303. package/vendor-core/docs-content/tuning/cold-start.md +81 -0
  304. package/vendor-core/docs-content/tuning/duplicate-plan-subtree.md +45 -0
  305. package/vendor-core/docs-content/tuning/failures.md +124 -0
  306. package/vendor-core/docs-content/tuning/gc.md +110 -0
  307. package/vendor-core/docs-content/tuning/job-failure-rate.md +101 -0
  308. package/vendor-core/docs-content/tuning/memory-utilization.md +58 -0
  309. package/vendor-core/docs-content/tuning/retry-waste.md +90 -0
  310. package/vendor-core/docs-content/tuning/shuffle.md +154 -0
  311. package/vendor-core/docs-content/tuning/skew.md +123 -0
  312. package/vendor-core/docs-content/tuning/slow-host.md +117 -0
  313. package/vendor-core/docs-content/tuning/small-files.md +99 -0
  314. package/vendor-core/docs-content/tuning/spill.md +114 -0
  315. package/vendor-core/docs-content/tuning/straggler.md +103 -0
  316. package/vendor-core/docs-content/tuning/tiny-tasks.md +94 -0
  317. package/vendor-core/docs-content/tuning/utilization.md +90 -0
  318. package/vendor-core/docs-site-config.js +10 -17
  319. package/vendor-core/efficiency-model.js +7 -13
  320. package/vendor-core/etl-phases.js +3 -5
  321. package/vendor-core/event-handlers.js +232 -134
  322. package/vendor-core/event-schemas.js +48 -114
  323. package/vendor-core/evidence-availability.js +5 -10
  324. package/vendor-core/evidence-report.js +72 -122
  325. package/vendor-core/export-data.js +48 -0
  326. package/vendor-core/finding-action-label.js +4 -10
  327. package/vendor-core/finding-filter-predicate.js +3 -7
  328. package/vendor-core/finding-generic-recommendation.js +112 -0
  329. package/vendor-core/finding-names.js +51 -0
  330. package/vendor-core/format-utils.js +112 -38
  331. package/vendor-core/impact-band.js +18 -24
  332. package/vendor-core/impact-estimator.js +38 -74
  333. package/vendor-core/ingest.js +7 -13
  334. package/vendor-core/job-groups.js +3 -6
  335. package/vendor-core/list-runs.js +278 -0
  336. package/vendor-core/load-vendored.js +6 -12
  337. package/vendor-core/log-header-peek.js +81 -0
  338. package/vendor-core/lz4-block.js +4 -6
  339. package/vendor-core/mcp-server-factory.js +38 -8
  340. package/vendor-core/mcp-tools.js +105 -76
  341. package/vendor-core/model-assembler.js +8 -16
  342. package/vendor-core/occupancy.js +5 -9
  343. package/vendor-core/parser-worker.js +18 -27
  344. package/vendor-core/plan-dot.js +2 -5
  345. package/vendor-core/plan-duration-attribution.js +78 -29
  346. package/vendor-core/plan-graph-model.js +126 -69
  347. package/vendor-core/plan-node-detail.js +31 -17
  348. package/vendor-core/plan-summary.js +19 -8
  349. package/vendor-core/recommendation-rollup.js +35 -39
  350. package/vendor-core/redact.js +72 -16
  351. package/vendor-core/rolling-log-reassembly.js +4 -6
  352. package/vendor-core/run-comparison.js +65 -70
  353. package/vendor-core/scaling-sim.js +5 -7
  354. package/vendor-core/session-snapshot.js +1 -1
  355. package/vendor-core/shs-fetch.js +4 -6
  356. package/vendor-core/shs-load.js +9 -13
  357. package/vendor-core/shs-request.js +1 -1
  358. package/vendor-core/stage-quantiles.js +14 -0
  359. package/vendor-core/types.js +78 -18
  360. package/vendor-core/wasted-core-hours.js +7 -12
@@ -0,0 +1 @@
1
+ import{_ as t,o as a,c as o,a5 as r}from"./chunks/framework.DSg0KOwT.js";const p=JSON.parse('{"title":"Anti-Patterns","description":"","frontmatter":{"title":"Anti-Patterns"},"headers":[],"relativePath":"tuning-reference/anti-patterns.md","filePath":"tuning-reference/anti-patterns.md"}'),s={name:"tuning-reference/anti-patterns.md"};function n(i,e,f,l,c,d){return a(),o("div",null,[...e[0]||(e[0]=[r('<h1 id="anti-patterns" tabindex="-1">Anti-Patterns <a class="header-anchor" href="#anti-patterns" aria-label="Permalink to &quot;Anti-Patterns {#anti-patterns}&quot;">​</a></h1><p>Practitioner checklists on Spark performance converge on a recurring handful of mistakes rather than exotic ones. Each entry below covers what the mistake is, how to detect it, why it hurts, and how to fix it.</p><ol><li><p><strong>Staying on the RDD API instead of DataFrame/Dataset</strong></p><ul><li><strong>What:</strong> Building jobs on the RDD API (or RDD-based MLlib) instead of the DataFrame/Dataset API.</li><li><strong>Detect:</strong> Code that drives its logic through <code>.map</code>/<code>.filter</code> lambdas over RDDs, or that calls into the RDD-based MLlib API.</li><li><strong>Why:</strong> RDDs are lambda-driven, so Spark cannot see inside the closures. It can&#39;t apply predicate pushdown, filter reordering, Adaptive Query Execution, or cost-based join reordering. The RDD-based MLlib API is in maintenance mode for the same reason.<sup class="footnote-ref"><a href="#fn1" id="fnref1">[1]</a></sup></li><li><strong>Fix:</strong> Use the DataFrame/Dataset API so Catalyst can optimize the query plan.</li></ul></li><li><p><strong>Reading CSV/JSON without an explicit schema</strong></p><ul><li><strong>What:</strong> Reading raw CSV or JSON files without supplying an explicit schema.</li><li><strong>Detect:</strong> A read step that spends noticeable time before any real work starts, because Spark is inferring types from the data itself.</li><li><strong>Why:</strong> Schema inference forces a full scan of the data just to determine types, and CSV/JSON don&#39;t support the column pruning, predicate pushdown, or stats-based file skipping that a columnar format does.<sup class="footnote-ref"><a href="#fn1" id="fnref1:1">[1:1]</a></sup></li><li><strong>Fix:</strong> Supply an explicit schema, or prefer Parquet for analytical workloads, where column pruning, predicate pushdown, and stats-based file skipping work out of the box.<sup class="footnote-ref"><a href="#fn1" id="fnref1:2">[1:2]</a></sup></li></ul></li><li><p><strong>Compressing input files with a non-splittable codec (GZIP)</strong></p><ul><li><strong>What:</strong> Storing input data as large GZIP files.</li><li><strong>Detect:</strong> One task/executor taking far longer than the rest on a stage reading GZIP-compressed input, while other executors sit idle.</li><li><strong>Why:</strong> A GZIP file can&#39;t be split across executors, so a single node has to decompress the whole file alone.<sup class="footnote-ref"><a href="#fn1" id="fnref1:3">[1:3]</a></sup></li><li><strong>Fix:</strong> Use a splittable codec instead: Snappy, LZ4, or ZSTD.<sup class="footnote-ref"><a href="#fn1" id="fnref1:4">[1:4]</a></sup></li></ul></li><li><p><strong>Storing data as JSON instead of a binary columnar/row format</strong></p><ul><li><strong>What:</strong> Persisting datasets as JSON rather than a binary columnar or row format.</li><li><strong>Detect:</strong> Read-heavy jobs where a large, recurring share of stage time goes to parsing the same JSON structures on every read.</li><li><strong>Why:</strong> JSON has to be re-parsed from text on every single read.<sup class="footnote-ref"><a href="#fn2" id="fnref2">[2]</a></sup></li><li><strong>Fix:</strong> Store data as Avro, Parquet, Thrift, or Protobuf structs in a sequence file to avoid the repeated parsing cost.<sup class="footnote-ref"><a href="#fn2" id="fnref2:1">[2:1]</a></sup></li></ul></li><li><p><strong>Not registering custom classes with Kryo</strong></p><ul><li><strong>What:</strong> Using Kryo serialization without registering the application&#39;s custom classes.</li><li><strong>Detect:</strong> Larger-than-expected serialized record sizes, and poor efficiency when using a serialized cache storage level such as <code>MEMORY_SER</code>.</li><li><strong>Why:</strong> Unregistered classes increase serialized-record size and hurt serialized cache storage levels.<sup class="footnote-ref"><a href="#fn2" id="fnref2:2">[2:2]</a></sup></li><li><strong>Fix:</strong> Register the application&#39;s custom classes with Kryo.</li></ul></li><li><p><strong>Calling <code>collect()</code> on a large DataFrame</strong></p><ul><li><p><strong>What:</strong> Materializing an entire DataFrame into driver memory with <code>collect()</code>.</p></li><li><p><strong>Detect:</strong> A driver <code>OutOfMemoryError</code>, or a job that aborts once <code>spark.driver.maxResultSize</code> (default 1 GB) is exceeded. Spark also logs a warning once a single task&#39;s serialized result passes roughly 1 MB.<sup class="footnote-ref"><a href="#fn3" id="fnref3">[3]</a></sup><sup class="footnote-ref"><a href="#fn1" id="fnref1:5">[1:5]</a></sup></p></li><li><p><strong>Why:</strong> <code>collect()</code> gathers every <code>Row</code> from every partition into a single in-memory <code>Array</code> that lives entirely on the driver, not the cluster. The driver JVM has to hold the whole result set in its own heap.<sup class="footnote-ref"><a href="#fn3" id="fnref3:1">[3:1]</a></sup><sup class="footnote-ref"><a href="#fn4" id="fnref4">[4]</a></sup><sup class="footnote-ref"><a href="#fn5" id="fnref5">[5]</a></sup><sup class="footnote-ref"><a href="#fn6" id="fnref6">[6]</a></sup></p></li><li><p><strong>Fix:</strong> Use aggregations or <code>take(n)</code> instead of collecting the full result set.</p><blockquote><p><strong>PySpark:</strong> <code>toPandas()</code> performs the same driver-side collection as <code>collect()</code> for Python users, so avoid it on large DataFrames too.<sup class="footnote-ref"><a href="#fn6" id="fnref6:1">[6:1]</a></sup></p></blockquote></li></ul></li><li><p><strong>Assuming <code>cache()</code> + <code>count()</code> guarantees the DataFrame stays fully cached</strong></p><ul><li><strong>What:</strong> Treating <code>df.cache(); df.count()</code> as a guarantee that the DataFrame remains fully persisted for later actions.</li><li><strong>Detect:</strong> Later actions on a supposedly-cached DataFrame recompute from source instead of hitting the cache. There&#39;s no built-in visibility into how much of a DataFrame is still actually cached, so this typically only surfaces as unexpectedly slow recomputation.<sup class="footnote-ref"><a href="#fn7" id="fnref7">[7]</a></sup></li><li><strong>Why:</strong> <code>count()</code> does force every partition to be computed and attempted for caching, unlike <code>take</code>/<code>limit</code>, which only touch the partitions they need.<sup class="footnote-ref"><a href="#fn5" id="fnref5:1">[5:1]</a></sup><sup class="footnote-ref"><a href="#fn7" id="fnref7:1">[7:1]</a></sup> But partitions can&#39;t be fractionally cached: if there isn&#39;t room for all of them, the ones that don&#39;t fit are silently dropped and recomputed on next access.<sup class="footnote-ref"><a href="#fn5" id="fnref5:2">[5:2]</a></sup> Cached blocks also compete with execution for the same memory pool and can be evicted under pressure without warning, and caching is tied to the <em>analyzed</em> (pre-optimization) logical plan, so a semantically identical query with a different analyzed plan bypasses the cache entirely and recomputes from source.<sup class="footnote-ref"><a href="#fn7" id="fnref7:2">[7:2]</a></sup> Losing an executor drops any cached block that wasn&#39;t stored with a replicated storage level like <code>MEMORY_AND_DISK_2</code>.<sup class="footnote-ref"><a href="#fn7" id="fnref7:3">[7:3]</a></sup></li><li><strong>Fix:</strong> Keep using <code>.cache()</code> followed by <code>.count()</code> to force materialization, but don&#39;t assume that guarantees persistence (there&#39;s no built-in way to confirm how much of the DataFrame is actually cached<sup class="footnote-ref"><a href="#fn7" id="fnref7:4">[7:4]</a></sup>), and use a replicated storage level if losing a cached partition to executor failure would be costly.</li></ul></li><li><p><strong>Row-at-a-time Python UDFs</strong></p><ul><li><strong>What:</strong> Writing Python UDFs that operate one row at a time instead of vectorized Pandas UDFs.</li><li><strong>Detect:</strong> A UDF-heavy stage where most of the time goes to serialization rather than actual computation.<sup class="footnote-ref"><a href="#fn8" id="fnref8">[8]</a></sup></li><li><strong>Why:</strong> The overhead is serialization plus per-row JVM↔Python data movement, not raw Python interpreter speed. Row-at-a-time UDFs &quot;suffer from high serialization and invocation overhead.&quot;<sup class="footnote-ref"><a href="#fn9" id="fnref9">[9]</a></sup> RDD-era Python UDFs pay a <em>double</em> serialization cost: Java/Scala objects are serialized, then re-serialized to Python via <code>cloudpickle</code> and back on every call.<sup class="footnote-ref"><a href="#fn8" id="fnref8:1">[8:1]</a></sup> A Databricks benchmark on a 10M-row DataFrame showed vectorized Pandas UDFs beating row-at-a-time UDFs &quot;across the board, ranging from 3x to over 100x.&quot;<sup class="footnote-ref"><a href="#fn9" id="fnref9:1">[9:1]</a></sup></li><li><strong>Fix:</strong> Use vectorized Pandas UDFs instead of row-at-a-time UDFs: once data is transferred via Arrow, there&#39;s no need to serialize/pickle it, since it&#39;s already in a format consumable by the Python process.<sup class="footnote-ref"><a href="#fn5" id="fnref5:3">[5:3]</a></sup></li></ul></li><li><p><strong>Leaving <code>spark.sql.shuffle.partitions</code> at its default of 200</strong></p><ul><li><strong>What:</strong> Never tuning <code>spark.sql.shuffle.partitions</code> away from its default of 200 for wide transformations (<code>join</code>, <code>groupBy</code>, aggregations).</li><li><strong>Detect:</strong> In the Spark UI Stages tab, too few partitions for the data size shows up as &quot;Spill (Memory)&quot; or &quot;Spill (Disk)&quot; entries against a stage; too many partitions for the data size shows up as a large task/stage count where each task&#39;s duration is dominated by scheduling and bookkeeping, often paired with a large number of tiny output files.<sup class="footnote-ref"><a href="#fn10" id="fnref10">[10]</a></sup></li><li><strong>Why:</strong> 200 is a fixed default regardless of whether you&#39;re joining 5MB or 5TB.<sup class="footnote-ref"><a href="#fn10" id="fnref10:1">[10:1]</a></sup><sup class="footnote-ref"><a href="#fn11" id="fnref11">[11]</a></sup> Too few partitions means each one holds more data than fits comfortably in executor memory, which increases load per executor and leads to spills once partition size exceeds available memory.<sup class="footnote-ref"><a href="#fn10" id="fnref10:2">[10:2]</a></sup> Too many partitions on a small dataset can shrink tasks to around ten rows each, so &quot;most of your CPUs will just be sitting there doing nothing.&quot;<sup class="footnote-ref"><a href="#fn10" id="fnref10:3">[10:3]</a></sup></li><li><strong>Fix:</strong> Increase <code>spark.sql.shuffle.partitions</code> if tasks are processing multiple GB each and spilling; decrease it if tasks finish in a couple of seconds and write tiny files. A rough starting target is 100–200 MB of data per task, tuned per dataset.<sup class="footnote-ref"><a href="#fn10" id="fnref10:4">[10:4]</a></sup></li></ul></li><li><p><strong>Broadcasting a table that&#39;s too large</strong></p><ul><li><strong>What:</strong> Forcing or allowing a broadcast join on a table that&#39;s too large to broadcast safely.</li><li><strong>Detect:</strong> A driver <code>OutOfMemoryError</code> during a broadcast join.</li><li><strong>Why:</strong> The small side of the join has to be collected onto the driver before it can be broadcast to every executor. That collection step is the same driver-memory operation as <code>.collect()</code>, and it&#39;s what fails, not executor-side replication or a serialization timeout: &quot;if you try to broadcast something too large, you can crash your driver node (because that collect is expensive).&quot;<sup class="footnote-ref"><a href="#fn4" id="fnref4:1">[4:1]</a></sup></li><li><strong>Fix:</strong> Use <code>spark.sql.autoBroadcastJoinThreshold</code> to control the maximum size Spark will broadcast, or increase driver memory.<sup class="footnote-ref"><a href="#fn4" id="fnref4:2">[4:2]</a></sup></li></ul></li><li><p><strong>Reaching for <code>coalesce(1)</code> before every single-file write</strong></p><ul><li><strong>What:</strong> Always using <code>coalesce(1)</code> instead of <code>repartition(1)</code> when a single output file is needed.</li><li><strong>Detect:</strong> The entire upstream stage (including filtering/transformation work that would otherwise run in parallel) collapses onto a single task/executor, leaving the rest of the cluster idle.</li><li><strong>Why:</strong> <code>coalesce</code> is a narrow transformation, so it &quot;causes the upstream partitions in the entire stage to execute with the level of parallelism assigned by coalesce,&quot; fusing all preceding work down to one task.<sup class="footnote-ref"><a href="#fn6" id="fnref6:2">[6:2]</a></sup> <code>repartition</code> triggers a full shuffle instead, which inserts an explicit shuffle boundary so the upstream stage keeps its original parallelism.<sup class="footnote-ref"><a href="#fn4" id="fnref4:3">[4:3]</a></sup><sup class="footnote-ref"><a href="#fn6" id="fnref6:3">[6:3]</a></sup></li><li><strong>Fix:</strong> Use <code>repartition(1)</code> when there is meaningful upstream computation you don&#39;t want collapsed onto a single executor; reserve <code>coalesce(1)</code> for when the upstream stage is already cheap and paying for a shuffle would be wasted cost.<sup class="footnote-ref"><a href="#fn6" id="fnref6:4">[6:4]</a></sup></li></ul></li><li><p><strong>Not verifying predicate pushdown is actually happening</strong></p><ul><li><strong>What:</strong> Assuming filters and column projections are pushed down to the data source without checking the physical plan.</li><li><strong>Detect:</strong> Run <code>.explain()</code> and look at the <code>Scan</code> node for a <code>PushedFilters</code> marker. For example, <code>*Scan JDBCRel... PushedFilters: [*In(DEST_COUNTRY_NAME, [Anguilla, Sweden])]</code>. Column pruning is verified the same way: a <code>.select()</code> on one column should show a narrowed <code>ReadSchema</code> at the scan rather than a full-table scan.<sup class="footnote-ref"><a href="#fn4" id="fnref4:4">[4:4]</a></sup></li><li><strong>Why:</strong> Spark pushes down simple filters (column equality, <code>IN</code>, <code>IS NULL</code>) to JDBC sources automatically, but anything with a computed column or cast won&#39;t push down and gets evaluated in Spark after the full read.<sup class="footnote-ref"><a href="#fn1" id="fnref1:6">[1:6]</a></sup> Some sources also only partially handle a filter and leave Spark to re-evaluate it as a safety mechanism; that&#39;s still a &quot;good&quot; pushdown since the amount of data read is reduced.<sup class="footnote-ref"><a href="#fn6" id="fnref6:5">[6:5]</a></sup></li><li><strong>Fix:</strong> Check <code>explain</code> output (paired with <code>printSchema</code>)<sup class="footnote-ref"><a href="#fn6" id="fnref6:6">[6:6]</a></sup> for <code>PushedFilters</code> after writing a filter, and rewrite filters that use casts or computed columns as simple predicates where possible so they can push down.<sup class="footnote-ref"><a href="#fn4" id="fnref4:5">[4:5]</a></sup><sup class="footnote-ref"><a href="#fn1" id="fnref1:7">[1:7]</a></sup></li></ul></li></ol><h2 id="sources" tabindex="-1">Sources <a class="header-anchor" href="#sources" aria-label="Permalink to &quot;Sources&quot;">​</a></h2><hr class="footnotes-sep"><section class="footnotes"><ol class="footnotes-list"><li id="fn1" class="footnote-item"><p><a href="https://luminousmen.com/post/the-apache-spark-optimization-checklist" target="_blank" rel="noreferrer">The Apache Spark Optimization Checklist</a> <a href="#fnref1" class="footnote-backref">↩︎</a> <a href="#fnref1:1" class="footnote-backref">↩︎</a> <a href="#fnref1:2" class="footnote-backref">↩︎</a> <a href="#fnref1:3" class="footnote-backref">↩︎</a> <a href="#fnref1:4" class="footnote-backref">↩︎</a> <a href="#fnref1:5" class="footnote-backref">↩︎</a> <a href="#fnref1:6" class="footnote-backref">↩︎</a> <a href="#fnref1:7" class="footnote-backref">↩︎</a></p></li><li id="fn2" class="footnote-item"><p><a href="https://blog.cloudera.com/how-to-tune-your-apache-spark-jobs-part-2/" target="_blank" rel="noreferrer">How to Tune Your Apache Spark Jobs (Part 2)</a> <a href="#fnref2" class="footnote-backref">↩︎</a> <a href="#fnref2:1" class="footnote-backref">↩︎</a> <a href="#fnref2:2" class="footnote-backref">↩︎</a></p></li><li id="fn3" class="footnote-item"><p><em>Advanced Analytics with PySpark</em>, Tandon, Ryza, Laserson et al., ch. 2 <a href="#fnref3" class="footnote-backref">↩︎</a> <a href="#fnref3:1" class="footnote-backref">↩︎</a></p></li><li id="fn4" class="footnote-item"><p><em>Spark: The Definitive Guide</em>, Chambers &amp; Zaharia, chs. 5, 8, 9, 18, 19 <a href="#fnref4" class="footnote-backref">↩︎</a> <a href="#fnref4:1" class="footnote-backref">↩︎</a> <a href="#fnref4:2" class="footnote-backref">↩︎</a> <a href="#fnref4:3" class="footnote-backref">↩︎</a> <a href="#fnref4:4" class="footnote-backref">↩︎</a> <a href="#fnref4:5" class="footnote-backref">↩︎</a></p></li><li id="fn5" class="footnote-item"><p><em>Learning Spark, 2nd Edition</em>, Damji, Wenig, Das &amp; Lee, chs. 3, 5, 7 <a href="#fnref5" class="footnote-backref">↩︎</a> <a href="#fnref5:1" class="footnote-backref">↩︎</a> <a href="#fnref5:2" class="footnote-backref">↩︎</a> <a href="#fnref5:3" class="footnote-backref">↩︎</a></p></li><li id="fn6" class="footnote-item"><p><em>High Performance Spark, 2nd Edition</em>, Karau, Polak &amp; Warren, chs. 5, 7 <a href="#fnref6" class="footnote-backref">↩︎</a> <a href="#fnref6:1" class="footnote-backref">↩︎</a> <a href="#fnref6:2" class="footnote-backref">↩︎</a> <a href="#fnref6:3" class="footnote-backref">↩︎</a> <a href="#fnref6:4" class="footnote-backref">↩︎</a> <a href="#fnref6:5" class="footnote-backref">↩︎</a> <a href="#fnref6:6" class="footnote-backref">↩︎</a></p></li><li id="fn7" class="footnote-item"><p><a href="https://luminousmen.com/post/explaining-the-mechanics-of-spark-caching" target="_blank" rel="noreferrer">Explaining the Mechanics of Spark Caching</a> <a href="#fnref7" class="footnote-backref">↩︎</a> <a href="#fnref7:1" class="footnote-backref">↩︎</a> <a href="#fnref7:2" class="footnote-backref">↩︎</a> <a href="#fnref7:3" class="footnote-backref">↩︎</a> <a href="#fnref7:4" class="footnote-backref">↩︎</a></p></li><li id="fn8" class="footnote-item"><p><a href="https://luminousmen.com/post/spark-tips-dataframe-api" target="_blank" rel="noreferrer">Spark Tips: DataFrame API</a> <a href="#fnref8" class="footnote-backref">↩︎</a> <a href="#fnref8:1" class="footnote-backref">↩︎</a></p></li><li id="fn9" class="footnote-item"><p><a href="https://www.databricks.com/blog/2017/10/30/introducing-vectorized-udfs-for-pyspark.html" target="_blank" rel="noreferrer">Introducing Vectorized UDFs for PySpark</a> <a href="#fnref9" class="footnote-backref">↩︎</a> <a href="#fnref9:1" class="footnote-backref">↩︎</a></p></li><li id="fn10" class="footnote-item"><p><a href="https://luminousmen.com/post/spark-partitions" target="_blank" rel="noreferrer">Spark Partitions</a> <a href="#fnref10" class="footnote-backref">↩︎</a> <a href="#fnref10:1" class="footnote-backref">↩︎</a> <a href="#fnref10:2" class="footnote-backref">↩︎</a> <a href="#fnref10:3" class="footnote-backref">↩︎</a> <a href="#fnref10:4" class="footnote-backref">↩︎</a></p></li><li id="fn11" class="footnote-item"><p><a href="https://spark.apache.org/docs/latest/configuration.html" target="_blank" rel="noreferrer">Configuration — Spark</a> <a href="#fnref11" class="footnote-backref">↩︎</a></p></li></ol></section>',6)])])}const u=t(s,[["render",n]]);export{p as __pageData,u as default};
@@ -0,0 +1 @@
1
+ import{_ as t,o as a,c as o,a5 as r}from"./chunks/framework.DSg0KOwT.js";const p=JSON.parse('{"title":"Anti-Patterns","description":"","frontmatter":{"title":"Anti-Patterns"},"headers":[],"relativePath":"tuning-reference/anti-patterns.md","filePath":"tuning-reference/anti-patterns.md"}'),s={name:"tuning-reference/anti-patterns.md"};function n(i,e,f,l,c,d){return a(),o("div",null,[...e[0]||(e[0]=[r("",6)])])}const u=t(s,[["render",n]]);export{p as __pageData,u as default};
@@ -0,0 +1 @@
1
+ import{_ as a,o as t,c as s,a5 as o}from"./chunks/framework.DSg0KOwT.js";const n="../assets/aqe-loop.IwQSATHw.svg",r="../assets/aqe-loop.dark.DGbaxqJE.svg",m=JSON.parse('{"title":"Adaptive Query Execution","description":"","frontmatter":{"title":"Adaptive Query Execution"},"headers":[],"relativePath":"tuning-reference/aqe.md","filePath":"tuning-reference/aqe.md"}'),i={name:"tuning-reference/aqe.md"};function f(l,e,c,h,d,p){return t(),s("div",null,[...e[0]||(e[0]=[o('<h1 id="aqe" tabindex="-1">Adaptive Query Execution <a class="header-anchor" href="#aqe" aria-label="Permalink to &quot;Adaptive Query Execution {#aqe}&quot;">​</a></h1><h2 id="what-it-is" tabindex="-1">What it is <a class="header-anchor" href="#what-it-is" aria-label="Permalink to &quot;What it is&quot;">​</a></h2><p>The static Catalyst optimizer plans a query once, before execution, using cost estimates derived from static statistics: row counts, min/max, NDVs, or defaults when statistics are missing.<sup class="footnote-ref"><a href="#fn1" id="fnref1">[1]</a></sup> Adaptive Query Execution (AQE), shipped in Spark 3.0, instead re-optimizes the plan mid-execution using runtime statistics gathered from completed &quot;query stages&quot;: the sections of a plan bounded by shuffle or broadcast exchange materialization points.<sup class="footnote-ref"><a href="#fn2" id="fnref2">[2]</a></sup> A shuffle or broadcast forces Spark to materialize its input before continuing, which makes it a natural checkpoint: once one or more leaf stages finish, AQE marks them complete, updates the logical plan with the real (not estimated) statistics, and reruns a selected set of logical and physical optimization rules (including AQE-specific rules such as partition coalescing and skew-join handling) before executing the next stages.<sup class="footnote-ref"><a href="#fn2" id="fnref2:1">[2:1]</a></sup></p><p>Shuffle statistics only become available once a stage is <em>fully</em> materialized, not incrementally as individual map tasks finish. A stage&#39;s successor can only proceed once every parallel process producing that stage&#39;s output has completed, which is exactly why materialization points are the reoptimization opportunity: it&#39;s the moment when statistics on all of a stage&#39;s partitions are known and the next stage hasn&#39;t started yet.<sup class="footnote-ref"><a href="#fn2" id="fnref2:2">[2:2]</a></sup> AQE kicks off all leaf stages (the ones with no upstream dependency) first; as each one finishes, the framework marks it complete, updates the plan, and re-optimizes before launching whichever next stages now have all their children materialized. This execute→reoptimize→execute loop repeats (once per completed stage, not just once) until the whole query finishes, so the number of reoptimizations scales with how many shuffle/broadcast boundaries the plan has.<sup class="footnote-ref"><a href="#fn2" id="fnref2:3">[2:3]</a></sup> The Spark SQL config description for <code>spark.sql.adaptive.enabled</code> frames this the same way: AQE re-optimizes &quot;the query plan in the middle of query execution, based on accurate runtime statistics.&quot;<sup class="footnote-ref"><a href="#fn3" id="fnref3">[3]</a></sup></p><img class="light-only" src="'+n+'" alt="AQE runs a loop that executes leaf stages, materializes at a shuffle or broadcast boundary, collects runtime statistics, re-applies rules to coalesce partitions and split skew and promote joins to broadcast, then launches the next stages until the plan is complete."><img class="dark-only" src="'+r+'" alt="AQE runs a loop that executes leaf stages, materializes at a shuffle or broadcast boundary, collects runtime statistics, re-applies rules to coalesce partitions and split skew and promote joins to broadcast, then launches the next stages until the plan is complete."><p>AQE re-optimizes three things the static optimizer cannot, because none of them are knowable before execution: the number of post-shuffle partitions (coalescing small partitions produced by wide transformations), the join strategy (converting a statically planned sort-merge join to a broadcast join once the actual materialized size of a join side is known), and skewed partitions in a shuffle join (splitting oversized partitions detected from shuffle file statistics).<sup class="footnote-ref"><a href="#fn2" id="fnref2:4">[2:4]</a></sup> High Performance Spark frames this as AQE using &quot;runtime information about the data it is processing along with the target output to go beyond static optimizations,&quot; with partitioning and join strategy singled out as the two biggest areas it impacts.<sup class="footnote-ref"><a href="#fn4" id="fnref4">[4]</a></sup> AQE also performs empty-relation propagation, replacing subqueries that turn out to be empty (impossible joins, empty unions) with a dummy empty <code>LocalRelation</code>, again something only knowable once data is materialized.<sup class="footnote-ref"><a href="#fn4" id="fnref4:1">[4:1]</a></sup></p><h2 id="how-it-s-detected" tabindex="-1">How it&#39;s detected <a class="header-anchor" href="#how-it-s-detected" aria-label="Permalink to &quot;How it&#39;s detected&quot;">​</a></h2><p>Because these are runtime decisions, none of them show up in a static <code>explain()</code> call; you have to run the query and check the Spark UI for the finalized, adapted plan.<sup class="footnote-ref"><a href="#fn4" id="fnref4:2">[4:2]</a></sup> The same caveat applies more broadly: since AQE adjusts plans at runtime, <code>explain()</code> might show one plan while the Spark UI shows a different one actually executed.<sup class="footnote-ref"><a href="#fn5" id="fnref5">[5]</a></sup></p><p>What you&#39;re looking for in that adaptive plan are the specific runtime rewrites AQE can make:</p><ul><li><strong>Partition coalescing.</strong> AQE combines <em>adjacent</em> small post-shuffle partitions into bigger ones by reading shuffle file statistics,<sup class="footnote-ref"><a href="#fn2" id="fnref2:5">[2:5]</a></sup> targeting <code>spark.sql.adaptive.advisoryPartitionSizeInBytes</code> (default 64 MB).<sup class="footnote-ref"><a href="#fn6" id="fnref6">[6]</a></sup> In a worked example, five post-shuffle partitions where three are small get coalesced into one, cutting the final-aggregation task count from five to three.<sup class="footnote-ref"><a href="#fn2" id="fnref2:6">[2:6]</a></sup></li><li><strong>Skew splitting.</strong> A partition is considered skewed if its size is larger than the median partition size times <code>spark.sql.adaptive.skewJoin.skewedPartitionFactor</code> (default 5.0) <em>and</em> larger than <code>spark.sql.adaptive.skewJoin.skewedPartitionThresholdInBytes</code> (default 256 MB).<sup class="footnote-ref"><a href="#fn7" id="fnref7">[7]</a></sup><sup class="footnote-ref"><a href="#fn8" id="fnref8">[8]</a></sup> The original design doc&#39;s worked example (table A join table B, where B&#39;s partition 0 is skewed) has the <code>OptimizeSkewedJoin</code> rule divide that partition into N smaller splits reading disjoint ranges of upstream map outputs, create N separate tasks each joining its slice of B&#39;s partition 0 against A&#39;s corresponding partition 0, then union the N results. That trades reading A&#39;s partition 0 N times against eliminating a straggler task.<sup class="footnote-ref"><a href="#fn7" id="fnref7:1">[7:1]</a></sup></li><li><strong>Join strategy promotion.</strong> When a join side&#39;s runtime-materialized size falls under the adaptive broadcast threshold, AQE converts a planned sort-merge join to a broadcast hash join. It reuses the shuffle output already written rather than re-materializing the build side. The perf-tuning guide frames the benefit as avoiding a re-sort of both join sides and reading the existing shuffle files locally instead, conditional on <code>spark.sql.adaptive.localShuffleReader.enabled</code>.<sup class="footnote-ref"><a href="#fn6" id="fnref6:1">[6:1]</a></sup> The same guide calls this conversion &quot;not as efficient as planning a broadcast hash join in the first place,&quot; consistent with reusing existing shuffle output rather than recomputing an equivalent build side from scratch.<sup class="footnote-ref"><a href="#fn6" id="fnref6:2">[6:2]</a></sup></li><li><strong>Local shuffle reads.</strong> Once that promotion happens, the side that no longer needs to be partitioned by join key can be read straight off the shuffle files each executor already has locally, instead of pulling blocks from remote executors over the network. This is what <code>spark.sql.adaptive.localShuffleReader.enabled</code> (default true since 3.0.0) does, and it applies whenever shuffle partitioning is no longer needed, such as after a sort-merge-to- broadcast conversion.<sup class="footnote-ref"><a href="#fn6" id="fnref6:3">[6:3]</a></sup><sup class="footnote-ref"><a href="#fn2" id="fnref2:7">[2:7]</a></sup></li></ul><p>Dynamic Partition Pruning is a separate mechanism worth distinguishing from AQE&#39;s loop when reading a plan: DPP inserts a <code>DynamicPruningSubquery</code>/<code>DynamicPruningExpression</code> node during logical optimization/physical planning, before execution starts, when an equi-join is on a partition column and pruning looks beneficial by static statistics.<sup class="footnote-ref"><a href="#fn9" id="fnref9">[9]</a></sup> Because DPP&#39;s calculation runs at planning time and AQE&#39;s runs later, mid-execution, off completed-stage statistics, the ordering implied is that DPP&#39;s pruning predicate is planned first, with AQE&#39;s adaptive rules applying afterward, during execution, on top of whatever DPP already pruned<sup class="footnote-ref"><a href="#fn9" id="fnref9:1">[9:1]</a></sup><sup class="footnote-ref"><a href="#fn2" id="fnref2:8">[2:8]</a></sup>, though the exact current-version integration between the two isn&#39;t detailed further in the available sources, so treat that ordering as directionally supported rather than exhaustively confirmed.</p><h2 id="why-it-matters" tabindex="-1">Why it matters <a class="header-anchor" href="#why-it-matters" aria-label="Permalink to &quot;Why it matters&quot;">​</a></h2><p>AQE exists precisely because partition sizing, join strategy, and skew are only knowable once real data has been materialized. That&#39;s the gap the static optimizer can&#39;t close on its own.<sup class="footnote-ref"><a href="#fn2" id="fnref2:9">[2:9]</a></sup><sup class="footnote-ref"><a href="#fn4" id="fnref4:3">[4:3]</a></sup> But it isn&#39;t a guaranteed win. High Performance Spark warns that &quot;while AQE generally does better than unoptimized code, there are cases, especially when targeting Iceberg tables, where AQE does more harm than good&quot;: AQE&#39;s partitioning-to-target-table matching is generally beneficial but can cause severe regressions if you&#39;ve already done deliberate work to avoid key skew via custom partitioning, since AQE&#39;s attempt to match the target table&#39;s write partitioning can undo that work, especially when there&#39;s a large amount of key skew.<sup class="footnote-ref"><a href="#fn4" id="fnref4:4">[4:4]</a></sup></p><p>A second, more general failure mode is bad input statistics feeding AQE&#39;s runtime decisions: if the source data has wrong stats (compressed JSON or Kafka streams are called out specifically), AQE can make things worse rather than better.<sup class="footnote-ref"><a href="#fn5" id="fnref5:1">[5:1]</a></sup> The same source&#39;s framing is that AQE &quot;isn&#39;t magic... it helps polish your plan; it doesn&#39;t design it for you. You still need good partitioning fundamentals&quot;. AQE can smooth over a reasonably partitioned plan but doesn&#39;t reliably rescue one built on bad upstream statistics or layout.<sup class="footnote-ref"><a href="#fn5" id="fnref5:2">[5:2]</a></sup></p><p>AQE&#39;s runtime re-optimization also doesn&#39;t reach into DataFrame caching, which is worth knowing so you don&#39;t reach for the wrong lever: <code>.cache()</code>/<code>.persist()</code> wrap the query&#39;s <em>analyzed</em> logical plan (the point after the analyzer phase but before optimization) in an <code>InMemoryRelation</code> node, and cache lookup compares the analyzed plan of the current query against what was cached, recomputing if they don&#39;t match exactly, even if the two queries would ultimately produce the same optimized physical plan.<sup class="footnote-ref"><a href="#fn10" id="fnref10">[10]</a></sup> Because AQE operates on the physical plan at runtime, well downstream of that analyzed-plan cache key, disabling AQE doesn&#39;t change whether a cached DataFrame&#39;s analyzed plan matches; it only changes how the physical/execution plan is chosen after that cache lookup already happened.<sup class="footnote-ref"><a href="#fn10" id="fnref10:1">[10:1]</a></sup></p><h2 id="how-to-fix-it" tabindex="-1">How to fix it <a class="header-anchor" href="#how-to-fix-it" aria-label="Permalink to &quot;How to fix it&quot;">​</a></h2><p>AQE&#39;s coalescing, skew-handling, and join-promotion rules only fire once <code>spark.sql.adaptive.enabled</code> is on, and each has its own knobs worth tuning rather than leaving at the defaults:</p><ul><li><code>spark.sql.adaptive.coalescePartitions.enabled</code> (default true) turns on partition coalescing; <code>spark.sql.adaptive.advisoryPartitionSizeInBytes</code> (default 64 MB) sets its target size.<sup class="footnote-ref"><a href="#fn6" id="fnref6:4">[6:4]</a></sup> <code>spark.sql.adaptive.coalescePartitions.minPartitionNum</code> bounds the minimum resulting parallelism,<sup class="footnote-ref"><a href="#fn4" id="fnref4:5">[4:5]</a></sup> and <code>spark.sql.adaptive.coalescePartitions.minPartitionSize</code> (default 1 MB, since 3.2) sets a floor on coalesced partition size for when the adaptively calculated target is too small.<sup class="footnote-ref"><a href="#fn6" id="fnref6:5">[6:5]</a></sup></li><li><code>spark.sql.adaptive.coalescePartitions.parallelismFirst</code> (default true, since 3.2) tells Spark to ignore the advisory target size entirely and instead calculate a (usually smaller) target based on cluster default parallelism, prioritizing task parallelism over hitting the 64 MB target. On a busy cluster, set it to <code>false</code> to respect the configured target size and avoid producing many small tasks.<sup class="footnote-ref"><a href="#fn6" id="fnref6:6">[6:6]</a></sup></li><li><code>spark.sql.adaptive.skewJoin.enabled</code> (default true) turns on skew-join handling for both sort-merge and shuffled hash joins.<sup class="footnote-ref"><a href="#fn8" id="fnref8:1">[8:1]</a></sup> Tune the detection thresholds via <code>spark.sql.adaptive.skewJoin.skewedPartitionFactor</code> (default 5.0) and <code>spark.sql.adaptive.skewJoin.skewedPartitionThresholdInBytes</code> (default 256 MB); set the byte threshold larger than <code>advisoryPartitionSizeInBytes</code>.<sup class="footnote-ref"><a href="#fn8" id="fnref8:2">[8:2]</a></sup> Use <code>spark.sql.adaptive.forceOptimizeSkewedJoin</code> (default false, since 3.3.0) to force the rule to fire even when it would introduce extra shuffle.<sup class="footnote-ref"><a href="#fn6" id="fnref6:7">[6:7]</a></sup></li><li><code>spark.sql.adaptive.localShuffleReader.enabled</code> (default true, since 3.0.0) keeps the local, per-mapper shuffle read enabled after a sort-merge-to-broadcast conversion.<sup class="footnote-ref"><a href="#fn6" id="fnref6:8">[6:8]</a></sup></li></ul><p>If AQE&#39;s target-table partition matching is regressing an Iceberg write that already has deliberate custom partitioning to avoid key skew, set the table&#39;s (or write-time) <code>write.distribution-mode</code> property to <code>none</code>, at the cost of a higher risk of many small files from unsorted/unhashed writers per partition.<sup class="footnote-ref"><a href="#fn4" id="fnref4:6">[4:6]</a></sup></p><h2 id="sources" tabindex="-1">Sources <a class="header-anchor" href="#sources" aria-label="Permalink to &quot;Sources&quot;">​</a></h2><hr class="footnotes-sep"><section class="footnotes"><ol class="footnotes-list"><li id="fn1" class="footnote-item"><p><a href="https://www.databricks.com/blog/2015/04/13/deep-dive-into-spark-sqls-catalyst-optimizer.html" target="_blank" rel="noreferrer">Deep Dive into Spark SQL&#39;s Catalyst Optimizer</a> <a href="#fnref1" class="footnote-backref">↩︎</a></p></li><li id="fn2" class="footnote-item"><p><a href="https://www.databricks.com/blog/2020/05/29/adaptive-query-execution-speeding-up-spark-sql-at-runtime.html" target="_blank" rel="noreferrer">Adaptive Query Execution: Speeding Up Spark SQL at Runtime</a> <a href="#fnref2" class="footnote-backref">↩︎</a> <a href="#fnref2:1" class="footnote-backref">↩︎</a> <a href="#fnref2:2" class="footnote-backref">↩︎</a> <a href="#fnref2:3" class="footnote-backref">↩︎</a> <a href="#fnref2:4" class="footnote-backref">↩︎</a> <a href="#fnref2:5" class="footnote-backref">↩︎</a> <a href="#fnref2:6" class="footnote-backref">↩︎</a> <a href="#fnref2:7" class="footnote-backref">↩︎</a> <a href="#fnref2:8" class="footnote-backref">↩︎</a> <a href="#fnref2:9" class="footnote-backref">↩︎</a></p></li><li id="fn3" class="footnote-item"><p><a href="https://raw.githubusercontent.com/apache/spark/v3.5.0/sql/catalyst/src/main/scala/org/apache/spark/sql/internal/SQLConf.scala" target="_blank" rel="noreferrer">SQLConf.scala</a> <a href="#fnref3" class="footnote-backref">↩︎</a></p></li><li id="fn4" class="footnote-item"><p><em>High Performance Spark, 2nd Edition</em>, Karau, Polak &amp; Warren, ch. 5 <a href="#fnref4" class="footnote-backref">↩︎</a> <a href="#fnref4:1" class="footnote-backref">↩︎</a> <a href="#fnref4:2" class="footnote-backref">↩︎</a> <a href="#fnref4:3" class="footnote-backref">↩︎</a> <a href="#fnref4:4" class="footnote-backref">↩︎</a> <a href="#fnref4:5" class="footnote-backref">↩︎</a> <a href="#fnref4:6" class="footnote-backref">↩︎</a></p></li><li id="fn5" class="footnote-item"><p><a href="https://luminousmen.com/post/spark-partitions" target="_blank" rel="noreferrer">Spark Partitions</a> <a href="#fnref5" class="footnote-backref">↩︎</a> <a href="#fnref5:1" class="footnote-backref">↩︎</a> <a href="#fnref5:2" class="footnote-backref">↩︎</a></p></li><li id="fn6" class="footnote-item"><p><a href="https://spark.apache.org/docs/latest/sql-performance-tuning.html" target="_blank" rel="noreferrer">Performance Tuning: Spark SQL, DataFrames and Datasets Guide</a> <a href="#fnref6" class="footnote-backref">↩︎</a> <a href="#fnref6:1" class="footnote-backref">↩︎</a> <a href="#fnref6:2" class="footnote-backref">↩︎</a> <a href="#fnref6:3" class="footnote-backref">↩︎</a> <a href="#fnref6:4" class="footnote-backref">↩︎</a> <a href="#fnref6:5" class="footnote-backref">↩︎</a> <a href="#fnref6:6" class="footnote-backref">↩︎</a> <a href="#fnref6:7" class="footnote-backref">↩︎</a> <a href="#fnref6:8" class="footnote-backref">↩︎</a></p></li><li id="fn7" class="footnote-item"><p><a href="https://issues.apache.org/jira/browse/SPARK-29544" target="_blank" rel="noreferrer">SPARK-29544: Optimize Skewed Join at Runtime</a> <a href="#fnref7" class="footnote-backref">↩︎</a> <a href="#fnref7:1" class="footnote-backref">↩︎</a></p></li><li id="fn8" class="footnote-item"><p><a href="https://spark.apache.org/docs/latest/configuration.html" target="_blank" rel="noreferrer">Configuration: Spark</a> <a href="#fnref8" class="footnote-backref">↩︎</a> <a href="#fnref8:1" class="footnote-backref">↩︎</a> <a href="#fnref8:2" class="footnote-backref">↩︎</a></p></li><li id="fn9" class="footnote-item"><p><a href="https://www.waitingforcode.com/apache-spark-sql/whats-new-apache-spark-3-dynamic-partition-pruning/read" target="_blank" rel="noreferrer">What&#39;s New in Apache Spark 3: Dynamic Partition Pruning</a> <a href="#fnref9" class="footnote-backref">↩︎</a> <a href="#fnref9:1" class="footnote-backref">↩︎</a></p></li><li id="fn10" class="footnote-item"><p><a href="https://luminousmen.com/post/explaining-the-mechanics-of-spark-caching" target="_blank" rel="noreferrer">Explaining the Mechanics of Spark Caching</a> <a href="#fnref10" class="footnote-backref">↩︎</a> <a href="#fnref10:1" class="footnote-backref">↩︎</a></p></li></ol></section>',23)])])}const g=a(i,[["render",f]]);export{m as __pageData,g as default};
@@ -0,0 +1 @@
1
+ import{_ as a,o as t,c as s,a5 as o}from"./chunks/framework.DSg0KOwT.js";const n="../assets/aqe-loop.IwQSATHw.svg",r="../assets/aqe-loop.dark.DGbaxqJE.svg",m=JSON.parse('{"title":"Adaptive Query Execution","description":"","frontmatter":{"title":"Adaptive Query Execution"},"headers":[],"relativePath":"tuning-reference/aqe.md","filePath":"tuning-reference/aqe.md"}'),i={name:"tuning-reference/aqe.md"};function f(l,e,c,h,d,p){return t(),s("div",null,[...e[0]||(e[0]=[o("",23)])])}const g=a(i,[["render",f]]);export{m as __pageData,g as default};
@@ -0,0 +1 @@
1
+ import{_ as a,o as t,c as o,a5 as s}from"./chunks/framework.DSg0KOwT.js";const r="../assets/broadcast-vs-shuffle.Db4WY1XK.svg",i="../assets/broadcast-vs-shuffle.dark.C7Bxs0mG.svg",b=JSON.parse('{"title":"Broadcast Sizing","description":"","frontmatter":{"title":"Broadcast Sizing"},"headers":[],"relativePath":"tuning-reference/bottleneck-broadcast-sizing.md","filePath":"tuning-reference/bottleneck-broadcast-sizing.md"}'),n={name:"tuning-reference/bottleneck-broadcast-sizing.md"};function d(l,e,c,h,f,p){return t(),o("div",null,[...e[0]||(e[0]=[s('<h1 id="bottleneck-broadcast-sizing" tabindex="-1">Broadcast Sizing <a class="header-anchor" href="#bottleneck-broadcast-sizing" aria-label="Permalink to &quot;Broadcast Sizing {#bottleneck-broadcast-sizing}&quot;">​</a></h1><p><span class="tag">BROADCASTSIZING</span></p><h2 id="what-it-is" tabindex="-1">What it is <a class="header-anchor" href="#what-it-is" aria-label="Permalink to &quot;What it is&quot;">​</a></h2><p>A join has two ways to move data. The <a href="./joins.html">shuffle join</a> forces all-to-all traffic across the cluster; the broadcast join ships one side out to every executor and skips the <a href="./shuffle.html">shuffle</a> entirely. <code>spark.sql.autoBroadcastJoinThreshold</code> is the size gate that picks between them: it sets the largest table, in bytes, that Spark will broadcast when planning a join, and it defaults to 10 MB.<sup class="footnote-ref"><a href="#fn1" id="fnref1">[1]</a></sup><sup class="footnote-ref"><a href="#fn2" id="fnref2">[2]</a></sup> Set it to <code>-1</code> and automatic broadcasting is off, so Spark always shuffles.<sup class="footnote-ref"><a href="#fn1" id="fnref1:1">[1:1]</a></sup></p><img class="light-only" src="'+r+'" alt="How the small side estimate against the broadcast threshold picks between broadcasting a copy to every executor for a local join and an all-to-all shuffle join, with the broadcast path&#39;s driver collect and executor memory risks noted."><img class="dark-only" src="'+i+'" alt="How the small side estimate against the broadcast threshold picks between broadcasting a copy to every executor for a local join and an all-to-all shuffle join, with the broadcast path&#39;s driver collect and executor memory risks noted."><p>Given a join&#39;s estimated table sizes and that gate, there are two ways the choice can go wrong:</p><ul><li><strong>underBroadcast</strong>: one side is small enough to broadcast but is being shuffled anyway. The join pays for an all-to-all exchange it could have avoided.</li><li><strong>overBroadcast</strong>: a side that is too large is being broadcast, which risks running the driver or the executors out of memory.</li></ul><h2 id="how-it-s-detected" tabindex="-1">How it&#39;s detected <a class="header-anchor" href="#how-it-s-detected" aria-label="Permalink to &quot;How it&#39;s detected&quot;">​</a></h2><p>Each join carries an estimated size for the side in question, which is compared against the broadcast threshold:</p><table tabindex="0"><thead><tr><th>Finding</th><th>Condition</th><th>What it means</th></tr></thead><tbody><tr><td><code>underBroadcast</code></td><td>small side is being shuffled instead of broadcast</td><td>a cheap broadcast join was left on the table</td></tr><tr><td><code>overBroadcast</code></td><td>an oversized side is being broadcast</td><td>the broadcast is a memory hazard</td></tr></tbody></table><h2 id="why-it-matters" tabindex="-1">Why it matters <a class="header-anchor" href="#why-it-matters" aria-label="Permalink to &quot;Why it matters&quot;">​</a></h2><p>Broadcasting the small side sends a full copy of that table to every executor. It pays off only when network bandwidth and executor memory are not the bottleneck; there, it is far cheaper than the shuffle it replaces.<sup class="footnote-ref"><a href="#fn3" id="fnref3">[3]</a></sup> That is the cost an <code>underBroadcast</code> finding is pointing at: a small dimension going through an all-to-all exchange when a copy on each executor would have run fast.</p><p><code>overBroadcast</code> is the opposite mistake, and it fails hard. Pushing a table that is too large through the broadcast path breaks in two ways: too-large task result errors and out-of-memory failures, both because the oversized table has to be collected on the driver and shipped whole to every executor.<sup class="footnote-ref"><a href="#fn4" id="fnref4">[4]</a></sup> The size gate exists for exactly this reason: a copy on every executor is harmless for a small table and a memory hazard for a large one.</p><h2 id="how-to-fix-it" tabindex="-1">How to fix it <a class="header-anchor" href="#how-to-fix-it" aria-label="Permalink to &quot;How to fix it&quot;">​</a></h2><p>For <code>underBroadcast</code>, let the small side broadcast. If its estimated size sits above the 10 MB default but still fits comfortably in executor memory, raise <code>spark.sql.autoBroadcastJoinThreshold</code> to cover it, or force the strategy with <code>broadcast(df)</code> on the small side.<sup class="footnote-ref"><a href="#fn1" id="fnref1:2">[1:2]</a></sup><sup class="footnote-ref"><a href="#fn5" id="fnref5">[5]</a></sup></p><p>For <code>overBroadcast</code>, keep the large side out of the broadcast path. Lower the threshold so the oversized table no longer qualifies, or drop the explicit <code>broadcast()</code> hint if one is forcing it, and let Spark fall back to a shuffle join.</p><h2 id="confidence" tabindex="-1">Confidence <a class="header-anchor" href="#confidence" aria-label="Permalink to &quot;Confidence&quot;">​</a></h2><p>Validated for both branches. <code>underBroadcast</code> and <code>overBroadcast</code> rest on the same well-documented mechanism: the broadcast threshold and its two failure modes are standard Spark behavior, so neither branch is experimental.</p><h2 id="limitations-false-positive-risk" tabindex="-1">Limitations / false-positive risk <a class="header-anchor" href="#limitations-false-positive-risk" aria-label="Permalink to &quot;Limitations / false-positive risk&quot;">​</a></h2><p>The comparison is only as good as the size estimate behind it. Table sizes come from statistics that may be stale or missing, so a join that looks mis-sized can be measuring the wrong number. A large broadcast can also be deliberate: an operator who has sized executor memory for it may broadcast a table well past the default on purpose.</p><hr class="footnotes-sep"><section class="footnotes"><ol class="footnotes-list"><li id="fn1" class="footnote-item"><p><a href="https://spark.apache.org/docs/latest/configuration.html" target="_blank" rel="noreferrer">Configuration (Spark)</a> <a href="#fnref1" class="footnote-backref">↩︎</a> <a href="#fnref1:1" class="footnote-backref">↩︎</a> <a href="#fnref1:2" class="footnote-backref">↩︎</a></p></li><li id="fn2" class="footnote-item"><p><a href="https://spark.apache.org/docs/latest/sql-performance-tuning.html" target="_blank" rel="noreferrer">Performance Tuning (Spark SQL, DataFrames and Datasets Guide)</a> <a href="#fnref2" class="footnote-backref">↩︎</a></p></li><li id="fn3" class="footnote-item"><p><em>Learning Spark, 2nd Edition</em>, Damji, Wenig, Das, Lee, ch. 7 <a href="#fnref3" class="footnote-backref">↩︎</a></p></li><li id="fn4" class="footnote-item"><p><em>High Performance Spark, 2nd Edition</em>, Karau, Polak &amp; Warren, ch. 6 <a href="#fnref4" class="footnote-backref">↩︎</a></p></li><li id="fn5" class="footnote-item"><p><a href="https://luminousmen.com/post/the-apache-spark-optimization-checklist" target="_blank" rel="noreferrer">The Apache Spark Optimization Checklist</a> <a href="#fnref5" class="footnote-backref">↩︎</a></p></li></ol></section>',23)])])}const u=a(n,[["render",d]]);export{b as __pageData,u as default};
@@ -0,0 +1 @@
1
+ import{_ as a,o as t,c as o,a5 as s}from"./chunks/framework.DSg0KOwT.js";const r="../assets/broadcast-vs-shuffle.Db4WY1XK.svg",i="../assets/broadcast-vs-shuffle.dark.C7Bxs0mG.svg",b=JSON.parse('{"title":"Broadcast Sizing","description":"","frontmatter":{"title":"Broadcast Sizing"},"headers":[],"relativePath":"tuning-reference/bottleneck-broadcast-sizing.md","filePath":"tuning-reference/bottleneck-broadcast-sizing.md"}'),n={name:"tuning-reference/bottleneck-broadcast-sizing.md"};function d(l,e,c,h,f,p){return t(),o("div",null,[...e[0]||(e[0]=[s("",23)])])}const u=a(n,[["render",d]]);export{b as __pageData,u as default};
@@ -0,0 +1,7 @@
1
+ import{_ as t,o as a,c as s,a5 as o}from"./chunks/framework.DSg0KOwT.js";const r="../assets/cold-start-timeline.DxC_Sc7w.svg",i="../assets/cold-start-timeline.dark.CZ17YcAG.svg",m=JSON.parse('{"title":"Cold Start","description":"","frontmatter":{"title":"Cold Start"},"headers":[],"relativePath":"tuning-reference/bottleneck-cold-start.md","filePath":"tuning-reference/bottleneck-cold-start.md"}'),n={name:"tuning-reference/bottleneck-cold-start.md"};function l(c,e,h,d,f,p){return a(),s("div",null,[...e[0]||(e[0]=[o('<h1 id="bottleneck-cold-start" tabindex="-1">Cold Start <a class="header-anchor" href="#bottleneck-cold-start" aria-label="Permalink to &quot;Cold Start {#bottleneck-cold-start}&quot;">​</a></h1><p><span class="tag">COLD</span></p><h2 id="what-it-is" tabindex="-1">What it is <a class="header-anchor" href="#what-it-is" aria-label="Permalink to &quot;What it is&quot;">​</a></h2><p>Cold start is the gap between when a Spark application starts and when it actually begins doing work: the delay before the <a href="./spark-architecture.html">DAGScheduler</a> hands off the first stage for execution. In the event log, that moment is marked by <code>SparkListenerStageSubmitted</code>, whose payload is a direct serialization of a <code>StageInfo</code>: the event name plus the <code>stageInfo</code> and any properties, with no task-level data attached<sup class="footnote-ref"><a href="#fn1" id="fnref1">[1]</a></sup>. That shape matches what the event records. The DAGScheduler doesn&#39;t pre-schedule the whole DAG upfront (it reacts to stage completions, unlocking child stages one at a time), and once it has carved the DAG into stages and handed a <code>TaskSet</code> for the next stage to the TaskScheduler, only then does the TaskScheduler assign individual tasks to executors<sup class="footnote-ref"><a href="#fn2" id="fnref2">[2]</a></sup>. <code>SparkListenerStageSubmitted</code> therefore captures the moment the DAGScheduler creates/submits that <code>TaskSet</code>, before any task has actually been placed on an executor.</p><h2 id="how-it-s-detected" tabindex="-1">How it&#39;s detected <a class="header-anchor" href="#how-it-s-detected" aria-label="Permalink to &quot;How it&#39;s detected&quot;">​</a></h2><p><code>firstStageSubmittedAt − app.startTime</code> &gt; 30 s → Warning. <code>firstStageSubmittedAt</code> is a synthetic field: the timestamp of the first <code>SparkListenerStageSubmitted</code> event in the event log.</p><img class="light-only" src="'+r+'" alt="A timeline from app.startTime through the gap where executors are acquired and no tasks run to firstStageSubmittedAt, with that gap marked as the cold-start delay."><img class="dark-only" src="'+i+`" alt="A timeline from app.startTime through the gap where executors are acquired and no tasks run to firstStageSubmittedAt, with that gap marked as the cold-start delay."><h2 id="why-it-matters" tabindex="-1">Why it matters <a class="header-anchor" href="#why-it-matters" aria-label="Permalink to &quot;Why it matters&quot;">​</a></h2><p>Because the event carries only stage-level metadata and fires ahead of any executor task placement<sup class="footnote-ref"><a href="#fn1" id="fnref1:1">[1:1]</a></sup><sup class="footnote-ref"><a href="#fn2" id="fnref2:1">[2:1]</a></sup>, a large gap here reflects time spent before the DAGScheduler could submit work at all, not time spent inside tasks. It&#39;s a signal about scheduling/startup latency, distinct from the per-task metrics that describe work once tasks are actually running on executors.</p><h2 id="how-to-fix-it" tabindex="-1">How to fix it <a class="header-anchor" href="#how-to-fix-it" aria-label="Permalink to &quot;How to fix it&quot;">​</a></h2><p>Dynamic-allocation ramp-up isn&#39;t directly tied to <code>firstStageSubmittedAt</code> timing, but the configuration below governs how many executors an application has available at launch, and how reliably that number can grow:</p><ul><li>Size executors deliberately instead of defaulting to one large executor per node. For a 10-node, 16-core/64GB cluster targeting ~5 cores per executor, the standard derivation reserves 1 core per node for daemons (15 usable cores/node, 150 total), divides by 5 cores/executor (30), and subtracts 1 for the YARN ApplicationMaster: 29 executors × 5 cores × ~18GB<sup class="footnote-ref"><a href="#fn3" id="fnref3">[3]</a></sup>. Cloudera&#39;s equivalent 6-node walkthrough lands on the same shape: 17 executors × 5 cores × 19G, rather than one fat executor per node<sup class="footnote-ref"><a href="#fn4" id="fnref4">[4]</a></sup>.</li><li>If dynamic allocation is enabled, its ability to add executors at all depends on one of a few supporting mechanisms being turned on. One such mechanism is <code>spark.dynamicAllocation.shuffleTracking.enabled</code> (default <code>true</code> since Spark 3.0), which &quot;enables shuffle file tracking for executors, which allows dynamic allocation without the need for an external shuffle service,&quot; and &quot;will try to keep alive executors that are storing shuffle data for active jobs&quot;<sup class="footnote-ref"><a href="#fn5" id="fnref5">[5]</a></sup>. Without shuffle tracking or an external shuffle service configured, dynamic allocation cannot be enabled at all<sup class="footnote-ref"><a href="#fn5" id="fnref5:1">[5:1]</a></sup>.</li></ul><p>Deliberate executor sizing plus the dynamic-allocation precondition (values from the 16-core/64GB-node derivation above):</p><div class="language-properties vp-adaptive-theme"><button title="Copy Code" class="copy"></button><span class="lang">properties</span><pre class="shiki shiki-themes github-light github-dark vp-code" tabindex="0"><code><span class="line"><span style="--shiki-light:#6A737D;--shiki-dark:#6A737D;"># ~5 cores per executor instead of one fat executor per node</span></span>
2
+ <span class="line"><span style="--shiki-light:#D73A49;--shiki-dark:#F97583;">spark.executor.cores</span><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;">=5</span></span>
3
+ <span class="line"><span style="--shiki-light:#D73A49;--shiki-dark:#F97583;">spark.executor.memory</span><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;">=18g</span></span>
4
+ <span class="line"></span>
5
+ <span class="line"><span style="--shiki-light:#6A737D;--shiki-dark:#6A737D;"># Dynamic allocation, with its required shuffle-tracking precondition (default true since Spark 3.0)</span></span>
6
+ <span class="line"><span style="--shiki-light:#D73A49;--shiki-dark:#F97583;">spark.dynamicAllocation.enabled</span><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;">=true</span></span>
7
+ <span class="line"><span style="--shiki-light:#D73A49;--shiki-dark:#F97583;">spark.dynamicAllocation.shuffleTracking.enabled</span><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;">=true</span></span></code></pre></div><h2 id="limitations-false-positive-risk" tabindex="-1">Limitations / false-positive risk <a class="header-anchor" href="#limitations-false-positive-risk" aria-label="Permalink to &quot;Limitations / false-positive risk&quot;">​</a></h2><p>Some startup latency is unavoidable: a cluster still has to provision before it can run anything. The signal measures the gap before the first stage is submitted, and that gap can legitimately vary by cluster manager, so a warning here doesn&#39;t always mean something is wrong.</p><h2 id="related" tabindex="-1">Related <a class="header-anchor" href="#related" aria-label="Permalink to &quot;Related&quot;">​</a></h2><ul><li><strong>Executor sizing &amp; dynamic allocation:</strong> <a href="./cluster-config.html">Cluster Tuning</a></li></ul><hr class="footnotes-sep"><section class="footnotes"><ol class="footnotes-list"><li id="fn1" class="footnote-item"><p><a href="https://raw.githubusercontent.com/apache/spark/v3.5.0/core/src/main/scala/org/apache/spark/util/JsonProtocol.scala" target="_blank" rel="noreferrer">JsonProtocol.scala</a> <a href="#fnref1" class="footnote-backref">↩︎</a> <a href="#fnref1:1" class="footnote-backref">↩︎</a></p></li><li id="fn2" class="footnote-item"><p><a href="https://luminousmen.com/post/spark-anatomy-of-spark-application" target="_blank" rel="noreferrer">Anatomy of a Spark Application</a> <a href="#fnref2" class="footnote-backref">↩︎</a> <a href="#fnref2:1" class="footnote-backref">↩︎</a></p></li><li id="fn3" class="footnote-item"><p><a href="https://raw.githubusercontent.com/spoddutur/spark-notes/master/distribution_of_executors_cores_and_memory_for_spark_application.md" target="_blank" rel="noreferrer">Distribution of Executors, Cores and Memory for a Spark Application</a> <a href="#fnref3" class="footnote-backref">↩︎</a></p></li><li id="fn4" class="footnote-item"><p><a href="https://blog.cloudera.com/how-to-tune-your-apache-spark-jobs-part-2/" target="_blank" rel="noreferrer">How to Tune Your Apache Spark Jobs (Part 2)</a> <a href="#fnref4" class="footnote-backref">↩︎</a></p></li><li id="fn5" class="footnote-item"><p><a href="https://spark.apache.org/docs/latest/configuration.html" target="_blank" rel="noreferrer">Configuration (Spark)</a> <a href="#fnref5" class="footnote-backref">↩︎</a> <a href="#fnref5:1" class="footnote-backref">↩︎</a></p></li></ol></section>`,21)])])}const k=t(n,[["render",l]]);export{m as __pageData,k as default};
@@ -0,0 +1 @@
1
+ import{_ as t,o as a,c as s,a5 as o}from"./chunks/framework.DSg0KOwT.js";const r="../assets/cold-start-timeline.DxC_Sc7w.svg",i="../assets/cold-start-timeline.dark.CZ17YcAG.svg",m=JSON.parse('{"title":"Cold Start","description":"","frontmatter":{"title":"Cold Start"},"headers":[],"relativePath":"tuning-reference/bottleneck-cold-start.md","filePath":"tuning-reference/bottleneck-cold-start.md"}'),n={name:"tuning-reference/bottleneck-cold-start.md"};function l(c,e,h,d,f,p){return a(),s("div",null,[...e[0]||(e[0]=[o("",21)])])}const k=t(n,[["render",l]]);export{m as __pageData,k as default};
@@ -0,0 +1 @@
1
+ import{_ as t,a}from"./chunks/duplicate-plan-subtree.dark.Cdp70QhV.js";import{_ as r,o as s,c as o,a5 as n}from"./chunks/framework.DSg0KOwT.js";const b=JSON.parse('{"title":"Duplicate Plan Subtree","description":"","frontmatter":{"title":"Duplicate Plan Subtree"},"headers":[],"relativePath":"tuning-reference/bottleneck-duplicate-plan-subtree.md","filePath":"tuning-reference/bottleneck-duplicate-plan-subtree.md"}'),i={name:"tuning-reference/bottleneck-duplicate-plan-subtree.md"};function l(h,e,c,f,u,d){return s(),o("div",null,[...e[0]||(e[0]=[n('<h1 id="bottleneck-duplicate-plan-subtree" tabindex="-1">Duplicate Plan Subtree <a class="header-anchor" href="#bottleneck-duplicate-plan-subtree" aria-label="Permalink to &quot;Duplicate Plan Subtree {#bottleneck-duplicate-plan-subtree}&quot;">​</a></h1><p><span class="tag">DUPPLAN</span></p><h2 id="what-it-is" tabindex="-1">What it is <a class="header-anchor" href="#what-it-is" aria-label="Permalink to &quot;What it is&quot;">​</a></h2><p>The physical plan is the form Spark actually runs: after optimizing the logical plan, the engine produces a plan that &quot;specifies how the logical plan will execute on the cluster,&quot; compiling the query into a series of RDDs and transformations<sup class="footnote-ref"><a href="#fn1" id="fnref1">[1]</a></sup>. A duplicated subtree is the same source read, together with the operators stacked above it, appearing in more than one branch of that plan. When the same logic sits in two different branches of a pipeline, you are &quot;basically signing up to recompute everything from scratch multiple times,&quot; because every action on an uncached DataFrame starts back at the source and rebuilds all the intermediate steps<sup class="footnote-ref"><a href="#fn2" id="fnref2">[2]</a></sup>.</p><h2 id="how-it-s-detected" tabindex="-1">How it&#39;s detected <a class="header-anchor" href="#how-it-s-detected" aria-label="Permalink to &quot;How it&#39;s detected&quot;">​</a></h2><p>Exchanges are the giveaway. A <a href="./shuffle.html">shuffle</a> or broadcast is an exchange, and these are the points where Spark&#39;s pipelining stops: &quot;Spark operators are often pipelined and executed in parallel processes. However, a shuffle or broadcast exchange breaks this pipeline. We call them materialization points and use the term &#39;query stages&#39; to denote subsections bounded by these materialization points&quot;<sup class="footnote-ref"><a href="#fn3" id="fnref3">[3]</a></sup>. The signal is the same source-plus-operators subtree feeding more than one branch across those materialization boundaries, which is the shape that gets recomputed unless something intervenes.</p><img class="light-only" src="'+t+'" alt="Before and after: the same scan and operators feed two branches and run twice, then a single shared node from exchange reuse or a cache feeds both consumers."><img class="dark-only" src="'+a+'" alt="Before and after: the same scan and operators feed two branches and run twice, then a single shared node from exchange reuse or a cache feeds both consumers."><table tabindex="0"><thead><tr><th>Signal</th><th>What it points to</th></tr></thead><tbody><tr><td>Identical scan + operators in two or more branches</td><td>Same subtree recomputed per branch</td></tr><tr><td>A materialization point (shuffle/broadcast) below the repeat</td><td>The pipeline breaks and the branch restarts from source</td></tr></tbody></table><h2 id="why-it-matters" tabindex="-1">Why it matters <a class="header-anchor" href="#why-it-matters" aria-label="Permalink to &quot;Why it matters&quot;">​</a></h2><p>Left alone, a repeated subtree is repeated work. Each branch re-reads the source and re-runs everything above it, so a subtree that shows up twice is roughly paid for twice<sup class="footnote-ref"><a href="#fn2" id="fnref2:1">[2:1]</a></sup>. That cost lands on exactly the expensive parts of a plan (scans and shuffles) rather than on cheap row-level operators.</p><h2 id="how-to-fix-it" tabindex="-1">How to fix it <a class="header-anchor" href="#how-to-fix-it" aria-label="Permalink to &quot;How to fix it&quot;">​</a></h2><p>Two mechanisms let a plan reuse an identical subtree instead of rebuilding it.</p><ul><li>Exchange reuse leans on the fact that a shuffle already writes its output to disk. Spark &quot;always executes shuffles by first having the &#39;source&#39; tasks ... write shuffle files to their local disks,&quot; so &quot;running a new job over data that&#39;s already been shuffled does not rerun the &#39;source&#39; side of the shuffle. Because the shuffle files were already written to disk earlier, Spark knows that it can use them to run the later stages of the job, and it need not redo the earlier ones. In the Spark UI and logs, you will see the pre-shuffle stages marked as &#39;skipped&#39;&quot;<sup class="footnote-ref"><a href="#fn1" id="fnref1:1">[1:1]</a></sup>. When the same materialized exchange feeds two consumers, the second reads those existing shuffle files instead of recomputing the subtree beneath it.</li><li><code>cache()</code> / <code>persist()</code> truncate recomputation by materializing the result and keying future lookups on it. Persisting &quot;means materializing an RDD (usually by storing it in memory on the executors) for reuse during the current job&quot;<sup class="footnote-ref"><a href="#fn2" id="fnref2:2">[2:2]</a></sup>, and in the Structured API the lookup is plan-keyed: &quot;caching is done based on the physical plan. This means that we effectively store the physical plan as our key ... and perform a lookup prior to the execution of a Structured job&quot;<sup class="footnote-ref"><a href="#fn1" id="fnref1:2">[1:2]</a></sup>. On a hit, Spark returns the stored data rather than rebuilding intermediate steps from the source<sup class="footnote-ref"><a href="#fn2" id="fnref2:3">[2:3]</a></sup>.</li></ul><blockquote><p><strong>PySpark:</strong> if the repeated subtree is a DataFrame you reuse across branches, <code>df.cache()</code> (or <code>df.persist()</code>) before the branches split lets both read the materialized result instead of recomputing it.</p></blockquote><h2 id="confidence" tabindex="-1">Confidence <a class="header-anchor" href="#confidence" aria-label="Permalink to &quot;Confidence&quot;">​</a></h2><p>Medium. Inferring a duplicated subtree from the static plan is a medium-confidence signal that requires runtime validation: whether a repeated subtree actually turns into avoided work depends on runtime facts the plan never shows, such as cache hits, retained shuffle files, and eviction<sup class="footnote-ref"><a href="#fn2" id="fnref2:4">[2:4]</a></sup>.</p><h2 id="limitations-false-positive-risk" tabindex="-1">Limitations / false-positive risk <a class="header-anchor" href="#limitations-false-positive-risk" aria-label="Permalink to &quot;Limitations / false-positive risk&quot;">​</a></h2><p>Two subtrees that look identical in the printed plan are not always the same recomputed work. The cache &quot;is tied to the analyzed logical plan, not the final optimized one,&quot; so semantically identical queries with different analyzed plans miss the cache and recompute even though &quot;the optimizer will eventually produce the same physical plan&quot;<sup class="footnote-ref"><a href="#fn2" id="fnref2:5">[2:5]</a></sup>. <a href="./caching.html">Caching</a> is also lazy and often partial: it happens &quot;only as they are accessed,&quot; &quot;the only blocks that get cached are the ones Spark was forced to touch,&quot; and Spark gives &quot;no built-in visibility into how much of your DataFrame is actually cached, or how much has already been evicted&quot;<sup class="footnote-ref"><a href="#fn2" id="fnref2:6">[2:6]</a></sup>. Plan-level inference can therefore over-count, flagging repetition that either never materializes or is only partly reused.</p><hr class="footnotes-sep"><section class="footnotes"><ol class="footnotes-list"><li id="fn1" class="footnote-item"><p><a href="https://www.oreilly.com/library/view/spark-the-definitive/9781491912201/" target="_blank" rel="noreferrer">Spark: The Definitive Guide</a>, Chambers &amp; Zaharia, chs. 4, 15, 19 <a href="#fnref1" class="footnote-backref">↩︎</a> <a href="#fnref1:1" class="footnote-backref">↩︎</a> <a href="#fnref1:2" class="footnote-backref">↩︎</a></p></li><li id="fn2" class="footnote-item"><p><a href="https://luminousmen.com/post/explaining-the-mechanics-of-spark-caching" target="_blank" rel="noreferrer">Explaining the mechanics of Spark caching</a> <a href="#fnref2" class="footnote-backref">↩︎</a> <a href="#fnref2:1" class="footnote-backref">↩︎</a> <a href="#fnref2:2" class="footnote-backref">↩︎</a> <a href="#fnref2:3" class="footnote-backref">↩︎</a> <a href="#fnref2:4" class="footnote-backref">↩︎</a> <a href="#fnref2:5" class="footnote-backref">↩︎</a> <a href="#fnref2:6" class="footnote-backref">↩︎</a></p></li><li id="fn3" class="footnote-item"><p><a href="https://www.databricks.com/blog/2020/05/29/adaptive-query-execution-speeding-up-spark-sql-at-runtime.html" target="_blank" rel="noreferrer">Adaptive Query Execution: Speeding Up Spark SQL at Runtime</a> <a href="#fnref3" class="footnote-backref">↩︎</a></p></li></ol></section>',21)])])}const g=r(i,[["render",l]]);export{b as __pageData,g as default};
@@ -0,0 +1 @@
1
+ import{_ as t,a}from"./chunks/duplicate-plan-subtree.dark.Cdp70QhV.js";import{_ as r,o as s,c as o,a5 as n}from"./chunks/framework.DSg0KOwT.js";const b=JSON.parse('{"title":"Duplicate Plan Subtree","description":"","frontmatter":{"title":"Duplicate Plan Subtree"},"headers":[],"relativePath":"tuning-reference/bottleneck-duplicate-plan-subtree.md","filePath":"tuning-reference/bottleneck-duplicate-plan-subtree.md"}'),i={name:"tuning-reference/bottleneck-duplicate-plan-subtree.md"};function l(h,e,c,f,u,d){return s(),o("div",null,[...e[0]||(e[0]=[n("",21)])])}const g=r(i,[["render",l]]);export{b as __pageData,g as default};
@@ -0,0 +1,6 @@
1
+ import{_ as a,a as t}from"./chunks/retry-escalation-ladder.dark.DHipdJgZ.js";import{_ as o,o as s,c as r,a5 as i}from"./chunks/framework.DSg0KOwT.js";const k=JSON.parse('{"title":"Task Failures","description":"","frontmatter":{"title":"Task Failures"},"headers":[],"relativePath":"tuning-reference/bottleneck-failures.md","filePath":"tuning-reference/bottleneck-failures.md"}'),n={name:"tuning-reference/bottleneck-failures.md"};function l(c,e,d,f,h,u){return s(),r("div",null,[...e[0]||(e[0]=[i(`<h1 id="bottleneck-failures" tabindex="-1">Task Failures <a class="header-anchor" href="#bottleneck-failures" aria-label="Permalink to &quot;Task Failures {#bottleneck-failures}&quot;">​</a></h1><p><span class="tag">FAIL</span></p><h2 id="what-it-is" tabindex="-1">What it is <a class="header-anchor" href="#what-it-is" aria-label="Permalink to &quot;What it is&quot;">​</a></h2><p>A task failure is any task end whose <code>Reason</code> field is something other than success. Spark&#39;s event log serializes this as the formatted class name of whichever <code>TaskEndReason</code> instance was assigned to that task attempt<sup class="footnote-ref"><a href="#fn1" id="fnref1">[1]</a></sup>, and the canonical set of reasons is <code>Success</code>, <code>FetchFailed</code>, <code>ExceptionFailure</code>, <code>TaskResultLost</code>, <code>TaskKilled</code>, <code>TaskCommitDenied</code>, <code>ExecutorLostFailure</code>, <code>Resubmitted</code>, and <code>UnknownReason</code><sup class="footnote-ref"><a href="#fn1" id="fnref1:1">[1:1]</a></sup>. Each carries its own extra detail: a <code>FetchFailed</code> records a reduce ID and message; an <code>ExceptionFailure</code> carries the exception&#39;s class name, description, and stack trace, plus accumulator updates (falling back to reading them out of the task&#39;s metrics object for logs written by Spark 1.x); a <code>TaskKilled</code> records a kill reason; a <code>TaskCommitDenied</code> records the job ID, partition ID, and attempt number; an <code>ExecutorLostFailure</code> records whether the loss was caused by the application, the executor ID, and a loss reason; <code>UnknownReason</code> carries no extra fields<sup class="footnote-ref"><a href="#fn1" id="fnref1:2">[1:2]</a></sup>.</p><h2 id="how-it-s-detected" tabindex="-1">How it&#39;s detected <a class="header-anchor" href="#how-it-s-detected" aria-label="Permalink to &quot;How it&#39;s detected&quot;">​</a></h2><table tabindex="0"><thead><tr><th>failed tasks (share)</th><th>Level</th></tr></thead><tbody><tr><td>&gt; 5%</td><td>Warning</td></tr><tr><td>&gt; 20%</td><td>Critical</td></tr></tbody></table><p>The task-metrics object attached to a <code>SparkListenerTaskEnd</code> event is optional in the schema, so it isn&#39;t guaranteed to be present for every failure. It tends to be populated for reasons where the attempt actually ran far enough to produce metrics (an <code>ExceptionFailure</code> or <code>TaskKilled</code>, for instance) but can be absent for a <code>Resubmitted</code> attempt that never completed on its executor at all<sup class="footnote-ref"><a href="#fn1" id="fnref1:3">[1:3]</a></sup>. Detection has to key off the <code>Reason</code> field itself rather than assume metrics will always be there to inspect.</p><h2 id="why-it-matters" tabindex="-1">Why it matters <a class="header-anchor" href="#why-it-matters" aria-label="Permalink to &quot;Why it matters&quot;">​</a></h2><p>Some reasons point at something worse than a one-off hiccup. An <code>ExecutorLostFailure</code> can be the knock-on effect of a container or pod killed for exceeding its memory grant: enabling off-heap memory without also raising <code>spark.executor.memoryOverhead</code> (or setting <code>spark.executor.pyspark.memory</code> explicitly) lets real usage exceed what YARN or Kubernetes allocated, and the executor is killed with a generic container-memory error rather than a Spark-level exception<sup class="footnote-ref"><a href="#fn2" id="fnref2">[2]</a></sup>. Losing an executor mid-shuffle can also cascade into failures elsewhere: if dynamic allocation removes an executor before a shuffle it wrote data for has completed, that shuffle output is gone, and downstream tasks reading it fail and force an unnecessary recompute<sup class="footnote-ref"><a href="#fn3" id="fnref3">[3]</a></sup>.</p><h2 id="how-to-fix-it" tabindex="-1">How to fix it <a class="header-anchor" href="#how-to-fix-it" aria-label="Permalink to &quot;How to fix it&quot;">​</a></h2><ul><li>Read the <code>Reason</code> field before treating all failures the same: an <code>ExceptionFailure</code> points at application code, and an <code>ExecutorLostFailure</code> usually points at memory sizing or infrastructure.</li><li>For memory-driven <code>ExecutorLostFailure</code>s, size <code>spark.executor.memoryOverhead</code> (which defaults to <code>max(384 MB, 10% of executor memory)</code>) to actually cover off-heap and PySpark memory, since neither is folded into that default budget automatically<sup class="footnote-ref"><a href="#fn2" id="fnref2:1">[2:1]</a></sup>.</li><li>For failures caused by dynamic allocation reclaiming an executor mid-shuffle, enable the external <a href="./shuffle.html">shuffle service</a> or <code>spark.dynamicAllocation.shuffleTracking.enabled</code> so shuffle output can outlive the executor that wrote it, instead of being lost and recomputed<sup class="footnote-ref"><a href="#fn3" id="fnref3:1">[3:1]</a></sup>.</li></ul><p>Config-only levers for memory-driven <code>ExecutorLostFailure</code>s and mid-shuffle executor loss:</p><div class="language-properties vp-adaptive-theme"><button title="Copy Code" class="copy"></button><span class="lang">properties</span><pre class="shiki shiki-themes github-light github-dark vp-code" tabindex="0"><code><span class="line"><span style="--shiki-light:#6A737D;--shiki-dark:#6A737D;"># Cover off-heap + PySpark memory the default overhead budget does NOT include.</span></span>
2
+ <span class="line"><span style="--shiki-light:#6A737D;--shiki-dark:#6A737D;"># Default is max(384m, 10% of executor memory); 2g is an example, size to real off-heap/PySpark use.</span></span>
3
+ <span class="line"><span style="--shiki-light:#D73A49;--shiki-dark:#F97583;">spark.executor.memoryOverhead</span><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;">=2g</span></span>
4
+ <span class="line"></span>
5
+ <span class="line"><span style="--shiki-light:#6A737D;--shiki-dark:#6A737D;"># Keep shuffle output alive when dynamic allocation reclaims an executor mid-shuffle</span></span>
6
+ <span class="line"><span style="--shiki-light:#D73A49;--shiki-dark:#F97583;">spark.dynamicAllocation.shuffleTracking.enabled</span><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;">=true</span></span></code></pre></div><h2 id="failed-stage" tabindex="-1">Failed stage <a class="header-anchor" href="#failed-stage" aria-label="Permalink to &quot;Failed stage&quot;">​</a></h2><p><span class="tag">SFAIL</span></p><p>A single task failing and a whole stage failing are handled by different parts of the scheduler. Task retries sit in the TaskScheduler; the stage-level verdict belongs to the <a href="./spark-architecture.html">DAGScheduler</a>, which decides whether the job as a whole lives or dies. When a stage fails, it is not abandoned on the spot: the DAGScheduler resubmits it as a fresh attempt, and the job only fails once the stage cannot make progress after repeated attempts<sup class="footnote-ref"><a href="#fn4" id="fnref4">[4]</a></sup>. So the difference between &quot;retried and succeeds&quot; and &quot;aborts the job&quot; is whether a later attempt finishes before the attempt budget runs out.</p><img class="light-only" src="`+a+'" alt="The layered budgets a repeated failure climbs through, from task retries to stage resubmission to the executor and application ceilings that abort the job."><img class="dark-only" src="'+t+'" alt="The layered budgets a repeated failure climbs through, from task retries to stage resubmission to the executor and application ceilings that abort the job."><p>That budget is <code>spark.stage.maxConsecutiveAttempts</code>, default 4: the number of consecutive stage attempts allowed before the stage is aborted<sup class="footnote-ref"><a href="#fn5" id="fnref5">[5]</a></sup>. Fail, resubmit, and complete within those attempts and the job stays alive; keep failing until the limit is spent and the job aborts.</p><p>A <code>FetchFailed</code> is the classic trigger. It counts against that stage attempt limit, which is why <code>spark.stage.ignoreDecommissionFetchFailure</code> (default <code>true</code> since 3.4.0) exists: it keeps a fetch failure caused by an executor being decommissioned from counting toward <code>spark.stage.maxConsecutiveAttempts</code>, which by implication means ordinary fetch failures otherwise do count toward the abort budget<sup class="footnote-ref"><a href="#fn5" id="fnref5:1">[5:1]</a></sup>.</p><p>Executor exclusion is the other stage-level lever. The <code>spark.excludeOnFailure.*</code> family is off by default (<code>spark.excludeOnFailure.enabled</code> is <code>false</code>) and tracks how many tasks must fail on one executor within a stage before that executor is excluded for the stage; it can also kill executors excluded on a fetch failure<sup class="footnote-ref"><a href="#fn5" id="fnref5:2">[5:2]</a></sup>. If exclusion leaves a TaskSet with no schedulable executor left, <code>spark.scheduler.excludeOnFailure.unschedulableTaskSetTimeout</code> (default 120s) caps how long Spark waits to acquire a new executor before aborting that TaskSet<sup class="footnote-ref"><a href="#fn5" id="fnref5:3">[5:3]</a></sup>.</p><h2 id="confidence" tabindex="-1">Confidence <a class="header-anchor" href="#confidence" aria-label="Permalink to &quot;Confidence&quot;">​</a></h2><p>The failed-task-share thresholds and the stage-failure behaviour both map to documented Spark scheduler mechanics; no branch here is experimental.</p><h2 id="limitations-false-positive-risk" tabindex="-1">Limitations / false-positive risk <a class="header-anchor" href="#limitations-false-positive-risk" aria-label="Permalink to &quot;Limitations / false-positive risk&quot;">​</a></h2><p>A high failed-task share is a blunt signal. It can be dominated by one systemic cause, such as a single lost executor whose tasks all fail at once, rather than many independent failures pointing at a code or data problem. Transient failures that a later attempt retries and succeeds are also not always harmful: task retries reset their count on success, so a job can carry a nonzero failure share and still finish correctly. Read the <code>Reason</code> distribution before treating the share as a defect.</p><h2 id="related" tabindex="-1">Related <a class="header-anchor" href="#related" aria-label="Permalink to &quot;Related&quot;">​</a></h2><ul><li><strong>Executor sizing:</strong> <a href="./cluster-config.html">Cluster Tuning</a></li><li><strong>Memory overhead &amp; off-heap:</strong> <a href="./memory-model.html">Memory Management</a></li></ul><hr class="footnotes-sep"><section class="footnotes"><ol class="footnotes-list"><li id="fn1" class="footnote-item"><p><a href="https://raw.githubusercontent.com/apache/spark/v3.5.0/core/src/main/scala/org/apache/spark/util/JsonProtocol.scala" target="_blank" rel="noreferrer">JsonProtocol.scala</a> <a href="#fnref1" class="footnote-backref">↩︎</a> <a href="#fnref1:1" class="footnote-backref">↩︎</a> <a href="#fnref1:2" class="footnote-backref">↩︎</a> <a href="#fnref1:3" class="footnote-backref">↩︎</a></p></li><li id="fn2" class="footnote-item"><p><a href="https://luminousmen.com/post/dive-into-spark-memory" target="_blank" rel="noreferrer">Dive Into Spark Memory</a> <a href="#fnref2" class="footnote-backref">↩︎</a> <a href="#fnref2:1" class="footnote-backref">↩︎</a></p></li><li id="fn3" class="footnote-item"><p><a href="https://spark.apache.org/docs/latest/job-scheduling.html" target="_blank" rel="noreferrer">Job Scheduling (Apache Spark Documentation)</a> <a href="#fnref3" class="footnote-backref">↩︎</a> <a href="#fnref3:1" class="footnote-backref">↩︎</a></p></li><li id="fn4" class="footnote-item"><p><a href="https://luminousmen.com/post/spark-anatomy-of-spark-application" target="_blank" rel="noreferrer">Anatomy of a Spark Application</a> <a href="#fnref4" class="footnote-backref">↩︎</a></p></li><li id="fn5" class="footnote-item"><p><a href="https://spark.apache.org/docs/latest/configuration.html" target="_blank" rel="noreferrer">Spark Configuration (Apache Spark Documentation)</a> <a href="#fnref5" class="footnote-backref">↩︎</a> <a href="#fnref5:1" class="footnote-backref">↩︎</a> <a href="#fnref5:2" class="footnote-backref">↩︎</a> <a href="#fnref5:3" class="footnote-backref">↩︎</a></p></li></ol></section>',29)])])}const b=o(n,[["render",l]]);export{k as __pageData,b as default};
@@ -0,0 +1 @@
1
+ import{_ as a,a as t}from"./chunks/retry-escalation-ladder.dark.DHipdJgZ.js";import{_ as o,o as s,c as r,a5 as i}from"./chunks/framework.DSg0KOwT.js";const k=JSON.parse('{"title":"Task Failures","description":"","frontmatter":{"title":"Task Failures"},"headers":[],"relativePath":"tuning-reference/bottleneck-failures.md","filePath":"tuning-reference/bottleneck-failures.md"}'),n={name:"tuning-reference/bottleneck-failures.md"};function l(c,e,d,f,h,u){return s(),r("div",null,[...e[0]||(e[0]=[i("",29)])])}const b=o(n,[["render",l]]);export{k as __pageData,b as default};
@@ -0,0 +1,6 @@
1
+ import{_ as a,o as t,c as o,a5 as s}from"./chunks/framework.DSg0KOwT.js";const p=JSON.parse('{"title":"GC Pressure","description":"","frontmatter":{"title":"GC Pressure"},"headers":[],"relativePath":"tuning-reference/bottleneck-gc.md","filePath":"tuning-reference/bottleneck-gc.md"}'),r={name:"tuning-reference/bottleneck-gc.md"};function i(n,e,l,c,h,f){return t(),o("div",null,[...e[0]||(e[0]=[s(`<h1 id="bottleneck-gc" tabindex="-1">GC Pressure <a class="header-anchor" href="#bottleneck-gc" aria-label="Permalink to &quot;GC Pressure {#bottleneck-gc}&quot;">​</a></h1><p><span class="tag">GC</span></p><h2 id="what-it-is" tabindex="-1">What it is <a class="header-anchor" href="#what-it-is" aria-label="Permalink to &quot;What it is&quot;">​</a></h2><p>GC pressure describes how much of an executor&#39;s time goes to JVM garbage collection instead of running task code. Spark tracks this directly in its per-task metrics: <code>jvmGCTime</code> sits alongside <code>executorRunTime</code>, the elapsed wall-clock time the executor spent running the task (as opposed to <code>executorCpuTime</code>, which measures CPU time specifically).<sup class="footnote-ref"><a href="#fn1" id="fnref1">[1]</a></sup> <code>jvmGCTime</code> is not additive on top of that duration: it&#39;s defined as the elapsed time the JVM spent in garbage collection <em>while executing the task</em>, so GC pauses fall inside the same wall-clock window <code>executorRunTime</code> already measures, not outside it. Summing the two would double-count the GC pauses and overstate how long the task actually ran.<sup class="footnote-ref"><a href="#fn1" id="fnref1:1">[1:1]</a></sup></p><h2 id="how-it-s-detected" tabindex="-1">How it&#39;s detected <a class="header-anchor" href="#how-it-s-detected" aria-label="Permalink to &quot;How it&#39;s detected&quot;">​</a></h2><p>Because <code>jvmGCTime</code> sits inside <code>executorRunTime</code> rather than alongside it, the ratio between the two gives a bounded read on how much of a task&#39;s wall-clock time went to garbage collection:</p><table tabindex="0"><thead><tr><th>gcPct = jvmGCTime / executorRunTime</th><th>Level</th></tr></thead><tbody><tr><td>&gt; 10%</td><td>Warning</td></tr><tr><td>&gt; 20%</td><td>Critical</td></tr></tbody></table><h2 id="why-it-matters" tabindex="-1">Why it matters <a class="header-anchor" href="#why-it-matters" aria-label="Permalink to &quot;Why it matters&quot;">​</a></h2><p>As gcPct climbs past these thresholds, a growing share of every task&#39;s wall-clock time is consumed by garbage collection instead of actual computation, so the executor&#39;s useful throughput drops even while it appears busy. Two causes show up repeatedly: oversized executors, and an on-heap memory split pushed too far toward execution/storage.</p><p>Sizing an executor with too many cores relative to its heap is one documented trigger: a fat-executor configuration using all 16 cores of a node was observed to not only hurt HDFS throughput but also &quot;result in excessive garbage [collection].&quot;<sup class="footnote-ref"><a href="#fn2" id="fnref2">[2]</a></sup></p><p>Pushing <code>spark.memory.fraction</code> too far toward its upper end has a similar effect through a different path. Its complement, <code>1 − spark.memory.fraction</code>, is untracked &quot;User Memory&quot; reserved for UDFs, Python/Arrow glue, and native library buffers; Spark doesn&#39;t manage this region at all, and starving it (which is what raising <code>spark.memory.fraction</code> toward something like 0.9 does) leads to &quot;GC pressure or random OOMs,&quot; with no warning before the failure.<sup class="footnote-ref"><a href="#fn3" id="fnref3">[3]</a></sup></p><h2 id="how-to-fix-it" tabindex="-1">How to fix it <a class="header-anchor" href="#how-to-fix-it" aria-label="Permalink to &quot;How to fix it&quot;">​</a></h2><ul><li>Keep cores per executor down rather than packing an entire node&#39;s cores onto one JVM (the common ~5-cores-per-executor guideline). That avoids the excessive-garbage-collection failure mode documented for fat-executor configurations.<sup class="footnote-ref"><a href="#fn2" id="fnref2:1">[2:1]</a></sup></li><li>Don&#39;t raise <code>spark.memory.fraction</code> past a comfortable range chasing more execution/storage memory: doing so starves the untracked user-memory region and risks the same GC-pressure/OOM failure mode.<sup class="footnote-ref"><a href="#fn3" id="fnref3:1">[3:1]</a></sup></li><li>Move Tungsten&#39;s execution and storage buffers off the JVM heap with <code>spark.memory.offHeap.enabled=true</code> (and a positive <code>spark.memory.offHeap.size</code>, which is required whenever off-heap is enabled<sup class="footnote-ref"><a href="#fn4" id="fnref4">[4]</a></sup>). Off-heap buffers are invisible to the garbage collector, so fewer and smaller live objects need to be tracked, scanned, and copied on the heap, reducing both the frequency and duration of GC pauses.<sup class="footnote-ref"><a href="#fn3" id="fnref3:2">[3:2]</a></sup></li><li>Tune or switch the GC collector on large heaps. G1GC is Spark&#39;s default collector since 4.0 (which defaults to JDK 17), and large executor heaps may need <code>-XX:G1HeapRegionSize</code> raised as well.<sup class="footnote-ref"><a href="#fn5" id="fnref5">[5]</a></sup> G1&#39;s stop-the-world evacuation and full-GC pauses can still spike under heavy old-generation pressure on large heaps: one documented 88GB-heap benchmark saw a job&#39;s pause spike to nearly 100 seconds under default G1 settings, and only tuning <code>InitiatingHeapOccupancyPercent</code>, <code>ConcGCThreads</code>, and RSet-update parameters brought G1 back ahead of Parallel/CMS on both throughput and latency.<sup class="footnote-ref"><a href="#fn6" id="fnref6">[6]</a></sup> ZGC is a documented alternative built for pause-time control: it performs its expensive work concurrently, without stopping application threads for more than a millisecond, with pause times designed to stay independent of heap size from a few hundred megabytes up to 16TB.<sup class="footnote-ref"><a href="#fn7" id="fnref7">[7]</a></sup></li></ul><p>Move buffers off-heap and give G1 room on large heaps (sizes are examples; set to your workload):</p><div class="language-properties vp-adaptive-theme"><button title="Copy Code" class="copy"></button><span class="lang">properties</span><pre class="shiki shiki-themes github-light github-dark vp-code" tabindex="0"><code><span class="line"><span style="--shiki-light:#6A737D;--shiki-dark:#6A737D;"># Move Tungsten execution/storage buffers off the JVM heap so the GC has fewer objects to scan</span></span>
2
+ <span class="line"><span style="--shiki-light:#D73A49;--shiki-dark:#F97583;">spark.memory.offHeap.enabled</span><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;">=true</span></span>
3
+ <span class="line"><span style="--shiki-light:#D73A49;--shiki-dark:#F97583;">spark.memory.offHeap.size</span><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;">=2g </span><span style="--shiki-light:#6A737D;--shiki-dark:#6A737D;"># must be &gt; 0 whenever off-heap is enabled; 2g is an example</span></span>
4
+ <span class="line"></span>
5
+ <span class="line"><span style="--shiki-light:#6A737D;--shiki-dark:#6A737D;"># G1 is the default collector since Spark 4.0 (JDK 17); raise the region size on large heaps</span></span>
6
+ <span class="line"><span style="--shiki-light:#D73A49;--shiki-dark:#F97583;">spark.executor.extraJavaOptions</span><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;">=-XX:+UseG1GC -XX:</span><span style="--shiki-light:#D73A49;--shiki-dark:#F97583;">G1HeapRegionSize</span><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;">=16m </span><span style="--shiki-light:#6A737D;--shiki-dark:#6A737D;"># 16m is an example</span></span></code></pre></div><h2 id="confidence" tabindex="-1">Confidence <a class="header-anchor" href="#confidence" aria-label="Permalink to &quot;Confidence&quot;">​</a></h2><p>The warning and critical thresholds on gcPct are validated: they read a bounded ratio of <code>jvmGCTime</code> to <code>executorRunTime</code>, both first-class per-task metrics Spark records directly,<sup class="footnote-ref"><a href="#fn1" id="fnref1:2">[1:2]</a></sup> and crossing 10% or 20% marks a real, measurable share of wall-clock time lost to garbage collection.</p><p><span class="tag">EXPERIMENTAL</span> A separate signal treats suspiciously <em>low</em> GC time as a possible cost-model or noise-floor indicator, and that one is an unsourced heuristic with no external basis, so treat it as exploratory rather than a confirmed diagnostic.</p><h2 id="limitations-false-positive-risk" tabindex="-1">Limitations / false-positive risk <a class="header-anchor" href="#limitations-false-positive-risk" aria-label="Permalink to &quot;Limitations / false-positive risk&quot;">​</a></h2><p>A high gcPct can reflect a transient memory spike (a single skewed task, a brief burst of large allocations) rather than chronic pressure, so a threshold crossing on one stage does not by itself prove a sustained sizing problem; corroborate against the executor&#39;s behavior over the whole run. The low-GC signal has no external basis at all, and a near-zero reading is at least as likely to mean the workload simply is not allocation-heavy as to mean anything is wrong.</p><h2 id="related" tabindex="-1">Related <a class="header-anchor" href="#related" aria-label="Permalink to &quot;Related&quot;">​</a></h2><ul><li><strong>Why it happens:</strong> <a href="./memory-model.html">Memory Management</a></li><li><strong>Executor sizing:</strong> <a href="./cluster-config.html">Cluster Tuning</a></li></ul><hr class="footnotes-sep"><section class="footnotes"><ol class="footnotes-list"><li id="fn1" class="footnote-item"><p><a href="https://spark.apache.org/docs/latest/monitoring.html#spark-history-server" target="_blank" rel="noreferrer">Monitoring and Instrumentation</a> <a href="#fnref1" class="footnote-backref">↩︎</a> <a href="#fnref1:1" class="footnote-backref">↩︎</a> <a href="#fnref1:2" class="footnote-backref">↩︎</a></p></li><li id="fn2" class="footnote-item"><p><a href="https://raw.githubusercontent.com/spoddutur/spark-notes/master/distribution_of_executors_cores_and_memory_for_spark_application.md" target="_blank" rel="noreferrer">Distribution of Executors, Cores, and Memory for a Spark Application</a> <a href="#fnref2" class="footnote-backref">↩︎</a> <a href="#fnref2:1" class="footnote-backref">↩︎</a></p></li><li id="fn3" class="footnote-item"><p><a href="https://luminousmen.com/post/dive-into-spark-memory" target="_blank" rel="noreferrer">Diving into Spark Memory Management</a> <a href="#fnref3" class="footnote-backref">↩︎</a> <a href="#fnref3:1" class="footnote-backref">↩︎</a> <a href="#fnref3:2" class="footnote-backref">↩︎</a></p></li><li id="fn4" class="footnote-item"><p><a href="https://spark.apache.org/docs/latest/configuration.html" target="_blank" rel="noreferrer">Configuration (Spark)</a> <a href="#fnref4" class="footnote-backref">↩︎</a></p></li><li id="fn5" class="footnote-item"><p><a href="https://spark.apache.org/docs/latest/tuning.html" target="_blank" rel="noreferrer">Spark Tuning Guide</a> <a href="#fnref5" class="footnote-backref">↩︎</a></p></li><li id="fn6" class="footnote-item"><p><a href="https://www.databricks.com/blog/2015/05/28/tuning-java-garbage-collection-for-spark-applications.html" target="_blank" rel="noreferrer">Tuning Java Garbage Collection for Apache Spark Applications</a> <a href="#fnref6" class="footnote-backref">↩︎</a></p></li><li id="fn7" class="footnote-item"><p><a href="https://wiki.openjdk.org/display/zgc" target="_blank" rel="noreferrer">The Z Garbage Collector (ZGC)</a> <a href="#fnref7" class="footnote-backref">↩︎</a></p></li></ol></section>`,24)])])}const u=a(r,[["render",i]]);export{p as __pageData,u as default};
@@ -0,0 +1 @@
1
+ import{_ as a,o as t,c as o,a5 as s}from"./chunks/framework.DSg0KOwT.js";const p=JSON.parse('{"title":"GC Pressure","description":"","frontmatter":{"title":"GC Pressure"},"headers":[],"relativePath":"tuning-reference/bottleneck-gc.md","filePath":"tuning-reference/bottleneck-gc.md"}'),r={name:"tuning-reference/bottleneck-gc.md"};function i(n,e,l,c,h,f){return t(),o("div",null,[...e[0]||(e[0]=[s("",24)])])}const u=a(r,[["render",i]]);export{p as __pageData,u as default};
@@ -0,0 +1,8 @@
1
+ import{_ as t,a}from"./chunks/retry-escalation-ladder.dark.DHipdJgZ.js";import{_ as s,o,c as r,a5 as i}from"./chunks/framework.DSg0KOwT.js";const b=JSON.parse('{"title":"Job Failure Rate","description":"","frontmatter":{"title":"Job Failure Rate"},"headers":[],"relativePath":"tuning-reference/bottleneck-job-failure-rate.md","filePath":"tuning-reference/bottleneck-job-failure-rate.md"}'),n={name:"tuning-reference/bottleneck-job-failure-rate.md"};function l(c,e,u,f,d,h){return o(),r("div",null,[...e[0]||(e[0]=[i('<h1 id="bottleneck-job-failure-rate" tabindex="-1">Job Failure Rate <a class="header-anchor" href="#bottleneck-job-failure-rate" aria-label="Permalink to &quot;Job Failure Rate {#bottleneck-job-failure-rate}&quot;">​</a></h1><p><span class="tag">FAIL-RATE</span></p><h2 id="what-it-is" tabindex="-1">What it is <a class="header-anchor" href="#what-it-is" aria-label="Permalink to &quot;What it is&quot;">​</a></h2><p>Not every task retry threatens the job. <code>spark.task.maxFailures</code> (default <code>4</code>) counts &quot;continuous failures of any particular task before giving up on the job. The total number of failures spread across different tasks will not cause the job to fail; a particular task has to fail this number of attempts continuously. If any attempt succeeds, the failure count for the task will be reset&quot;<sup class="footnote-ref"><a href="#fn1" id="fnref1">[1]</a></sup>. A task that fails once and then succeeds on retry resets its counter and never counts toward job abandonment: that&#39;s a transient retry, and by itself it&#39;s harmless. Whole-job abandonment is a different, layered escalation above that: it only happens once an unbroken run of failures on the same task reaches the configured limit (3 retries allowed by default, since allowed retries = value − 1), or once higher-level retry budgets above the task level are themselves exhausted.</p><h2 id="how-it-s-detected" tabindex="-1">How it&#39;s detected <a class="header-anchor" href="#how-it-s-detected" aria-label="Permalink to &quot;How it&#39;s detected&quot;">​</a></h2><p>The layers above task-level retries are what turn isolated failures into a failed job. At the stage level, <code>spark.stage.maxConsecutiveAttempts</code> (default <code>4</code>) bounds &quot;the number of consecutive stage attempts allowed before a stage is aborted&quot;<sup class="footnote-ref"><a href="#fn1" id="fnref1:1">[1:1]</a></sup>. By default, <code>spark.stage.ignoreDecommissionFetchFailure</code> (<code>true</code>, since 3.4.0) excludes fetch failures caused by graceful executor decommission from counting toward that limit, so a decommission-triggered <code>FetchFailed</code> doesn&#39;t push a stage toward abortion the way a genuine repeated fetch failure would<sup class="footnote-ref"><a href="#fn1" id="fnref1:2">[1:2]</a></sup>. On YARN and Kubernetes there&#39;s a further application-level ceiling: <code>spark.executor.maxNumFailures</code> (default <code>numExecutors * 2</code>, minimum 3, since 3.5.0) is &quot;the maximum number of executor failures before failing the application,&quot; while <code>spark.executor.failuresValidityInterval</code> (since 3.5.0) lets failures spaced far enough apart be &quot;considered independent and not accumulate towards the attempt count&quot;<sup class="footnote-ref"><a href="#fn1" id="fnref1:3">[1:3]</a></sup>. The TaskScheduler owns retries at the task level, but it is &quot;the DAGScheduler that ultimately declares the job to have failed&quot;<sup class="footnote-ref"><a href="#fn2" id="fnref2">[2]</a></sup> once those retry budgets are exhausted, and because &quot;a Spark job corresponds to one action&quot; with a fixed DAG once that action is called<sup class="footnote-ref"><a href="#fn3" id="fnref3">[3]</a></sup>, an aborted stage cascades directly into failure of the job built on it.</p><img class="light-only" src="'+t+'" alt="How a failure escalates from a task retry up through stage resubmission, the executor failure ceiling, and a failed application attempt before the job is aborted."><img class="dark-only" src="'+a+`" alt="How a failure escalates from a task retry up through stage resubmission, the executor failure ceiling, and a failed application attempt before the job is aborted."><h2 id="why-it-matters" tabindex="-1">Why it matters <a class="header-anchor" href="#why-it-matters" aria-label="Permalink to &quot;Why it matters&quot;">​</a></h2><p>A rising failure rate that traces back to exhausted retry budgets, rather than to a single one-off task failure, points at a systemic problem (an unstable executor, a genuinely broken fetch path, or a resource ceiling being hit repeatedly) since none of these budgets trip on a single transient hiccup. Job-level failure is also distinct from application-attempt failure: on YARN, &quot;each application may have multiple attempts,&quot; and the history server displays failed attempts alongside &quot;any ongoing incomplete attempt or the final successful attempt&quot;<sup class="footnote-ref"><a href="#fn4" id="fnref4">[4]</a></sup>: a separate, higher layer of retry (driver/application restart) above the in-job task and stage retries covered here.</p><h2 id="how-to-fix-it" tabindex="-1">How to fix it <a class="header-anchor" href="#how-to-fix-it" aria-label="Permalink to &quot;How to fix it&quot;">​</a></h2><ul><li>Check which budget was exhausted before treating a job failure as a single root cause: a single task failing <code>spark.task.maxFailures</code> times consecutively, a stage being resubmitted until <code>spark.stage.maxConsecutiveAttempts</code> is reached and aborted, or (on YARN/Kubernetes) cumulative distinct executor failures exceeding <code>spark.executor.maxNumFailures</code><sup class="footnote-ref"><a href="#fn1" id="fnref1:4">[1:4]</a></sup> each point at a different underlying problem.</li><li>If failures are decommission-triggered <code>FetchFailed</code>s during a normal scale-down, confirm <code>spark.stage.ignoreDecommissionFetchFailure</code> is enabled (default <code>true</code> since 3.4.0) so they aren&#39;t inflating the stage-abort count<sup class="footnote-ref"><a href="#fn1" id="fnref1:5">[1:5]</a></sup>.</li><li>On YARN/Kubernetes, if unrelated executor failures spread far apart in time are tripping <code>spark.executor.maxNumFailures</code>, <code>spark.executor.failuresValidityInterval</code> (since 3.5.0) can let sufficiently-spaced failures stop accumulating toward the same count<sup class="footnote-ref"><a href="#fn1" id="fnref1:6">[1:6]</a></sup>.</li><li>See also <a href="./bottleneck-retry-waste.html">retry waste</a>: the same executor-loss and <code>FetchFailed</code> causes that eventually exhaust these budgets and fail a job are, below that threshold, also the source of wasted executor time on jobs that ultimately succeed.</li></ul><blockquote><p><strong>PySpark:</strong> these are session-level configs, settable without a <code>spark-submit</code> flag: <code>spark.conf.set(&quot;spark.task.maxFailures&quot;, &quot;4&quot;)</code>, and the stage/executor-level equivalents the same way.</p></blockquote><p>The retry-budget ladder: defaults shown; raise a budget only once you know which layer is tripping:</p><div class="language-properties vp-adaptive-theme"><button title="Copy Code" class="copy"></button><span class="lang">properties</span><pre class="shiki shiki-themes github-light github-dark vp-code" tabindex="0"><code><span class="line"><span style="--shiki-light:#6A737D;--shiki-dark:#6A737D;"># Task level: consecutive failures of one task before the job is abandoned (default 4 =&gt; 3 retries)</span></span>
2
+ <span class="line"><span style="--shiki-light:#D73A49;--shiki-dark:#F97583;">spark.task.maxFailures</span><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;">=4</span></span>
3
+ <span class="line"></span>
4
+ <span class="line"><span style="--shiki-light:#6A737D;--shiki-dark:#6A737D;"># Stage level: consecutive stage attempts before the stage is aborted (default 4)</span></span>
5
+ <span class="line"><span style="--shiki-light:#D73A49;--shiki-dark:#F97583;">spark.stage.maxConsecutiveAttempts</span><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;">=4</span></span>
6
+ <span class="line"></span>
7
+ <span class="line"><span style="--shiki-light:#6A737D;--shiki-dark:#6A737D;"># Don&#39;t count graceful-decommission fetch failures toward the stage-abort limit (default true since Spark 3.4)</span></span>
8
+ <span class="line"><span style="--shiki-light:#D73A49;--shiki-dark:#F97583;">spark.stage.ignoreDecommissionFetchFailure</span><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;">=true</span></span></code></pre></div><h2 id="limitations-false-positive-risk" tabindex="-1">Limitations / false-positive risk <a class="header-anchor" href="#limitations-false-positive-risk" aria-label="Permalink to &quot;Limitations / false-positive risk&quot;">​</a></h2><p>A nonzero job-failure rate can be dominated by a single repeatedly-failing job rather than a systemic issue, so the rate alone doesn&#39;t tell you whether the problem is broad or isolated. Retried jobs that eventually succeed still count toward attempts, which can inflate the rate above what the eventual outcomes justify.</p><h2 id="related" tabindex="-1">Related <a class="header-anchor" href="#related" aria-label="Permalink to &quot;Related&quot;">​</a></h2><ul><li><strong>Retry budgets &amp; executor stability:</strong> <a href="./cluster-config.html">Cluster Tuning</a></li><li><strong>Avoiding the failures upstream:</strong> <a href="./anti-patterns.html">Anti-Patterns</a></li></ul><hr class="footnotes-sep"><section class="footnotes"><ol class="footnotes-list"><li id="fn1" class="footnote-item"><p><a href="https://spark.apache.org/docs/latest/configuration.html" target="_blank" rel="noreferrer">Configuration (Spark)</a> <a href="#fnref1" class="footnote-backref">↩︎</a> <a href="#fnref1:1" class="footnote-backref">↩︎</a> <a href="#fnref1:2" class="footnote-backref">↩︎</a> <a href="#fnref1:3" class="footnote-backref">↩︎</a> <a href="#fnref1:4" class="footnote-backref">↩︎</a> <a href="#fnref1:5" class="footnote-backref">↩︎</a> <a href="#fnref1:6" class="footnote-backref">↩︎</a></p></li><li id="fn2" class="footnote-item"><p><a href="https://luminousmen.com/post/spark-anatomy-of-spark-application" target="_blank" rel="noreferrer">Spark: Anatomy of Spark Application</a> <a href="#fnref2" class="footnote-backref">↩︎</a></p></li><li id="fn3" class="footnote-item"><p><em>High Performance Spark, 2nd Edition</em>, Karau, Polak &amp; Warren, ch. 2 <a href="#fnref3" class="footnote-backref">↩︎</a></p></li><li id="fn4" class="footnote-item"><p><a href="https://spark.apache.org/docs/latest/monitoring.html#spark-history-server" target="_blank" rel="noreferrer">Monitoring and Instrumentation</a> <a href="#fnref4" class="footnote-backref">↩︎</a></p></li></ol></section>`,21)])])}const k=s(n,[["render",l]]);export{b as __pageData,k as default};
@@ -0,0 +1 @@
1
+ import{_ as t,a}from"./chunks/retry-escalation-ladder.dark.DHipdJgZ.js";import{_ as s,o,c as r,a5 as i}from"./chunks/framework.DSg0KOwT.js";const b=JSON.parse('{"title":"Job Failure Rate","description":"","frontmatter":{"title":"Job Failure Rate"},"headers":[],"relativePath":"tuning-reference/bottleneck-job-failure-rate.md","filePath":"tuning-reference/bottleneck-job-failure-rate.md"}'),n={name:"tuning-reference/bottleneck-job-failure-rate.md"};function l(c,e,u,f,d,h){return o(),r("div",null,[...e[0]||(e[0]=[i("",21)])])}const k=s(n,[["render",l]]);export{b as __pageData,k as default};
@@ -0,0 +1,7 @@
1
+ import{_ as a,o as t,c as o,a5 as s}from"./chunks/framework.DSg0KOwT.js";const p=JSON.parse('{"title":"Memory Utilization","description":"","frontmatter":{"title":"Memory Utilization"},"headers":[],"relativePath":"tuning-reference/bottleneck-memory-utilization.md","filePath":"tuning-reference/bottleneck-memory-utilization.md"}'),r={name:"tuning-reference/bottleneck-memory-utilization.md"};function n(i,e,l,c,d,f){return t(),o("div",null,[...e[0]||(e[0]=[s(`<h1 id="bottleneck-memory-utilization" tabindex="-1">Memory Utilization <a class="header-anchor" href="#bottleneck-memory-utilization" aria-label="Permalink to &quot;Memory Utilization {#bottleneck-memory-utilization}&quot;">​</a></h1><p><span class="tag">MEM</span></p><h2 id="what-it-is" tabindex="-1">What it is <a class="header-anchor" href="#what-it-is" aria-label="Permalink to &quot;What it is&quot;">​</a></h2><p>An executor is a single JVM process that gets a fixed memory allocation when the application starts, and it holds that whole allocation for its entire lifetime, whether or not it has work to run<sup class="footnote-ref"><a href="#fn1" id="fnref1">[1]</a></sup>. Two things can leave that memory band poorly used: cores sitting idle inside a held executor, and a held executor whose memory drains of active work but is never released. This finding covers both.</p><p>An executor&#39;s core count is its concurrency ceiling. <code>spark.executor.cores</code> sets how many tasks it can run at once, so <code>--executor-cores 5</code> caps that executor at five concurrent tasks<sup class="footnote-ref"><a href="#fn2" id="fnref2">[2]</a></sup>. The scheduler turns cores into task slots from <code>spark.executor.cores</code> and <code>spark.task.cpus</code> (minimum 1), with <code>spark.task.cpus</code> defaulting to one core per task<sup class="footnote-ref"><a href="#fn3" id="fnref3">[3]</a></sup>. Cores go idle whenever there are fewer runnable tasks than slots: a stage with fewer partitions than the total slots across executors leaves slots empty, and so does the tail of a stage where a <a href="./bottleneck-straggler.html">straggler</a> or two keep running after their peers finish. Raising <code>spark.task.cpus</code> above 1 has the same effect from the other side, since each task then reserves several cores and fewer run side by side. Through all of it the JVM keeps its fixed heap<sup class="footnote-ref"><a href="#fn1" id="fnref1:1">[1:1]</a></sup>.</p><h2 id="how-it-s-detected" tabindex="-1">How it&#39;s detected <a class="header-anchor" href="#how-it-s-detected" aria-label="Permalink to &quot;How it&#39;s detected&quot;">​</a></h2><table tabindex="0"><thead><tr><th>Rule</th><th>Signal</th><th>Status</th></tr></thead><tbody><tr><td>idleCores</td><td>Task parallelism below allocated cores while the executor is held</td><td>Validated</td></tr><tr><td>memoryBand</td><td>A whole executor&#39;s memory band held with little or no active work (dynamic-allocation idle timeouts)</td><td>Validated</td></tr><tr><td>wasteModel</td><td>Peak used memory well below allocated memory (unused headroom)</td><td>Experimental</td></tr></tbody></table><h2 id="why-it-matters" tabindex="-1">Why it matters <a class="header-anchor" href="#why-it-matters" aria-label="Permalink to &quot;Why it matters&quot;">​</a></h2><p>A held executor is allocation you pay for regardless of how busy it is. On YARN that is the memory YARN grants the container; on Kubernetes it is the pod memory limit<sup class="footnote-ref"><a href="#fn1" id="fnref1:2">[1:2]</a></sup>. When cores idle or a whole executor lingers with nothing to do, that fixed band is reserved without returning work, so the cost lands whether or not tasks are running.</p><h2 id="how-to-fix-it" tabindex="-1">How to fix it <a class="header-anchor" href="#how-to-fix-it" aria-label="Permalink to &quot;How to fix it&quot;">​</a></h2><p>Keep executors from being oversized in the first place. The guidance is against packing every core of a node into one fat executor: assigning all 16 cores of a node to a single executor hurts HDFS throughput and drives excessive <a href="./bottleneck-gc.html">garbage collection</a>, so aim for a balance between tiny (one core per executor) and fat (one executor per node) sizing<sup class="footnote-ref"><a href="#fn4" id="fnref4">[4]</a></sup>. On YARN, leave cores for the OS and Hadoop daemons instead of handing 100% of a node to Spark containers<sup class="footnote-ref"><a href="#fn2" id="fnref2:1">[2:1]</a></sup>.</p><p>For the held memory band, lean on <a href="./cluster-config.html">dynamic allocation</a> to reclaim executors once the work drains. It requests executors when tasks back up and frees them when they go idle<sup class="footnote-ref"><a href="#fn2" id="fnref2:2">[2:2]</a></sup>. Executors are added in rounds once tasks have been pending for <code>spark.dynamicAllocation.schedulerBacklogTimeout</code> (default 1s), then again every <code>spark.dynamicAllocation.sustainedSchedulerBacklogTimeout</code> while the backlog holds<sup class="footnote-ref"><a href="#fn5" id="fnref5">[5]</a></sup>. On the release side, an executor is removed after it has been idle longer than <code>spark.dynamicAllocation.executorIdleTimeout</code><sup class="footnote-ref"><a href="#fn5" id="fnref5:1">[5:1]</a></sup>.</p><p>Cached data is the trap here. By default an executor holding <a href="./caching.html">cached blocks</a> is never removed, governed by <code>spark.dynamicAllocation.cachedExecutorIdleTimeout</code>, whose default is infinity<sup class="footnote-ref"><a href="#fn3" id="fnref3:1">[3:1]</a></sup>. Such an executor keeps its full memory band indefinitely with no active tasks unless you set that timeout to a finite value, or turn on <code>spark.shuffle.service.fetch.rdd.enabled</code> so executors holding only disk-persisted blocks are treated as idle after <code>spark.dynamicAllocation.executorIdleTimeout</code> and released<sup class="footnote-ref"><a href="#fn5" id="fnref5:2">[5:2]</a></sup>. To avoid holding barely-used executors at all, lower <code>spark.dynamicAllocation.executorAllocationRatio</code> (default 1.0, full parallelism) toward 0.5, since with small tasks full-parallelism allocation can request executors that never do any work<sup class="footnote-ref"><a href="#fn3" id="fnref3:2">[3:2]</a></sup>.</p><div class="language-properties vp-adaptive-theme"><button title="Copy Code" class="copy"></button><span class="lang">properties</span><pre class="shiki shiki-themes github-light github-dark vp-code" tabindex="0"><code><span class="line"><span style="--shiki-light:#6A737D;--shiki-dark:#6A737D;"># Balanced executor sizing, not one fat executor per node</span></span>
2
+ <span class="line"><span style="--shiki-light:#D73A49;--shiki-dark:#F97583;">spark.executor.cores</span><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;">=5</span></span>
3
+ <span class="line"></span>
4
+ <span class="line"><span style="--shiki-light:#6A737D;--shiki-dark:#6A737D;"># Reclaim idle executors; give cached holders a finite timeout</span></span>
5
+ <span class="line"><span style="--shiki-light:#D73A49;--shiki-dark:#F97583;">spark.dynamicAllocation.enabled</span><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;">=true</span></span>
6
+ <span class="line"><span style="--shiki-light:#D73A49;--shiki-dark:#F97583;">spark.dynamicAllocation.cachedExecutorIdleTimeout</span><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;">=300s</span></span>
7
+ <span class="line"><span style="--shiki-light:#D73A49;--shiki-dark:#F97583;">spark.dynamicAllocation.executorAllocationRatio</span><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;">=0.5</span></span></code></pre></div><h3 id="reading-the-wasted-memory-estimate-experimental" tabindex="-1">Reading the wasted-memory estimate <span class="tag">EXPERIMENTAL</span> <a class="header-anchor" href="#reading-the-wasted-memory-estimate-experimental" aria-label="Permalink to &quot;Reading the wasted-memory estimate &lt;span class=&quot;tag&quot;&gt;EXPERIMENTAL&lt;/span&gt;&quot;">​</a></h3><p>The wasteModel rule estimates unused (&quot;wasted&quot;) allocated memory from the gap between an executor&#39;s peak used memory and its allocated total. Treat it as a rough buffer heuristic, not a measurement. Peak used memory is a high-water mark rather than a ceiling: Spark records it under <code>peakMemoryMetrics.*</code>, and each figure is the maximum its pool ever reached, so peak sits at or below the allocated total by construction and the space between them is the executor&#39;s unused headroom<sup class="footnote-ref"><a href="#fn6" id="fnref6">[6]</a></sup>. Turning that headroom into a wasted number stays approximate for concrete reasons. The peak heap figure counts garbage, since <code>JVMHeapMemory</code> is the peak used heap including &quot;the amount of memory occupied by both live objects and garbage objects that have not been collected,&quot; so it overstates the live footprint<sup class="footnote-ref"><a href="#fn6" id="fnref6:1">[6:1]</a></sup>. The region boundaries also move: <code>totalOnHeapStorageMemory</code> and <code>totalOffHeapStorageMemory</code> &quot;can vary over time, depending on the MemoryManager implementation,&quot; so there is no single fixed allocated-to-storage value to subtract a peak from<sup class="footnote-ref"><a href="#fn6" id="fnref6:2">[6:2]</a></sup>. Peak metrics only reach the event log when <code>spark.eventLog.logStageExecutorMetrics</code> is true<sup class="footnote-ref"><a href="#fn6" id="fnref6:3">[6:3]</a></sup>, so without that setting there is nothing to estimate from. Use the number as a hint that an executor may be oversized, then confirm against sizing before acting on it.</p><h2 id="confidence" tabindex="-1">Confidence <a class="header-anchor" href="#confidence" aria-label="Permalink to &quot;Confidence&quot;">​</a></h2><p>The idleCores and memoryBand rules are validated: they rest on documented Spark concurrency and dynamic-allocation behavior. The wasteModel estimate is low-confidence and experimental. It is a buffer heuristic derived from peak-versus-allocated sampling, not an exact accounting of unused memory, so it should steer investigation rather than settle it.</p><h2 id="limitations-false-positive-risk" tabindex="-1">Limitations / false-positive risk <a class="header-anchor" href="#limitations-false-positive-risk" aria-label="Permalink to &quot;Limitations / false-positive risk&quot;">​</a></h2><p>Idle cores are not always waste. The tail of a stage legitimately leaves slots empty while a straggler or two finish<sup class="footnote-ref"><a href="#fn3" id="fnref3:3">[3:3]</a></sup>, so a snapshot of low parallelism can reflect a normal straggler tail rather than chronic under-utilization. The wasted-memory estimate depends on peak-memory sampling and is only as good as that sampling: the peak includes uncollected garbage, the managed-region boundaries shift under the MemoryManager, and the figures are absent entirely unless <code>spark.eventLog.logStageExecutorMetrics</code> is enabled<sup class="footnote-ref"><a href="#fn6" id="fnref6:4">[6:4]</a></sup>.</p><hr class="footnotes-sep"><section class="footnotes"><ol class="footnotes-list"><li id="fn1" class="footnote-item"><p><a href="https://luminousmen.com/post/dive-into-spark-memory" target="_blank" rel="noreferrer">Dive into Spark memory</a> <a href="#fnref1" class="footnote-backref">↩︎</a> <a href="#fnref1:1" class="footnote-backref">↩︎</a> <a href="#fnref1:2" class="footnote-backref">↩︎</a></p></li><li id="fn2" class="footnote-item"><p><a href="https://blog.cloudera.com/how-to-tune-your-apache-spark-jobs-part-2/" target="_blank" rel="noreferrer">How to Tune Your Apache Spark Jobs (Part 2)</a> <a href="#fnref2" class="footnote-backref">↩︎</a> <a href="#fnref2:1" class="footnote-backref">↩︎</a> <a href="#fnref2:2" class="footnote-backref">↩︎</a></p></li><li id="fn3" class="footnote-item"><p><a href="https://spark.apache.org/docs/latest/configuration.html" target="_blank" rel="noreferrer">Configuration — Spark</a> <a href="#fnref3" class="footnote-backref">↩︎</a> <a href="#fnref3:1" class="footnote-backref">↩︎</a> <a href="#fnref3:2" class="footnote-backref">↩︎</a> <a href="#fnref3:3" class="footnote-backref">↩︎</a></p></li><li id="fn4" class="footnote-item"><p><a href="https://raw.githubusercontent.com/spoddutur/spark-notes/master/distribution_of_executors_cores_and_memory_for_spark_application.md" target="_blank" rel="noreferrer">Distribution of executors, cores and memory for a Spark application</a> <a href="#fnref4" class="footnote-backref">↩︎</a></p></li><li id="fn5" class="footnote-item"><p><a href="https://spark.apache.org/docs/latest/job-scheduling.html" target="_blank" rel="noreferrer">Job Scheduling — Spark</a> <a href="#fnref5" class="footnote-backref">↩︎</a> <a href="#fnref5:1" class="footnote-backref">↩︎</a> <a href="#fnref5:2" class="footnote-backref">↩︎</a></p></li><li id="fn6" class="footnote-item"><p><a href="https://spark.apache.org/docs/latest/monitoring.html" target="_blank" rel="noreferrer">Monitoring — Spark</a> <a href="#fnref6" class="footnote-backref">↩︎</a> <a href="#fnref6:1" class="footnote-backref">↩︎</a> <a href="#fnref6:2" class="footnote-backref">↩︎</a> <a href="#fnref6:3" class="footnote-backref">↩︎</a> <a href="#fnref6:4" class="footnote-backref">↩︎</a></p></li></ol></section>`,22)])])}const u=a(r,[["render",n]]);export{p as __pageData,u as default};
@@ -0,0 +1 @@
1
+ import{_ as a,o as t,c as o,a5 as s}from"./chunks/framework.DSg0KOwT.js";const p=JSON.parse('{"title":"Memory Utilization","description":"","frontmatter":{"title":"Memory Utilization"},"headers":[],"relativePath":"tuning-reference/bottleneck-memory-utilization.md","filePath":"tuning-reference/bottleneck-memory-utilization.md"}'),r={name:"tuning-reference/bottleneck-memory-utilization.md"};function n(i,e,l,c,d,f){return t(),o("div",null,[...e[0]||(e[0]=[s("",22)])])}const u=a(r,[["render",n]]);export{p as __pageData,u as default};
@@ -0,0 +1 @@
1
+ import{_ as t,a}from"./chunks/retry-escalation-ladder.dark.DHipdJgZ.js";import{_ as r,o,c as s,a5 as i}from"./chunks/framework.DSg0KOwT.js";const k=JSON.parse('{"title":"Retry Waste","description":"","frontmatter":{"title":"Retry Waste"},"headers":[],"relativePath":"tuning-reference/bottleneck-retry-waste.md","filePath":"tuning-reference/bottleneck-retry-waste.md"}'),n={name:"tuning-reference/bottleneck-retry-waste.md"};function l(c,e,f,h,u,d){return o(),s("div",null,[...e[0]||(e[0]=[i('<h1 id="bottleneck-retry-waste" tabindex="-1">Retry Waste <a class="header-anchor" href="#bottleneck-retry-waste" aria-label="Permalink to &quot;Retry Waste {#bottleneck-retry-waste}&quot;">​</a></h1><p><span class="tag">RETRY</span></p><h2 id="what-it-is" tabindex="-1">What it is <a class="header-anchor" href="#what-it-is" aria-label="Permalink to &quot;What it is&quot;">​</a></h2><p>When a task attempt is superseded by a later retry, every bit of executor time the abandoned attempt spent counts for nothing. Every attempt, successful or not, accrues the standard task metrics: <code>executorRunTime</code> (&quot;elapsed time the executor spent running this task,&quot; including time fetching shuffle data) and <code>executorCpuTime</code><sup class="footnote-ref"><a href="#fn1" id="fnref1">[1]</a></sup>, but only the winning attempt&#39;s output survives. The TaskScheduler is what triggers the do-over: &quot;if an executor dies or a task throws an exception, the TaskScheduler resubmits the task to another executor, respecting Spark&#39;s task locality preferences&quot;<sup class="footnote-ref"><a href="#fn2" id="fnref2">[2]</a></sup>. Whatever the discarded attempt had already run is gone; none of it carries forward into the job&#39;s result.</p><img class="light-only" src="'+t+'" alt="Where a superseded retry sits in the escalation ladder, below the stage and application budgets that a job crosses only when the same failures keep recurring."><img class="dark-only" src="'+a+'" alt="Where a superseded retry sits in the escalation ladder, below the stage and application budgets that a job crosses only when the same failures keep recurring."><h2 id="how-it-s-detected" tabindex="-1">How it&#39;s detected <a class="header-anchor" href="#how-it-s-detected" aria-label="Permalink to &quot;How it&#39;s detected&quot;">​</a></h2><p>A stage&#39;s superseded task attempts are the signal: a later retry supersedes an earlier attempt after <a href="./bottleneck-failures.html">an executor is lost</a> (<code>ExecutorLostFailure</code>) or after a shuffle <code>FetchFailed</code>, since &quot;if an executor dies or a task throws an exception, the TaskScheduler resubmits the task to another executor, respecting Spark&#39;s task locality preferences&quot;<sup class="footnote-ref"><a href="#fn2" id="fnref2:1">[2:1]</a></sup>. Because only the winning attempt&#39;s output survives, the abandoned attempt&#39;s <code>executorRunTime</code> (the elapsed executor time defined above) is the wasted time.</p><p>The two causes are handled differently by Spark&#39;s own machinery, which is how they can be told apart after the fact. An <code>ExecutorLostFailure</code> removes the executor (and any shuffle map output it had already written) from the pool outright<sup class="footnote-ref"><a href="#fn3" id="fnref3">[3]</a></sup>. A shuffle <code>FetchFailed</code>, by contrast, is a read-side failure against a still-alive remote executor, and Spark&#39;s own exclusion machinery treats it as a distinct category from a general executor loss: <code>spark.excludeOnFailure.killExcludedExecutors</code> governs whether Spark kills executors &quot;excluded on fetch failure or excluded for the entire application&quot;<sup class="footnote-ref"><a href="#fn4" id="fnref4">[4]</a></sup>, a separate bucket from executor loss in Spark&#39;s accounting. That distinction (executor-loss bucket vs. fetch-failure bucket) is the basis for identifying which cause dominated a given stage&#39;s retry waste.</p><h2 id="why-it-matters" tabindex="-1">Why it matters <a class="header-anchor" href="#why-it-matters" aria-label="Permalink to &quot;Why it matters&quot;">​</a></h2><p>A lost executor can waste more than just the in-flight task&#39;s time: &quot;dynamic allocation may remove an executor before the shuffle completes, in which case the shuffle files written by that executor must be recomputed unnecessarily&quot;<sup class="footnote-ref"><a href="#fn3" id="fnref3:1">[3:1]</a></sup>, so prior map-output work from that same executor may need redoing too. A <code>FetchFailed</code> doesn&#39;t carry that same risk, since the remote executor stays alive and only the one failed read is lost<sup class="footnote-ref"><a href="#fn4" id="fnref4:1">[4:1]</a></sup>.</p><h2 id="how-to-fix-it" tabindex="-1">How to fix it <a class="header-anchor" href="#how-to-fix-it" aria-label="Permalink to &quot;How to fix it&quot;">​</a></h2><ul><li>Check the retried-attempt count and accumulated <code>executorRunTime</code>/<code>executorCpuTime</code> on the abandoned attempts directly, rather than trusting the stage&#39;s final wall-clock duration: a clean-looking stage can still be hiding significant wasted compute<sup class="footnote-ref"><a href="#fn1" id="fnref1:1">[1:1]</a></sup>.</li><li>If the dominant cause is executor loss, investigate node-level stability (OOM-kills, node death, aggressive dynamic-allocation deallocation) rather than the task logic itself, since the lost executor&#39;s prior shuffle-write work may also need recomputing<sup class="footnote-ref"><a href="#fn3" id="fnref3:2">[3:2]</a></sup>.</li><li>If the dominant cause is <code>FetchFailed</code>, check whether <code>spark.excludeOnFailure.killExcludedExecutors</code> is causing repeated exclusion churn on otherwise-live executors, and treat it separately from outright executor loss<sup class="footnote-ref"><a href="#fn4" id="fnref4:2">[4:2]</a></sup>.</li><li>See also <a href="./bottleneck-job-failure-rate.html">job failure rate</a>: retry waste can pile up quietly on a job that ultimately succeeds, while the same underlying causes (exhausted at a higher threshold) are what push a job to fail outright.</li></ul><blockquote><p><strong>PySpark:</strong> there&#39;s no attempt-level API to recover a discarded attempt&#39;s <code>executorRunTime</code>: pull it from the event log or history server UI&#39;s per-stage task list, filtering for tasks whose attempt number is greater than zero.</p></blockquote><p>The preventive lever: keep an executor&#39;s shuffle output available if it is lost, so retries don&#39;t recompute prior map work<sup class="footnote-ref"><a href="#fn5" id="fnref5">[5]</a></sup>:</p><div class="language-properties vp-adaptive-theme"><button title="Copy Code" class="copy"></button><span class="lang">properties</span><pre class="shiki shiki-themes github-light github-dark vp-code" tabindex="0"><code><span class="line"><span style="--shiki-light:#D73A49;--shiki-dark:#F97583;">spark.dynamicAllocation.shuffleTracking.enabled</span><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;">=true</span></span></code></pre></div><h2 id="limitations-false-positive-risk" tabindex="-1">Limitations / false-positive risk <a class="header-anchor" href="#limitations-false-positive-risk" aria-label="Permalink to &quot;Limitations / false-positive risk&quot;">​</a></h2><p>Some retry waste is unavoidable: transient cluster faults will always cost a few superseded attempts, and no tuning drives that to zero. The metric attributes burned executor time to the abandoned attempts, so its accuracy tracks the failure cause. When a lost executor also forces recompute of prior map output, the per-attempt <code>executorRunTime</code> undercounts the true cost; when an attempt fails almost immediately, it overcounts the compute actually lost. Read the number as a directional signal, not an exact ledger.</p><h2 id="related" tabindex="-1">Related <a class="header-anchor" href="#related" aria-label="Permalink to &quot;Related&quot;">​</a></h2><ul><li><strong>Shuffle recompute on executor loss:</strong> <a href="./shuffle.html">Shuffle</a></li><li><strong>Executor stability &amp; dynamic allocation:</strong> <a href="./cluster-config.html">Cluster Tuning</a></li></ul><hr class="footnotes-sep"><section class="footnotes"><ol class="footnotes-list"><li id="fn1" class="footnote-item"><p><a href="https://spark.apache.org/docs/latest/monitoring.html" target="_blank" rel="noreferrer">Monitoring and Instrumentation</a> <a href="#fnref1" class="footnote-backref">↩︎</a> <a href="#fnref1:1" class="footnote-backref">↩︎</a></p></li><li id="fn2" class="footnote-item"><p><a href="https://luminousmen.com/post/spark-anatomy-of-spark-application" target="_blank" rel="noreferrer">Spark: Anatomy of Spark Application</a> <a href="#fnref2" class="footnote-backref">↩︎</a> <a href="#fnref2:1" class="footnote-backref">↩︎</a></p></li><li id="fn3" class="footnote-item"><p><a href="https://spark.apache.org/docs/latest/job-scheduling.html" target="_blank" rel="noreferrer">Job Scheduling (Spark)</a> <a href="#fnref3" class="footnote-backref">↩︎</a> <a href="#fnref3:1" class="footnote-backref">↩︎</a> <a href="#fnref3:2" class="footnote-backref">↩︎</a></p></li><li id="fn4" class="footnote-item"><p><a href="https://spark.apache.org/docs/latest/configuration.html" target="_blank" rel="noreferrer">Configuration (Spark)</a> <a href="#fnref4" class="footnote-backref">↩︎</a> <a href="#fnref4:1" class="footnote-backref">↩︎</a> <a href="#fnref4:2" class="footnote-backref">↩︎</a></p></li><li id="fn5" class="footnote-item"><p><a href="https://spark.apache.org/docs/latest/configuration.html" target="_blank" rel="noreferrer">Configuration (Spark)</a> <a href="#fnref5" class="footnote-backref">↩︎</a></p></li></ol></section>',22)])])}const g=r(n,[["render",l]]);export{k as __pageData,g as default};
@@ -0,0 +1 @@
1
+ import{_ as t,a}from"./chunks/retry-escalation-ladder.dark.DHipdJgZ.js";import{_ as r,o,c as s,a5 as i}from"./chunks/framework.DSg0KOwT.js";const k=JSON.parse('{"title":"Retry Waste","description":"","frontmatter":{"title":"Retry Waste"},"headers":[],"relativePath":"tuning-reference/bottleneck-retry-waste.md","filePath":"tuning-reference/bottleneck-retry-waste.md"}'),n={name:"tuning-reference/bottleneck-retry-waste.md"};function l(c,e,f,h,u,d){return o(),s("div",null,[...e[0]||(e[0]=[i("",22)])])}const g=r(n,[["render",l]]);export{k as __pageData,g as default};
@@ -0,0 +1,12 @@
1
+ import{_ as a,o as t,c as s,a5 as o}from"./chunks/framework.DSg0KOwT.js";const p=JSON.parse('{"title":"Shuffle I/O","description":"","frontmatter":{"title":"Shuffle I/O"},"headers":[],"relativePath":"tuning-reference/bottleneck-shuffle.md","filePath":"tuning-reference/bottleneck-shuffle.md"}'),r={name:"tuning-reference/bottleneck-shuffle.md"};function i(n,e,f,l,c,d){return t(),s("div",null,[...e[0]||(e[0]=[o(`<h1 id="bottleneck-shuffle" tabindex="-1">Shuffle I/O <a class="header-anchor" href="#bottleneck-shuffle" aria-label="Permalink to &quot;Shuffle I/O {#bottleneck-shuffle}&quot;">​</a></h1><p><span class="tag">SHFL</span></p><h2 id="what-it-is" tabindex="-1">What it is <a class="header-anchor" href="#what-it-is" aria-label="Permalink to &quot;What it is&quot;">​</a></h2><p>Shuffle I/O is the data movement Spark performs between stages whenever an operation needs to redistribute data across the cluster so that related rows land on the same partition. At the DataFrame/SQL level, <code>groupBy()</code>, <code>join()</code>, <code>agg()</code>, <code>sortBy()</code>, and <code>reduceByKey()</code>-style aggregations are the classic wide transformations that force this exchange<sup class="footnote-ref"><a href="#fn1" id="fnref1">[1]</a></sup>. Non-broadcast join strategies (shuffle hash join, shuffle sort-merge join, and shuffle-and-replicated nested loop / Cartesian product join) all shuffle data across executors the same way<sup class="footnote-ref"><a href="#fn1" id="fnref1:1">[1:1]</a></sup>. At the RDD layer, the operations that can cause a shuffle are repartitioning (<code>repartition</code>, <code>coalesce</code>), <code>&#39;ByKey</code> operations other than counting (<code>groupByKey</code>, <code>reduceByKey</code>), and join-family operations (<code>cogroup</code>, <code>join</code>)<sup class="footnote-ref"><a href="#fn2" id="fnref2">[2]</a></sup>.</p><p>Spark can skip the shuffle when it already knows the data layout: Storage Partition Join avoids it entirely when Spark can use partitioning already reported by a compatible V2 data source<sup class="footnote-ref"><a href="#fn3" id="fnref3">[3]</a></sup>, and classic Hive-style bucketing has the same effect: once both sides of a join are bucketed and sorted the same way, the physical plan shows no <code>Exchange</code> operator<sup class="footnote-ref"><a href="#fn4" id="fnref4">[4]</a></sup>. <code>DataFrameWriter.partitionBy</code>, by contrast, does not trigger a shuffle by itself on write<sup class="footnote-ref"><a href="#fn5" id="fnref5">[5]</a></sup>.</p><p>At the metrics level, shuffle reads split cross-node traffic from same-host traffic: <code>remoteBytesRead</code> counts bytes read from a remote executor, <code>localBytesRead</code> counts bytes read from local disk, and <code>totalBytesRead</code> is their sum<sup class="footnote-ref"><a href="#fn6" id="fnref6">[6]</a></sup>.</p><h2 id="how-it-s-detected" tabindex="-1">How it&#39;s detected <a class="header-anchor" href="#how-it-s-detected" aria-label="Permalink to &quot;How it&#39;s detected&quot;">​</a></h2><table tabindex="0"><thead><tr><th>Shuffle read/write bytes</th><th>Level</th></tr></thead><tbody><tr><td>&gt; 50 MB</td><td>Info</td></tr><tr><td>&gt; 500 MB</td><td>Warning</td></tr><tr><td>&gt; 1 GB</td><td>Critical</td></tr></tbody></table><p>Beyond raw byte volume, the executor-side wait is captured by <code>fetchWaitTime</code>: time a task spends blocked on a remote shuffle block it needs next, not counting time spent prefetching other blocks in the background<sup class="footnote-ref"><a href="#fn6" id="fnref6:1">[6:1]</a></sup>. That wait time counts toward the task&#39;s overall <code>executorRunTime</code>, since Spark&#39;s wall-clock task-time metric explicitly includes time fetching shuffle data<sup class="footnote-ref"><a href="#fn6" id="fnref6:2">[6:2]</a></sup>.</p><h2 id="why-it-matters" tabindex="-1">Why it matters <a class="header-anchor" href="#why-it-matters" aria-label="Permalink to &quot;Why it matters&quot;">​</a></h2><p>As shuffle volume grows, the underlying pull-based mechanism gets less efficient: the number of shuffle blocks grows quadratically with mapper × reducer count while individual block sizes shrink to only tens of KB, which is inefficient for disk-backed random reads<sup class="footnote-ref"><a href="#fn7" id="fnref7">[7]</a></sup>. Because fetch-wait time is part of a task&#39;s measured execution window, that inefficiency shows up directly as added task time rather than as separate, hidden overhead<sup class="footnote-ref"><a href="#fn6" id="fnref6:3">[6:3]</a></sup>.</p><h2 id="how-to-fix-it" tabindex="-1">How to fix it <a class="header-anchor" href="#how-to-fix-it" aria-label="Permalink to &quot;How to fix it&quot;">​</a></h2><ul><li>Let <a href="./aqe.html">Adaptive Query Execution</a> coalesce small post-shuffle partitions automatically at runtime (<code>spark.sql.adaptive.coalescePartitions.enabled</code>, default <code>true</code>) instead of only hand-tuning a fixed partition count<sup class="footnote-ref"><a href="#fn3" id="fnref3:1">[3:1]</a></sup>.</li><li>Review <code>spark.sql.shuffle.partitions</code> (default <code>200</code>), which applies uniformly to every <code>join()</code>, <code>groupBy()</code>, and aggregation regardless of how much data is actually moving<sup class="footnote-ref"><a href="#fn8" id="fnref8">[8]</a></sup>. There&#39;s no fixed formula for the right value: pull it down toward the executor core count for small or streaming workloads<sup class="footnote-ref"><a href="#fn1" id="fnref1:2">[1:2]</a></sup>, and otherwise favor over- to under-provisioning, since Spark&#39;s low per-task overhead makes it safer to have too many tasks than too few<sup class="footnote-ref"><a href="#fn9" id="fnref9">[9]</a></sup>.</li><li>Where possible, avoid the shuffle altogether: bucket both sides of a join identically, or rely on Storage Partition Join for compatible sources, so the physical plan drops the <code>Exchange</code> node<sup class="footnote-ref"><a href="#fn3" id="fnref3:2">[3:2]</a></sup><sup class="footnote-ref"><a href="#fn4" id="fnref4:1">[4:1]</a></sup>.</li><li>If map tasks are I/O-bound writing many shuffle files, increase <code>spark.shuffle.file.buffer</code> (default <code>32k</code>) to reduce disk seeks and system calls on the write side<sup class="footnote-ref"><a href="#fn8" id="fnref8:1">[8:1]</a></sup>; Learning Spark&#39;s tuning table recommends bumping it to 1 MB for large jobs<sup class="footnote-ref"><a href="#fn1" id="fnref1:3">[1:3]</a></sup>.</li><li>If executors have memory to spare, increase <code>spark.reducer.maxSizeInFlight</code> (default <code>48m</code>) so reducers can pull more map output concurrently and need fewer fetch rounds<sup class="footnote-ref"><a href="#fn8" id="fnref8:2">[8:2]</a></sup>.</li><li>At larger scale, enable the external shuffle service (effectively required for <a href="./cluster-config.html">dynamic allocation</a>, since it lets shuffle files be served after an executor is removed<sup class="footnote-ref"><a href="#fn10" id="fnref10">[10]</a></sup>) and consider push-based (Magnet) shuffle, which converts many small random reads into large sequential reads of pre-merged chunks<sup class="footnote-ref"><a href="#fn7" id="fnref7:1">[7:1]</a></sup>.</li></ul><blockquote><p><strong>PySpark:</strong> adjust the shuffle partition count directly from a running session with <code>spark.conf.set(&quot;spark.sql.shuffle.partitions&quot;, 100)</code>.</p></blockquote><p>A shuffle-tuning starting point: defaults shown, comments say which way to move:</p><div class="language-properties vp-adaptive-theme"><button title="Copy Code" class="copy"></button><span class="lang">properties</span><pre class="shiki shiki-themes github-light github-dark vp-code" tabindex="0"><code><span class="line"><span style="--shiki-light:#6A737D;--shiki-dark:#6A737D;"># Let AQE coalesce small post-shuffle partitions at runtime</span></span>
2
+ <span class="line"><span style="--shiki-light:#D73A49;--shiki-dark:#F97583;">spark.sql.adaptive.enabled</span><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;">=true</span></span>
3
+ <span class="line"><span style="--shiki-light:#D73A49;--shiki-dark:#F97583;">spark.sql.adaptive.coalescePartitions.enabled</span><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;">=true</span></span>
4
+ <span class="line"></span>
5
+ <span class="line"><span style="--shiki-light:#6A737D;--shiki-dark:#6A737D;"># Baseline shuffle partition count (default 200); pull toward executor-core count for small jobs</span></span>
6
+ <span class="line"><span style="--shiki-light:#D73A49;--shiki-dark:#F97583;">spark.sql.shuffle.partitions</span><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;">=200</span></span>
7
+ <span class="line"></span>
8
+ <span class="line"><span style="--shiki-light:#6A737D;--shiki-dark:#6A737D;"># Reduce write-side disk seeks on large shuffles (default 32k; Learning Spark suggests 1m)</span></span>
9
+ <span class="line"><span style="--shiki-light:#D73A49;--shiki-dark:#F97583;">spark.shuffle.file.buffer</span><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;">=1m</span></span>
10
+ <span class="line"></span>
11
+ <span class="line"><span style="--shiki-light:#6A737D;--shiki-dark:#6A737D;"># Let reducers pull more map output per fetch round when executors have spare memory (default 48m)</span></span>
12
+ <span class="line"><span style="--shiki-light:#D73A49;--shiki-dark:#F97583;">spark.reducer.maxSizeInFlight</span><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;">=48m</span></span></code></pre></div><h2 id="partition-sizing-part" tabindex="-1">Partition sizing <span class="tag">PART</span> <a class="header-anchor" href="#partition-sizing-part" aria-label="Permalink to &quot;Partition sizing &lt;span class=&quot;tag&quot;&gt;PART&lt;/span&gt;&quot;">​</a></h2><p>Adaptive Query Execution re-optimizes the plan while the query runs: as each shuffle stage materializes, it reads the real shuffle-file sizes and resizes partitions before launching the stages downstream<sup class="footnote-ref"><a href="#fn11" id="fnref11">[11]</a></sup>. That runtime feedback is what corrects a partition grid that static estimates would get wrong.</p><p>The coarse starting grid is <code>spark.sql.shuffle.partitions</code> (default <code>200</code>), the fixed partition count Spark uses when shuffling for joins or aggregations<sup class="footnote-ref"><a href="#fn12" id="fnref12">[12]</a></sup>. A single number cannot fit every stage. Spread a few megabytes across 200 partitions and most cores sit idle on a handful of rows each; push hundreds of gigabytes through the same 200 and every executor is overloaded, often into memory errors. For small or streaming workloads the 200 default is usually too high, and pulling it toward the executor-core count thins out the flood of tiny partitions crossing the network<sup class="footnote-ref"><a href="#fn1" id="fnref1:4">[1:4]</a></sup>. The pattern AQE encourages is the opposite of hand-tuning that number: leave the initial count large and let runtime coalescing combine adjacent small partitions instead<sup class="footnote-ref"><a href="#fn11" id="fnref11:1">[11:1]</a></sup>.</p><p><strong>Partitions too small (low parallelism payoff).</strong> <code>spark.sql.adaptive.coalescePartitions.enabled</code> (default <code>true</code>) merges contiguous shuffle partitions up to a target size so a stage does not end up with many tiny tasks<sup class="footnote-ref"><a href="#fn13" id="fnref13">[13]</a></sup>. The target is <code>spark.sql.adaptive.advisoryPartitionSizeInBytes</code>, the advisory shuffle-partition size AQE steers toward during adaptive optimization, default <code>64 MB</code><sup class="footnote-ref"><a href="#fn13" id="fnref13:1">[13:1]</a></sup>.</p><p><strong>Partitions too big or skewed.</strong> <code>spark.sql.adaptive.skewJoin.enabled</code> (default <code>true</code>) handles skew in shuffled joins by splitting the oversized partitions and replicating the matching side where needed<sup class="footnote-ref"><a href="#fn14" id="fnref14">[14]</a></sup>. A partition only counts as skewed when it clears both bars: larger than <code>spark.sql.adaptive.skewJoin.skewedPartitionFactor</code> (default <code>5.0</code>) times the median partition size, and larger than <code>spark.sql.adaptive.skewJoin.skewedPartitionThresholdInBytes</code> (default <code>256 MB</code>)<sup class="footnote-ref"><a href="#fn14" id="fnref14:1">[14:1]</a></sup>. Once it qualifies, the partition is split back down toward the advisory size.</p><p>So the two knobs divide the labor: <code>spark.sql.shuffle.partitions</code> lays down the coarse initial grid, and <code>advisoryPartitionSizeInBytes</code> (64 MB) is the size AQE aims each partition at from both directions. Partitions below it get coalesced; skewed partitions past the 256 MB / 5.0-factor bar get split toward it.</p><h2 id="limitations-false-positive-risk" tabindex="-1">Limitations / false-positive risk <a class="header-anchor" href="#limitations-false-positive-risk" aria-label="Permalink to &quot;Limitations / false-positive risk&quot;">​</a></h2><p>A high shuffle-byte count is not automatically a defect. A wide transformation such as a large <code>join()</code> or <code>groupBy()</code> legitimately has to move that data, so the volume can be an inherent property of the query rather than something worth fixing. The byte thresholds that raise Info, Warning, and Critical levels are heuristic cutoffs, not measured limits for a given cluster, so treat them as a prompt to look rather than a verdict.</p><h2 id="related" tabindex="-1">Related <a class="header-anchor" href="#related" aria-label="Permalink to &quot;Related&quot;">​</a></h2><ul><li><strong>The mechanism:</strong> <a href="./shuffle.html">Shuffle</a></li><li><strong>Tuning parallelism:</strong> <a href="./partitioning.html">Partitioning</a></li></ul><hr class="footnotes-sep"><section class="footnotes"><ol class="footnotes-list"><li id="fn1" class="footnote-item"><p><em>Learning Spark, 2nd Edition</em>, Damji, Wenig, Das, Lee, ch. 7 <a href="#fnref1" class="footnote-backref">↩︎</a> <a href="#fnref1:1" class="footnote-backref">↩︎</a> <a href="#fnref1:2" class="footnote-backref">↩︎</a> <a href="#fnref1:3" class="footnote-backref">↩︎</a> <a href="#fnref1:4" class="footnote-backref">↩︎</a></p></li><li id="fn2" class="footnote-item"><p><a href="https://spark.apache.org/docs/latest/rdd-programming-guide.html" target="_blank" rel="noreferrer">RDD Programming Guide</a> <a href="#fnref2" class="footnote-backref">↩︎</a></p></li><li id="fn3" class="footnote-item"><p><a href="https://spark.apache.org/docs/latest/sql-performance-tuning.html" target="_blank" rel="noreferrer">Performance Tuning (Spark SQL, DataFrames and Datasets Guide)</a> <a href="#fnref3" class="footnote-backref">↩︎</a> <a href="#fnref3:1" class="footnote-backref">↩︎</a> <a href="#fnref3:2" class="footnote-backref">↩︎</a></p></li><li id="fn4" class="footnote-item"><p><a href="https://books.japila.pl/spark-sql-internals/bucketing/" target="_blank" rel="noreferrer">Bucketing (The Internals of Spark SQL)</a> <a href="#fnref4" class="footnote-backref">↩︎</a> <a href="#fnref4:1" class="footnote-backref">↩︎</a></p></li><li id="fn5" class="footnote-item"><p><a href="https://luminousmen.com/post/spark-tips-partition-tuning" target="_blank" rel="noreferrer">Spark Tips: Partition Tuning</a> <a href="#fnref5" class="footnote-backref">↩︎</a></p></li><li id="fn6" class="footnote-item"><p><a href="https://spark.apache.org/docs/latest/monitoring.html#spark-history-server" target="_blank" rel="noreferrer">Monitoring and Instrumentation</a> <a href="#fnref6" class="footnote-backref">↩︎</a> <a href="#fnref6:1" class="footnote-backref">↩︎</a> <a href="#fnref6:2" class="footnote-backref">↩︎</a> <a href="#fnref6:3" class="footnote-backref">↩︎</a></p></li><li id="fn7" class="footnote-item"><p><a href="https://issues.apache.org/jira/browse/SPARK-30602" target="_blank" rel="noreferrer">SPARK-30602: Support push-based shuffle to improve shuffle efficiency</a> <a href="#fnref7" class="footnote-backref">↩︎</a> <a href="#fnref7:1" class="footnote-backref">↩︎</a></p></li><li id="fn8" class="footnote-item"><p><a href="https://spark.apache.org/docs/latest/configuration.html" target="_blank" rel="noreferrer">Configuration (Spark)</a> <a href="#fnref8" class="footnote-backref">↩︎</a> <a href="#fnref8:1" class="footnote-backref">↩︎</a> <a href="#fnref8:2" class="footnote-backref">↩︎</a></p></li><li id="fn9" class="footnote-item"><p><a href="https://blog.cloudera.com/how-to-tune-your-apache-spark-jobs-part-2/" target="_blank" rel="noreferrer">How to Tune Your Apache Spark Jobs (Part 2): Cloudera Engineering Blog</a> <a href="#fnref9" class="footnote-backref">↩︎</a></p></li><li id="fn10" class="footnote-item"><p><a href="https://spark.apache.org/docs/latest/job-scheduling.html" target="_blank" rel="noreferrer">Job Scheduling: Dynamic Resource Allocation</a> <a href="#fnref10" class="footnote-backref">↩︎</a></p></li><li id="fn11" class="footnote-item"><p><a href="https://www.databricks.com/blog/2020/05/29/adaptive-query-execution-speeding-up-spark-sql-at-runtime.html" target="_blank" rel="noreferrer">Adaptive Query Execution: Speeding Up Spark SQL at Runtime (Databricks)</a> <a href="#fnref11" class="footnote-backref">↩︎</a> <a href="#fnref11:1" class="footnote-backref">↩︎</a></p></li><li id="fn12" class="footnote-item"><p><a href="https://raw.githubusercontent.com/apache/spark/v3.5.0/sql/catalyst/src/main/scala/org/apache/spark/sql/internal/SQLConf.scala" target="_blank" rel="noreferrer">SQLConf: shuffle-partition defaults (Spark source)</a> <a href="#fnref12" class="footnote-backref">↩︎</a></p></li><li id="fn13" class="footnote-item"><p><a href="https://spark.apache.org/docs/latest/sql-performance-tuning.html" target="_blank" rel="noreferrer">Performance Tuning: Coalescing Post Shuffle Partitions</a> <a href="#fnref13" class="footnote-backref">↩︎</a> <a href="#fnref13:1" class="footnote-backref">↩︎</a></p></li><li id="fn14" class="footnote-item"><p><a href="https://spark.apache.org/docs/latest/configuration.html" target="_blank" rel="noreferrer">Configuration (Spark)</a> <a href="#fnref14" class="footnote-backref">↩︎</a> <a href="#fnref14:1" class="footnote-backref">↩︎</a></p></li></ol></section>`,28)])])}const u=a(r,[["render",i]]);export{p as __pageData,u as default};
@@ -0,0 +1 @@
1
+ import{_ as a,o as t,c as s,a5 as o}from"./chunks/framework.DSg0KOwT.js";const p=JSON.parse('{"title":"Shuffle I/O","description":"","frontmatter":{"title":"Shuffle I/O"},"headers":[],"relativePath":"tuning-reference/bottleneck-shuffle.md","filePath":"tuning-reference/bottleneck-shuffle.md"}'),r={name:"tuning-reference/bottleneck-shuffle.md"};function i(n,e,f,l,c,d){return t(),s("div",null,[...e[0]||(e[0]=[o("",28)])])}const u=a(r,[["render",i]]);export{p as __pageData,u as default};