sparkforensics-cli 0.1.0 → 0.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (360) hide show
  1. package/README.md +6 -0
  2. package/bin/sparkforensics-analyze.mjs +113 -48
  3. package/export-template/docs/404.html +25 -0
  4. package/export-template/docs/assets/app.CndaAS6v.js +1 -0
  5. package/export-template/docs/assets/aqe-loop.IwQSATHw.svg +1 -0
  6. package/export-template/docs/assets/aqe-loop.dark.DGbaxqJE.svg +1 -0
  7. package/export-template/docs/assets/broadcast-vs-shuffle.Db4WY1XK.svg +1 -0
  8. package/export-template/docs/assets/broadcast-vs-shuffle.dark.C7Bxs0mG.svg +1 -0
  9. package/export-template/docs/assets/cache-lifecycle.dark.B-hS7AgU.svg +1 -0
  10. package/export-template/docs/assets/cache-lifecycle.rEOVYQNU.svg +1 -0
  11. package/export-template/docs/assets/chunks/@localSearchIndexroot.DNY8bVcl.js +1 -0
  12. package/export-template/docs/assets/chunks/VPLocalSearchBox.yJbZbsEo.js +9 -0
  13. package/export-template/docs/assets/chunks/duplicate-plan-subtree.dark.Cdp70QhV.js +1 -0
  14. package/export-template/docs/assets/chunks/framework.DSg0KOwT.js +20 -0
  15. package/export-template/docs/assets/chunks/retry-escalation-ladder.dark.DHipdJgZ.js +1 -0
  16. package/export-template/docs/assets/chunks/theme.Df2VAG9w.js +2 -0
  17. package/export-template/docs/assets/cold-start-timeline.DxC_Sc7w.svg +1 -0
  18. package/export-template/docs/assets/cold-start-timeline.dark.CZ17YcAG.svg +1 -0
  19. package/export-template/docs/assets/columnar-layout.PghGeOEA.svg +1 -0
  20. package/export-template/docs/assets/columnar-layout.dark.BVNlz0ff.svg +1 -0
  21. package/export-template/docs/assets/container-memory.DIO0AnIm.svg +1 -0
  22. package/export-template/docs/assets/container-memory.dark.CP-5zuCl.svg +1 -0
  23. package/export-template/docs/assets/contributor-guide_architecture_board-widgets.md.B-OsL91z.js +1 -0
  24. package/export-template/docs/assets/contributor-guide_architecture_board-widgets.md.B-OsL91z.lean.js +1 -0
  25. package/export-template/docs/assets/contributor-guide_architecture_detector-contract.md.BOeH4d1J.js +1 -0
  26. package/export-template/docs/assets/contributor-guide_architecture_detector-contract.md.BOeH4d1J.lean.js +1 -0
  27. package/export-template/docs/assets/contributor-guide_architecture_drill-down.md.BtPdlM7r.js +1 -0
  28. package/export-template/docs/assets/contributor-guide_architecture_drill-down.md.BtPdlM7r.lean.js +1 -0
  29. package/export-template/docs/assets/contributor-guide_architecture_impact-estimation.md.DYCDPgkh.js +1 -0
  30. package/export-template/docs/assets/contributor-guide_architecture_impact-estimation.md.DYCDPgkh.lean.js +1 -0
  31. package/export-template/docs/assets/contributor-guide_architecture_index.md.3TO9ic6w.js +1 -0
  32. package/export-template/docs/assets/contributor-guide_architecture_index.md.3TO9ic6w.lean.js +1 -0
  33. package/export-template/docs/assets/contributor-guide_architecture_overview.md.CehiRmGn.js +1 -0
  34. package/export-template/docs/assets/contributor-guide_architecture_overview.md.CehiRmGn.lean.js +1 -0
  35. package/export-template/docs/assets/contributor-guide_architecture_state-and-history.md.m3S3UdMk.js +1 -0
  36. package/export-template/docs/assets/contributor-guide_architecture_state-and-history.md.m3S3UdMk.lean.js +1 -0
  37. package/export-template/docs/assets/contributor-guide_architecture_widget-rendering.md.DbqPf2OT.js +1 -0
  38. package/export-template/docs/assets/contributor-guide_architecture_widget-rendering.md.DbqPf2OT.lean.js +1 -0
  39. package/export-template/docs/assets/contributor-guide_architecture_worker-protocol.md.B93qJ_tT.js +6 -0
  40. package/export-template/docs/assets/contributor-guide_architecture_worker-protocol.md.B93qJ_tT.lean.js +1 -0
  41. package/export-template/docs/assets/contributor-guide_contributing.md.CvRsdr6J.js +1 -0
  42. package/export-template/docs/assets/contributor-guide_contributing.md.CvRsdr6J.lean.js +1 -0
  43. package/export-template/docs/assets/contributor-guide_development-setup.md.DvAN_9mK.js +12 -0
  44. package/export-template/docs/assets/contributor-guide_development-setup.md.DvAN_9mK.lean.js +1 -0
  45. package/export-template/docs/assets/contributor-guide_testing.md.6rIKqSyY.js +1 -0
  46. package/export-template/docs/assets/contributor-guide_testing.md.6rIKqSyY.lean.js +1 -0
  47. package/export-template/docs/assets/dag-stages.DSz_S937.svg +1 -0
  48. package/export-template/docs/assets/dag-stages.dark.F72UzxH4.svg +1 -0
  49. package/export-template/docs/assets/driver-executor.D5pQ7YN1.svg +1 -0
  50. package/export-template/docs/assets/driver-executor.dark.BmX9cPvh.svg +1 -0
  51. package/export-template/docs/assets/duplicate-plan-subtree.B4cvN6fj.svg +1 -0
  52. package/export-template/docs/assets/duplicate-plan-subtree.dark.Dw8wS0Ag.svg +1 -0
  53. package/export-template/docs/assets/index.md.CHJVslga.js +1 -0
  54. package/export-template/docs/assets/index.md.CHJVslga.lean.js +1 -0
  55. package/export-template/docs/assets/inter-italic-cyrillic-ext.r48I6akx.woff2 +0 -0
  56. package/export-template/docs/assets/inter-italic-cyrillic.By2_1cv3.woff2 +0 -0
  57. package/export-template/docs/assets/inter-italic-greek-ext.1u6EdAuj.woff2 +0 -0
  58. package/export-template/docs/assets/inter-italic-greek.DJ8dCoTZ.woff2 +0 -0
  59. package/export-template/docs/assets/inter-italic-latin-ext.CN1xVJS-.woff2 +0 -0
  60. package/export-template/docs/assets/inter-italic-latin.C2AdPX0b.woff2 +0 -0
  61. package/export-template/docs/assets/inter-italic-vietnamese.BSbpV94h.woff2 +0 -0
  62. package/export-template/docs/assets/inter-roman-cyrillic-ext.BBPuwvHQ.woff2 +0 -0
  63. package/export-template/docs/assets/inter-roman-cyrillic.C5lxZ8CY.woff2 +0 -0
  64. package/export-template/docs/assets/inter-roman-greek-ext.CqjqNYQ-.woff2 +0 -0
  65. package/export-template/docs/assets/inter-roman-greek.BBVDIX6e.woff2 +0 -0
  66. package/export-template/docs/assets/inter-roman-latin-ext.4ZJIpNVo.woff2 +0 -0
  67. package/export-template/docs/assets/inter-roman-latin.Di8DUHzh.woff2 +0 -0
  68. package/export-template/docs/assets/inter-roman-vietnamese.BjW4sHH5.woff2 +0 -0
  69. package/export-template/docs/assets/join-strategy.C_FvrCEo.svg +1 -0
  70. package/export-template/docs/assets/join-strategy.dark.ChMLnNII.svg +1 -0
  71. package/export-template/docs/assets/memory-borrowing.BqQRJg0u.svg +1 -0
  72. package/export-template/docs/assets/memory-borrowing.dark.Yhh20O9C.svg +1 -0
  73. package/export-template/docs/assets/memory-regions.XHvO7jHG.svg +1 -0
  74. package/export-template/docs/assets/memory-regions.dark.D4TP9_08.svg +1 -0
  75. package/export-template/docs/assets/repartition-vs-coalesce.BovLRrpj.svg +1 -0
  76. package/export-template/docs/assets/repartition-vs-coalesce.dark.BhAczKZQ.svg +1 -0
  77. package/export-template/docs/assets/retry-escalation-ladder.DyTKJJmZ.svg +1 -0
  78. package/export-template/docs/assets/retry-escalation-ladder.dark.BdsabtU3.svg +1 -0
  79. package/export-template/docs/assets/shuffle-map-reduce.KuOEZVmg.svg +1 -0
  80. package/export-template/docs/assets/shuffle-map-reduce.dark.BgQZnFSb.svg +1 -0
  81. package/export-template/docs/assets/spill-classification.BU2euYDO.svg +1 -0
  82. package/export-template/docs/assets/spill-classification.dark.D7i1M40d.svg +1 -0
  83. package/export-template/docs/assets/style.DXOMCXxn.css +1 -0
  84. package/export-template/docs/assets/tuning-reference_anti-patterns.md.Df1YMIHu.js +1 -0
  85. package/export-template/docs/assets/tuning-reference_anti-patterns.md.Df1YMIHu.lean.js +1 -0
  86. package/export-template/docs/assets/tuning-reference_aqe.md.BIsCtLzm.js +1 -0
  87. package/export-template/docs/assets/tuning-reference_aqe.md.BIsCtLzm.lean.js +1 -0
  88. package/export-template/docs/assets/tuning-reference_bottleneck-broadcast-sizing.md.CEstB3Ia.js +1 -0
  89. package/export-template/docs/assets/tuning-reference_bottleneck-broadcast-sizing.md.CEstB3Ia.lean.js +1 -0
  90. package/export-template/docs/assets/tuning-reference_bottleneck-cold-start.md.CEuy-72y.js +7 -0
  91. package/export-template/docs/assets/tuning-reference_bottleneck-cold-start.md.CEuy-72y.lean.js +1 -0
  92. package/export-template/docs/assets/tuning-reference_bottleneck-duplicate-plan-subtree.md.CIohQDfn.js +1 -0
  93. package/export-template/docs/assets/tuning-reference_bottleneck-duplicate-plan-subtree.md.CIohQDfn.lean.js +1 -0
  94. package/export-template/docs/assets/tuning-reference_bottleneck-failures.md.4z5BXGJ2.js +6 -0
  95. package/export-template/docs/assets/tuning-reference_bottleneck-failures.md.4z5BXGJ2.lean.js +1 -0
  96. package/export-template/docs/assets/tuning-reference_bottleneck-gc.md.DSxzZRK7.js +6 -0
  97. package/export-template/docs/assets/tuning-reference_bottleneck-gc.md.DSxzZRK7.lean.js +1 -0
  98. package/export-template/docs/assets/tuning-reference_bottleneck-job-failure-rate.md.BaJl__1W.js +8 -0
  99. package/export-template/docs/assets/tuning-reference_bottleneck-job-failure-rate.md.BaJl__1W.lean.js +1 -0
  100. package/export-template/docs/assets/tuning-reference_bottleneck-memory-utilization.md.DbP-SJZc.js +7 -0
  101. package/export-template/docs/assets/tuning-reference_bottleneck-memory-utilization.md.DbP-SJZc.lean.js +1 -0
  102. package/export-template/docs/assets/tuning-reference_bottleneck-retry-waste.md.D5JMjVOt.js +1 -0
  103. package/export-template/docs/assets/tuning-reference_bottleneck-retry-waste.md.D5JMjVOt.lean.js +1 -0
  104. package/export-template/docs/assets/tuning-reference_bottleneck-shuffle.md.CM-nTmIH.js +12 -0
  105. package/export-template/docs/assets/tuning-reference_bottleneck-shuffle.md.CM-nTmIH.lean.js +1 -0
  106. package/export-template/docs/assets/tuning-reference_bottleneck-skew.md.BdUwiDhn.js +14 -0
  107. package/export-template/docs/assets/tuning-reference_bottleneck-skew.md.BdUwiDhn.lean.js +1 -0
  108. package/export-template/docs/assets/tuning-reference_bottleneck-slow-host.md.BlIo6UDW.js +7 -0
  109. package/export-template/docs/assets/tuning-reference_bottleneck-slow-host.md.BlIo6UDW.lean.js +1 -0
  110. package/export-template/docs/assets/tuning-reference_bottleneck-small-files.md.B8kloyx8.js +5 -0
  111. package/export-template/docs/assets/tuning-reference_bottleneck-small-files.md.B8kloyx8.lean.js +1 -0
  112. package/export-template/docs/assets/tuning-reference_bottleneck-spill.md.PNH7mITt.js +6 -0
  113. package/export-template/docs/assets/tuning-reference_bottleneck-spill.md.PNH7mITt.lean.js +1 -0
  114. package/export-template/docs/assets/tuning-reference_bottleneck-straggler.md.DY36fHN5.js +7 -0
  115. package/export-template/docs/assets/tuning-reference_bottleneck-straggler.md.DY36fHN5.lean.js +1 -0
  116. package/export-template/docs/assets/tuning-reference_bottleneck-tiny-tasks.md.QTV7O8kU.js +8 -0
  117. package/export-template/docs/assets/tuning-reference_bottleneck-tiny-tasks.md.QTV7O8kU.lean.js +1 -0
  118. package/export-template/docs/assets/tuning-reference_bottleneck-utilization.md.DTiueZC3.js +5 -0
  119. package/export-template/docs/assets/tuning-reference_bottleneck-utilization.md.DTiueZC3.lean.js +1 -0
  120. package/export-template/docs/assets/tuning-reference_caching.md.B7aQ8asB.js +1 -0
  121. package/export-template/docs/assets/tuning-reference_caching.md.B7aQ8asB.lean.js +1 -0
  122. package/export-template/docs/assets/tuning-reference_cluster-config.md.ZVmDGsQ3.js +1 -0
  123. package/export-template/docs/assets/tuning-reference_cluster-config.md.ZVmDGsQ3.lean.js +1 -0
  124. package/export-template/docs/assets/tuning-reference_config.md.UvveiWG3.js +1 -0
  125. package/export-template/docs/assets/tuning-reference_config.md.UvveiWG3.lean.js +1 -0
  126. package/export-template/docs/assets/tuning-reference_data-formats.md.bjCAWH3N.js +1 -0
  127. package/export-template/docs/assets/tuning-reference_data-formats.md.bjCAWH3N.lean.js +1 -0
  128. package/export-template/docs/assets/tuning-reference_index.md.BQ_NooMV.js +1 -0
  129. package/export-template/docs/assets/tuning-reference_index.md.BQ_NooMV.lean.js +1 -0
  130. package/export-template/docs/assets/tuning-reference_intro.md.CobD-lGB.js +1 -0
  131. package/export-template/docs/assets/tuning-reference_intro.md.CobD-lGB.lean.js +1 -0
  132. package/export-template/docs/assets/tuning-reference_joins.md.BtKs_CuW.js +1 -0
  133. package/export-template/docs/assets/tuning-reference_joins.md.BtKs_CuW.lean.js +1 -0
  134. package/export-template/docs/assets/tuning-reference_memory-model.md.DhT-n4y3.js +1 -0
  135. package/export-template/docs/assets/tuning-reference_memory-model.md.DhT-n4y3.lean.js +1 -0
  136. package/export-template/docs/assets/tuning-reference_metrics.md.mLOh7Apj.js +1 -0
  137. package/export-template/docs/assets/tuning-reference_metrics.md.mLOh7Apj.lean.js +1 -0
  138. package/export-template/docs/assets/tuning-reference_partitioning.md.q0zKF_8X.js +1 -0
  139. package/export-template/docs/assets/tuning-reference_partitioning.md.q0zKF_8X.lean.js +1 -0
  140. package/export-template/docs/assets/tuning-reference_pyspark.md.DDCfvN9t.js +6 -0
  141. package/export-template/docs/assets/tuning-reference_pyspark.md.DDCfvN9t.lean.js +1 -0
  142. package/export-template/docs/assets/tuning-reference_shuffle.md.BZZ7R4Ix.js +1 -0
  143. package/export-template/docs/assets/tuning-reference_shuffle.md.BZZ7R4Ix.lean.js +1 -0
  144. package/export-template/docs/assets/tuning-reference_spark-architecture.md.Dwzm5avO.js +1 -0
  145. package/export-template/docs/assets/tuning-reference_spark-architecture.md.Dwzm5avO.lean.js +1 -0
  146. package/export-template/docs/assets/tuning-reference_table-formats.md.D6wj-2dX.js +1 -0
  147. package/export-template/docs/assets/tuning-reference_table-formats.md.D6wj-2dX.lean.js +1 -0
  148. package/export-template/docs/assets/udf-execution-models.BUFDICuG.svg +1 -0
  149. package/export-template/docs/assets/udf-execution-models.dark.YTNS6GDq.svg +1 -0
  150. package/export-template/docs/assets/user-guide_alternative-log-retrieval.md.sU3KGarf.js +1 -0
  151. package/export-template/docs/assets/user-guide_alternative-log-retrieval.md.sU3KGarf.lean.js +1 -0
  152. package/export-template/docs/assets/user-guide_getting-started.md.DtEM37MK.js +3 -0
  153. package/export-template/docs/assets/user-guide_getting-started.md.DtEM37MK.lean.js +1 -0
  154. package/export-template/docs/assets/user-guide_mcp-tools.md.C8MiIu7F.js +125 -0
  155. package/export-template/docs/assets/user-guide_mcp-tools.md.C8MiIu7F.lean.js +1 -0
  156. package/export-template/docs/assets/user-guide_run-comparison.md.S0TWWmLY.js +1 -0
  157. package/export-template/docs/assets/user-guide_run-comparison.md.S0TWWmLY.lean.js +1 -0
  158. package/export-template/docs/assets/user-guide_understanding-findings.md.D0R_Y-R2.js +1 -0
  159. package/export-template/docs/assets/user-guide_understanding-findings.md.D0R_Y-R2.lean.js +1 -0
  160. package/export-template/docs/contributor-guide/architecture/board-widgets.html +25 -0
  161. package/export-template/docs/contributor-guide/architecture/detector-contract.html +25 -0
  162. package/export-template/docs/contributor-guide/architecture/drill-down.html +25 -0
  163. package/export-template/docs/contributor-guide/architecture/impact-estimation.html +25 -0
  164. package/export-template/docs/contributor-guide/architecture/index.html +25 -0
  165. package/export-template/docs/contributor-guide/architecture/overview.html +25 -0
  166. package/export-template/docs/contributor-guide/architecture/state-and-history.html +25 -0
  167. package/export-template/docs/contributor-guide/architecture/widget-rendering.html +25 -0
  168. package/export-template/docs/contributor-guide/architecture/worker-protocol.html +30 -0
  169. package/export-template/docs/contributor-guide/contributing.html +25 -0
  170. package/export-template/docs/contributor-guide/development-setup.html +36 -0
  171. package/export-template/docs/contributor-guide/testing.html +25 -0
  172. package/export-template/docs/favicon.svg +4 -0
  173. package/export-template/docs/hashmap.json +1 -0
  174. package/export-template/docs/index.html +25 -0
  175. package/export-template/docs/package.json +1 -0
  176. package/export-template/docs/tuning-reference/anti-patterns.html +25 -0
  177. package/export-template/docs/tuning-reference/aqe.html +25 -0
  178. package/export-template/docs/tuning-reference/bottleneck-broadcast-sizing.html +25 -0
  179. package/export-template/docs/tuning-reference/bottleneck-cold-start.html +31 -0
  180. package/export-template/docs/tuning-reference/bottleneck-duplicate-plan-subtree.html +25 -0
  181. package/export-template/docs/tuning-reference/bottleneck-failures.html +30 -0
  182. package/export-template/docs/tuning-reference/bottleneck-gc.html +30 -0
  183. package/export-template/docs/tuning-reference/bottleneck-job-failure-rate.html +32 -0
  184. package/export-template/docs/tuning-reference/bottleneck-memory-utilization.html +31 -0
  185. package/export-template/docs/tuning-reference/bottleneck-retry-waste.html +25 -0
  186. package/export-template/docs/tuning-reference/bottleneck-shuffle.html +36 -0
  187. package/export-template/docs/tuning-reference/bottleneck-skew.html +38 -0
  188. package/export-template/docs/tuning-reference/bottleneck-slow-host.html +31 -0
  189. package/export-template/docs/tuning-reference/bottleneck-small-files.html +29 -0
  190. package/export-template/docs/tuning-reference/bottleneck-spill.html +30 -0
  191. package/export-template/docs/tuning-reference/bottleneck-straggler.html +31 -0
  192. package/export-template/docs/tuning-reference/bottleneck-tiny-tasks.html +32 -0
  193. package/export-template/docs/tuning-reference/bottleneck-utilization.html +29 -0
  194. package/export-template/docs/tuning-reference/caching.html +25 -0
  195. package/export-template/docs/tuning-reference/cluster-config.html +25 -0
  196. package/export-template/docs/tuning-reference/config.html +25 -0
  197. package/export-template/docs/tuning-reference/data-formats.html +25 -0
  198. package/export-template/docs/tuning-reference/index.html +25 -0
  199. package/export-template/docs/tuning-reference/intro.html +25 -0
  200. package/export-template/docs/tuning-reference/joins.html +25 -0
  201. package/export-template/docs/tuning-reference/memory-model.html +25 -0
  202. package/export-template/docs/tuning-reference/metrics.html +25 -0
  203. package/export-template/docs/tuning-reference/partitioning.html +25 -0
  204. package/export-template/docs/tuning-reference/pyspark.html +30 -0
  205. package/export-template/docs/tuning-reference/shuffle.html +25 -0
  206. package/export-template/docs/tuning-reference/spark-architecture.html +25 -0
  207. package/export-template/docs/tuning-reference/table-formats.html +25 -0
  208. package/export-template/docs/user-guide/alternative-log-retrieval.html +25 -0
  209. package/export-template/docs/user-guide/getting-started.html +27 -0
  210. package/export-template/docs/user-guide/mcp-tools.html +149 -0
  211. package/export-template/docs/user-guide/run-comparison.html +25 -0
  212. package/export-template/docs/user-guide/understanding-findings.html +25 -0
  213. package/export-template/docs/vp-icons.css +0 -0
  214. package/export-template/favicon.svg +4 -0
  215. package/export-template/index.html +115 -0
  216. package/export-template/parser-worker-QqyEE4m9.js +64 -0
  217. package/package.json +16 -3
  218. package/vendor-core/analyzer.js +74 -74
  219. package/vendor-core/cli/budgets.js +13 -27
  220. package/vendor-core/cli/collect-run.js +43 -19
  221. package/vendor-core/core-count.js +25 -27
  222. package/vendor-core/core-locality-ratio.js +4 -11
  223. package/vendor-core/core-time-series.js +6 -12
  224. package/vendor-core/core-usage-locality.js +3 -4
  225. package/vendor-core/detectors.js +256 -375
  226. package/vendor-core/docs-config.js +69 -21
  227. package/vendor-core/docs-content/chapters/01-intro.md +32 -0
  228. package/vendor-core/docs-content/chapters/02-spark-architecture.md +76 -0
  229. package/vendor-core/docs-content/chapters/03-memory-model.md +73 -0
  230. package/vendor-core/docs-content/chapters/04-partitioning.md +65 -0
  231. package/vendor-core/docs-content/chapters/05-joins.md +62 -0
  232. package/vendor-core/docs-content/chapters/06-shuffle.md +59 -0
  233. package/vendor-core/docs-content/chapters/07-data-formats.md +81 -0
  234. package/vendor-core/docs-content/chapters/07b-table-formats.md +56 -0
  235. package/vendor-core/docs-content/chapters/08-caching.md +58 -0
  236. package/vendor-core/docs-content/chapters/09-pyspark.md +78 -0
  237. package/vendor-core/docs-content/chapters/10-aqe.md +167 -0
  238. package/vendor-core/docs-content/chapters/11-cluster-config.md +170 -0
  239. package/vendor-core/docs-content/chapters/12-anti-patterns.md +171 -0
  240. package/vendor-core/docs-content/chapters/14-metrics.md +87 -0
  241. package/vendor-core/docs-content/chapters/15-config.md +93 -0
  242. package/vendor-core/docs-content/chapters/nav-index.json +370 -0
  243. package/vendor-core/docs-content/detection/cache.md +6 -0
  244. package/vendor-core/docs-content/detection/cfg.md +15 -0
  245. package/vendor-core/docs-content/detection/chrn.md +7 -0
  246. package/vendor-core/docs-content/detection/cold.md +4 -0
  247. package/vendor-core/docs-content/detection/cstor.md +4 -0
  248. package/vendor-core/docs-content/detection/fail.md +5 -0
  249. package/vendor-core/docs-content/detection/gc.md +4 -0
  250. package/vendor-core/docs-content/detection/host.md +5 -0
  251. package/vendor-core/docs-content/detection/incmp.md +6 -0
  252. package/vendor-core/docs-content/detection/jobs.md +4 -0
  253. package/vendor-core/docs-content/detection/local.md +7 -0
  254. package/vendor-core/docs-content/detection/mem.md +10 -0
  255. package/vendor-core/docs-content/detection/part.md +5 -0
  256. package/vendor-core/docs-content/detection/plan.md +14 -0
  257. package/vendor-core/docs-content/detection/retry.md +4 -0
  258. package/vendor-core/docs-content/detection/sfail.md +5 -0
  259. package/vendor-core/docs-content/detection/shape.md +5 -0
  260. package/vendor-core/docs-content/detection/shfl.md +4 -0
  261. package/vendor-core/docs-content/detection/skew.md +6 -0
  262. package/vendor-core/docs-content/detection/slow.md +6 -0
  263. package/vendor-core/docs-content/detection/spec.md +7 -0
  264. package/vendor-core/docs-content/detection/spill.md +7 -0
  265. package/vendor-core/docs-content/detection/strag.md +5 -0
  266. package/vendor-core/docs-content/detection/tiny.md +4 -0
  267. package/vendor-core/docs-content/detection/util.md +4 -0
  268. package/vendor-core/docs-content/diagrams/aqe-loop.dark.svg +1 -0
  269. package/vendor-core/docs-content/diagrams/aqe-loop.svg +1 -0
  270. package/vendor-core/docs-content/diagrams/broadcast-vs-shuffle.dark.svg +1 -0
  271. package/vendor-core/docs-content/diagrams/broadcast-vs-shuffle.svg +1 -0
  272. package/vendor-core/docs-content/diagrams/cache-lifecycle.dark.svg +1 -0
  273. package/vendor-core/docs-content/diagrams/cache-lifecycle.svg +1 -0
  274. package/vendor-core/docs-content/diagrams/cold-start-timeline.dark.svg +1 -0
  275. package/vendor-core/docs-content/diagrams/cold-start-timeline.svg +1 -0
  276. package/vendor-core/docs-content/diagrams/columnar-layout.dark.svg +1 -0
  277. package/vendor-core/docs-content/diagrams/columnar-layout.svg +1 -0
  278. package/vendor-core/docs-content/diagrams/container-memory.dark.svg +1 -0
  279. package/vendor-core/docs-content/diagrams/container-memory.svg +1 -0
  280. package/vendor-core/docs-content/diagrams/dag-stages.dark.svg +1 -0
  281. package/vendor-core/docs-content/diagrams/dag-stages.svg +1 -0
  282. package/vendor-core/docs-content/diagrams/driver-executor.dark.svg +1 -0
  283. package/vendor-core/docs-content/diagrams/driver-executor.svg +1 -0
  284. package/vendor-core/docs-content/diagrams/duplicate-plan-subtree.dark.svg +1 -0
  285. package/vendor-core/docs-content/diagrams/duplicate-plan-subtree.svg +1 -0
  286. package/vendor-core/docs-content/diagrams/join-strategy.dark.svg +1 -0
  287. package/vendor-core/docs-content/diagrams/join-strategy.svg +1 -0
  288. package/vendor-core/docs-content/diagrams/memory-borrowing.dark.svg +1 -0
  289. package/vendor-core/docs-content/diagrams/memory-borrowing.svg +1 -0
  290. package/vendor-core/docs-content/diagrams/memory-regions.dark.svg +1 -0
  291. package/vendor-core/docs-content/diagrams/memory-regions.svg +1 -0
  292. package/vendor-core/docs-content/diagrams/repartition-vs-coalesce.dark.svg +1 -0
  293. package/vendor-core/docs-content/diagrams/repartition-vs-coalesce.svg +1 -0
  294. package/vendor-core/docs-content/diagrams/retry-escalation-ladder.dark.svg +1 -0
  295. package/vendor-core/docs-content/diagrams/retry-escalation-ladder.svg +1 -0
  296. package/vendor-core/docs-content/diagrams/shuffle-map-reduce.dark.svg +1 -0
  297. package/vendor-core/docs-content/diagrams/shuffle-map-reduce.svg +1 -0
  298. package/vendor-core/docs-content/diagrams/spill-classification.dark.svg +1 -0
  299. package/vendor-core/docs-content/diagrams/spill-classification.svg +1 -0
  300. package/vendor-core/docs-content/diagrams/udf-execution-models.dark.svg +1 -0
  301. package/vendor-core/docs-content/diagrams/udf-execution-models.svg +1 -0
  302. package/vendor-core/docs-content/tuning/broadcast-sizing.md +78 -0
  303. package/vendor-core/docs-content/tuning/cold-start.md +81 -0
  304. package/vendor-core/docs-content/tuning/duplicate-plan-subtree.md +45 -0
  305. package/vendor-core/docs-content/tuning/failures.md +124 -0
  306. package/vendor-core/docs-content/tuning/gc.md +110 -0
  307. package/vendor-core/docs-content/tuning/job-failure-rate.md +101 -0
  308. package/vendor-core/docs-content/tuning/memory-utilization.md +58 -0
  309. package/vendor-core/docs-content/tuning/retry-waste.md +90 -0
  310. package/vendor-core/docs-content/tuning/shuffle.md +154 -0
  311. package/vendor-core/docs-content/tuning/skew.md +123 -0
  312. package/vendor-core/docs-content/tuning/slow-host.md +117 -0
  313. package/vendor-core/docs-content/tuning/small-files.md +99 -0
  314. package/vendor-core/docs-content/tuning/spill.md +114 -0
  315. package/vendor-core/docs-content/tuning/straggler.md +103 -0
  316. package/vendor-core/docs-content/tuning/tiny-tasks.md +94 -0
  317. package/vendor-core/docs-content/tuning/utilization.md +90 -0
  318. package/vendor-core/docs-site-config.js +10 -17
  319. package/vendor-core/efficiency-model.js +7 -13
  320. package/vendor-core/etl-phases.js +3 -5
  321. package/vendor-core/event-handlers.js +232 -134
  322. package/vendor-core/event-schemas.js +48 -114
  323. package/vendor-core/evidence-availability.js +5 -10
  324. package/vendor-core/evidence-report.js +72 -122
  325. package/vendor-core/export-data.js +48 -0
  326. package/vendor-core/finding-action-label.js +4 -10
  327. package/vendor-core/finding-filter-predicate.js +3 -7
  328. package/vendor-core/finding-generic-recommendation.js +112 -0
  329. package/vendor-core/finding-names.js +51 -0
  330. package/vendor-core/format-utils.js +112 -38
  331. package/vendor-core/impact-band.js +18 -24
  332. package/vendor-core/impact-estimator.js +38 -74
  333. package/vendor-core/ingest.js +7 -13
  334. package/vendor-core/job-groups.js +3 -6
  335. package/vendor-core/list-runs.js +278 -0
  336. package/vendor-core/load-vendored.js +6 -12
  337. package/vendor-core/log-header-peek.js +81 -0
  338. package/vendor-core/lz4-block.js +4 -6
  339. package/vendor-core/mcp-server-factory.js +38 -8
  340. package/vendor-core/mcp-tools.js +105 -76
  341. package/vendor-core/model-assembler.js +8 -16
  342. package/vendor-core/occupancy.js +5 -9
  343. package/vendor-core/parser-worker.js +18 -27
  344. package/vendor-core/plan-dot.js +2 -5
  345. package/vendor-core/plan-duration-attribution.js +78 -29
  346. package/vendor-core/plan-graph-model.js +126 -69
  347. package/vendor-core/plan-node-detail.js +31 -17
  348. package/vendor-core/plan-summary.js +19 -8
  349. package/vendor-core/recommendation-rollup.js +35 -39
  350. package/vendor-core/redact.js +72 -16
  351. package/vendor-core/rolling-log-reassembly.js +4 -6
  352. package/vendor-core/run-comparison.js +65 -70
  353. package/vendor-core/scaling-sim.js +5 -7
  354. package/vendor-core/session-snapshot.js +1 -1
  355. package/vendor-core/shs-fetch.js +4 -6
  356. package/vendor-core/shs-load.js +9 -13
  357. package/vendor-core/shs-request.js +1 -1
  358. package/vendor-core/stage-quantiles.js +14 -0
  359. package/vendor-core/types.js +78 -18
  360. package/vendor-core/wasted-core-hours.js +7 -12
@@ -0,0 +1 @@
1
+ import{_ as a,o,c as t,a5 as r}from"./chunks/framework.DSg0KOwT.js";const u=JSON.parse('{"title":"Spark Config Quick-Reference","description":"","frontmatter":{"title":"Spark Config Quick-Reference"},"headers":[],"relativePath":"tuning-reference/config.md","filePath":"tuning-reference/config.md"}'),s={name:"tuning-reference/config.md"};function f(n,e,c,i,d,l){return o(),t("div",null,[...e[0]||(e[0]=[r('<h1 id="config" tabindex="-1">Spark Config Quick-Reference <a class="header-anchor" href="#config" aria-label="Permalink to &quot;Spark Config Quick-Reference {#config}&quot;">​</a></h1><p>A cross-reference of the Spark and PySpark configuration properties discussed elsewhere in this guide, plus the JVM garbage-collection flags relevant to executor tuning. Defaults and version notes below are traced to Spark&#39;s own configuration docs, its SQLConf source, and the cited literature. Where a source doesn&#39;t give a hard number (a range, a ceiling, a recommended workload profile), the table leaves it out rather than guess.</p><h2 id="spark-properties" tabindex="-1">Spark properties <a class="header-anchor" href="#spark-properties" aria-label="Permalink to &quot;Spark properties&quot;">​</a></h2><table tabindex="0"><thead><tr><th>Key</th><th>Default</th><th>Recommended</th><th>Notes</th></tr></thead><tbody><tr><td><code>spark.sql.shuffle.partitions</code></td><td>200 (unchanged since Spark 1.1.0)<sup class="footnote-ref"><a href="#fn1" id="fnref1">[1]</a></sup><sup class="footnote-ref"><a href="#fn2" id="fnref2">[2]</a></sup><sup class="footnote-ref"><a href="#fn3" id="fnref3">[3]</a></sup></td><td></td><td>AQE doesn&#39;t override this config value; see notes below for the effective runtime count.</td></tr><tr><td><code>spark.sql.adaptive.enabled</code></td><td><code>true</code><sup class="footnote-ref"><a href="#fn4" id="fnref4">[4]</a></sup></td><td><code>true</code></td><td></td></tr><tr><td><code>spark.sql.adaptive.coalescePartitions.enabled</code></td><td><code>true</code> (since 3.0.0)<sup class="footnote-ref"><a href="#fn1" id="fnref1:1">[1:1]</a></sup></td><td><code>true</code></td><td>Requires AQE; merges contiguous small post-shuffle partitions instead of one task per configured partition<sup class="footnote-ref"><a href="#fn1" id="fnref1:2">[1:2]</a></sup>.</td></tr><tr><td><code>spark.sql.adaptive.coalescePartitions.initialPartitionNum</code></td><td>unset → falls back to <code>spark.sql.shuffle.partitions</code> (200)<sup class="footnote-ref"><a href="#fn1" id="fnref1:3">[1:3]</a></sup></td><td></td><td>Sets the partition count entering the coalescing step.</td></tr><tr><td><code>spark.sql.adaptive.coalescePartitions.parallelismFirst</code></td><td><code>true</code> (since 3.2.0)<sup class="footnote-ref"><a href="#fn1" id="fnref1:4">[1:4]</a></sup><sup class="footnote-ref"><a href="#fn5" id="fnref5">[5]</a></sup></td><td><code>false</code> on busy clusters<sup class="footnote-ref"><a href="#fn1" id="fnref1:5">[1:5]</a></sup></td><td>When <code>true</code>, ignores <code>advisoryPartitionSizeInBytes</code> and derives a target from the cluster&#39;s default parallelism, respecting only the <code>minPartitionSize</code> floor.</td></tr><tr><td><code>spark.sql.adaptive.advisoryPartitionSizeInBytes</code></td><td>64MB (since 3.0.0)<sup class="footnote-ref"><a href="#fn1" id="fnref1:6">[1:6]</a></sup><sup class="footnote-ref"><a href="#fn4" id="fnref4:1">[4:1]</a></sup><sup class="footnote-ref"><a href="#fn5" id="fnref5:1">[5:1]</a></sup></td><td></td><td>Ignored during coalescing when <code>parallelismFirst=true</code> (the default).</td></tr><tr><td><code>spark.sql.adaptive.coalescePartitions.minPartitionSize</code></td><td>1MB (since 3.2.0)<sup class="footnote-ref"><a href="#fn5" id="fnref5:2">[5:2]</a></sup></td><td></td><td>Floor enforced when the target size is ignored: the default <code>parallelismFirst=true</code> case<sup class="footnote-ref"><a href="#fn1" id="fnref1:7">[1:7]</a></sup>.</td></tr><tr><td><code>spark.memory.fraction</code></td><td>0.6<sup class="footnote-ref"><a href="#fn6" id="fnref6">[6]</a></sup></td><td></td><td>See notes below; no hard 0–1 ceiling documented in the cited sources.</td></tr><tr><td><code>spark.memory.storageFraction</code></td><td>0.5<sup class="footnote-ref"><a href="#fn7" id="fnref7">[7]</a></sup></td><td></td><td>A fraction <em>of</em> the region sized by <code>spark.memory.fraction</code>, not an independent pool; see notes below.</td></tr><tr><td><code>spark.sql.execution.arrow.pyspark.enabled</code></td><td><code>false</code><sup class="footnote-ref"><a href="#fn8" id="fnref8">[8]</a></sup></td><td></td><td>See PySpark callout below.</td></tr><tr><td><code>spark.sql.execution.arrow.pyspark.fallback.enabled</code></td><td><code>true</code> (inherited from the deprecated <code>arrow.fallback.enabled</code>)<sup class="footnote-ref"><a href="#fn8" id="fnref8:1">[8:1]</a></sup><sup class="footnote-ref"><a href="#fn5" id="fnref5:3">[5:3]</a></sup><sup class="footnote-ref"><a href="#fn4" id="fnref4:2">[4:2]</a></sup></td><td></td><td>Falls back to the non-Arrow path automatically if a conversion error occurs, before computation runs<sup class="footnote-ref"><a href="#fn8" id="fnref8:2">[8:2]</a></sup>.</td></tr><tr><td><code>spark.dynamicAllocation.shuffleTracking.enabled</code></td><td><code>true</code> (since 3.0.0)<sup class="footnote-ref"><a href="#fn4" id="fnref4:3">[4:3]</a></sup></td><td></td><td>Lets dynamic allocation track shuffle files per executor without an external shuffle service.</td></tr><tr><td><code>spark.dynamicAllocation.shuffleTracking.timeout</code></td><td><code>infinity</code><sup class="footnote-ref"><a href="#fn4" id="fnref4:4">[4:4]</a></sup></td><td></td><td>Executors holding shuffle data wait for it to be garbage collected before release, by default; set a finite value if GC isn&#39;t keeping up.</td></tr><tr><td><code>spark.shuffle.service.enabled</code></td><td></td><td></td><td>External shuffle service; see <code>#config-shuffle-service</code> below.</td></tr><tr><td><code>spark.dynamicAllocation.minExecutors</code> / <code>maxExecutors</code></td><td><code>0</code> / <code>infinity</code><sup class="footnote-ref"><a href="#fn4" id="fnref4:5">[4:5]</a></sup></td><td></td><td>Autoscale bounds; see <code>#config-autoscale-bounds</code> below.</td></tr><tr><td><code>spark.serializer</code></td><td><code>org.apache.spark.serializer.JavaSerializer</code><sup class="footnote-ref"><a href="#fn4" id="fnref4:6">[4:6]</a></sup></td><td><code>KryoSerializer</code></td><td>See <code>#config-serializer</code> below.</td></tr><tr><td><code>spark.executor.memoryOverhead</code></td><td>greater of 10% of executor memory or 384MB<sup class="footnote-ref"><a href="#fn7" id="fnref7:1">[7:1]</a></sup></td><td></td><td>Off-heap overhead; see <code>#config-memory-overhead</code> below.</td></tr></tbody></table><h3 id="shuffle-partitions-and-aqe-coalescing" tabindex="-1">Shuffle partitions and AQE coalescing <a class="header-anchor" href="#shuffle-partitions-and-aqe-coalescing" aria-label="Permalink to &quot;Shuffle partitions and AQE coalescing&quot;">​</a></h3><p>The 200 default for <code>spark.sql.shuffle.partitions</code> hasn&#39;t moved since 1.1.0, and AQE (on by default since Spark 3.0) doesn&#39;t change that config value<sup class="footnote-ref"><a href="#fn1" id="fnref1:8">[1:8]</a></sup><sup class="footnote-ref"><a href="#fn2" id="fnref2:1">[2:1]</a></sup><sup class="footnote-ref"><a href="#fn3" id="fnref3:1">[3:1]</a></sup>. What it changes is the effective number of partitions used at runtime. With <code>coalescePartitions.enabled</code> also on by default, Spark merges small contiguous post-shuffle partitions instead of running one task per configured partition<sup class="footnote-ref"><a href="#fn1" id="fnref1:9">[1:9]</a></sup>. Because <code>parallelismFirst</code> defaults to <code>true</code> since 3.2.0, that merge target usually isn&#39;t the 64MB <code>advisoryPartitionSizeInBytes</code> value; it&#39;s derived from the cluster&#39;s default parallelism, with <code>minPartitionSize</code> (1MB) as the only enforced floor<sup class="footnote-ref"><a href="#fn1" id="fnref1:10">[1:10]</a></sup><sup class="footnote-ref"><a href="#fn5" id="fnref5:4">[5:4]</a></sup>. Databricks&#39; own writeup on AQE shows the reduce-task count actually shrinking based on measured data volume<sup class="footnote-ref"><a href="#fn9" id="fnref9">[9]</a></sup>, and Learning Spark accordingly calls the static 200 default &quot;too high for smaller or streaming workloads&quot;<sup class="footnote-ref"><a href="#fn3" id="fnref3:2">[3:2]</a></sup>. A related, distinct knob (<code>spark.sql.adaptive.coalescePartitions.minPartitionNum</code>) sets a minimum parallelism floor for data that&#39;s slow to compute despite being small, per High Performance Spark<sup class="footnote-ref"><a href="#fn10" id="fnref10">[10]</a></sup>.</p><h3 id="spark-memory-fraction-and-spark-memory-storagefraction" tabindex="-1"><code>spark.memory.fraction</code> and <code>spark.memory.storageFraction</code> <a class="header-anchor" href="#spark-memory-fraction-and-spark-memory-storagefraction" aria-label="Permalink to &quot;`spark.memory.fraction` and `spark.memory.storageFraction`&quot;">​</a></h3><p><code>spark.memory.fraction</code> sizes the unified execution/storage region (M) as a fraction of (JVM heap − 300MiB); the tuning guide frames it as a knob to fit M &quot;comfortably within the JVM&#39;s old or tenured generation&quot; rather than stating an enforced numeric range<sup class="footnote-ref"><a href="#fn6" id="fnref6:1">[6:1]</a></sup>. The cited sources document no failure threshold tied to a specific value like 0.9. What they do document is the risk of pushing the fraction high: the complement, <code>1 − spark.memory.fraction</code>, is untracked &quot;User Memory&quot; for UDFs, Python/Arrow glue, and native buffers, and starving it (which is what raising the fraction toward 0.9 does) leads to &quot;GC pressure or random OOMs,&quot; with no warning from Spark<sup class="footnote-ref"><a href="#fn7" id="fnref7:2">[7:2]</a></sup>.</p><p><code>spark.memory.storageFraction</code> isn&#39;t independent of that: it&#39;s a fraction <em>of</em> M, the same region <code>spark.memory.fraction</code> sizes<sup class="footnote-ref"><a href="#fn6" id="fnref6:2">[6:2]</a></sup>. Inside that shared pool, execution can evict storage down to the threshold set by <code>storageFraction</code>, but storage can never evict execution<sup class="footnote-ref"><a href="#fn11" id="fnref11">[11]</a></sup><sup class="footnote-ref"><a href="#fn6" id="fnref6:3">[6:3]</a></sup>. With the defaults (0.6 and 0.5), that works out to execution and storage each getting 30% of usable heap; raising <code>memory.fraction</code> scales both pools at once, while <code>storageFraction</code> only re-splits the pool that&#39;s already been carved out<sup class="footnote-ref"><a href="#fn7" id="fnref7:3">[7:3]</a></sup>.</p><blockquote><p><strong>PySpark:</strong> <code>spark.sql.execution.arrow.pyspark.enabled</code> is off by default and governs Arrow use for <code>DataFrame.toPandas()</code> and <code>SparkSession.createDataFrame()</code> from a Pandas DataFrame or NumPy array<sup class="footnote-ref"><a href="#fn8" id="fnref8:3">[8:3]</a></sup>. Its documented risk is a type-coverage gap, not silent data corruption: <code>ArrayType</code> of <code>TimestampType</code> is explicitly unsupported<sup class="footnote-ref"><a href="#fn4" id="fnref4:7">[4:7]</a></sup><sup class="footnote-ref"><a href="#fn5" id="fnref5:5">[5:5]</a></sup>, and more generally an unsupported column type raises an error rather than converting wrongly<sup class="footnote-ref"><a href="#fn8" id="fnref8:4">[8:4]</a></sup>. <code>spark.sql.execution.arrow.pyspark.fallback.enabled</code> defaults to <code>true</code> and automatically falls back to the non-Arrow path if a conversion error occurs, before any computation runs<sup class="footnote-ref"><a href="#fn8" id="fnref8:5">[8:5]</a></sup><sup class="footnote-ref"><a href="#fn5" id="fnref5:6">[5:6]</a></sup><sup class="footnote-ref"><a href="#fn4" id="fnref4:8">[4:8]</a></sup>. So the failure mode is an exception plus fallback, not a wrong result. Separately, and not specific to this Arrow config, PySpark&#39;s own type coercion has its own hazards: numeric values passed for <code>ByteType</code>/<code>ShortType</code>/<code>IntegerType</code> must fall within fixed ranges or get rejected or converted unexpectedly<sup class="footnote-ref"><a href="#fn2" id="fnref2:2">[2:2]</a></sup>.</p></blockquote><h3 id="config-shuffle-service" tabindex="-1">Shuffle service <a class="header-anchor" href="#config-shuffle-service" aria-label="Permalink to &quot;Shuffle service {#config-shuffle-service}&quot;">​</a></h3><p>During a shuffle, an executor writes its map output to local disk and then serves fetch requests for that data itself<sup class="footnote-ref"><a href="#fn12" id="fnref12">[12]</a></sup>. Dynamic allocation can reclaim an idle executor before a later stage has fetched all the shuffle blocks it holds (especially with stragglers, tasks that run much longer than their peers), and once that executor is gone, any stage that still needs its shuffle output gets a <code>FetchFailed</code> and has to recompute it<sup class="footnote-ref"><a href="#fn12" id="fnref12:1">[12:1]</a></sup>. The external shuffle service fixes this by moving shuffle-block serving out of the executor process: it&#39;s a long-running process on each node, independent of any particular application&#39;s executors, and once <code>spark.shuffle.service.enabled</code> is <code>true</code>, executors fetch shuffle blocks from it instead of from each other, so an executor&#39;s shuffle output keeps being served after that executor is reclaimed<sup class="footnote-ref"><a href="#fn12" id="fnref12:2">[12:2]</a></sup><sup class="footnote-ref"><a href="#fn4" id="fnref4:9">[4:9]</a></sup>. Setting up dynamic allocation requires enabling this service on every worker node in addition to <code>spark.dynamicAllocation.enabled</code><sup class="footnote-ref"><a href="#fn2" id="fnref2:3">[2:3]</a></sup>. As of Spark 3.0, shuffle-file tracking (<code>spark.dynamicAllocation.shuffleTracking.enabled</code>) is an alternative to the external shuffle service for the same safe-removal problem<sup class="footnote-ref"><a href="#fn4" id="fnref4:10">[4:10]</a></sup>.</p><p><strong>Limitations / false-positive risk:</strong> the service can be intentionally left off when dynamic allocation relies on shuffle-file tracking instead, so a disabled setting is not automatically a misconfiguration.</p><h3 id="config-autoscale-bounds" tabindex="-1">Autoscale bounds <a class="header-anchor" href="#config-autoscale-bounds" aria-label="Permalink to &quot;Autoscale bounds {#config-autoscale-bounds}&quot;">​</a></h3><p>By default, <code>spark.dynamicAllocation.minExecutors</code> is <code>0</code> and <code>spark.dynamicAllocation.maxExecutors</code> is <code>infinity</code><sup class="footnote-ref"><a href="#fn4" id="fnref4:11">[4:11]</a></sup>. Leaving <code>maxExecutors</code> unset therefore leaves dynamic allocation unbounded on the high end as far as Spark itself is concerned; in practice the ceiling ends up being whatever the cluster or scheduler enforces outside Spark<sup class="footnote-ref"><a href="#fn4" id="fnref4:12">[4:12]</a></sup>. <code>spark.dynamicAllocation.initialExecutors</code> defaults to <code>minExecutors</code>, unless <code>--num-executors</code> (or <code>spark.executor.instances</code>) sets a larger starting value<sup class="footnote-ref"><a href="#fn4" id="fnref4:13">[4:13]</a></sup>. Whatever executor count <code>executorAllocationRatio</code> computes to maximize parallelism, that target is still clamped by the <code>minExecutors</code>/<code>maxExecutors</code> floor and ceiling<sup class="footnote-ref"><a href="#fn4" id="fnref4:14">[4:14]</a></sup>.</p><p><strong>Limitations / false-positive risk:</strong> an unbounded <code>maxExecutors</code> is often deliberate on clusters where the scheduler enforces the real ceiling, so an unset high end is not always wrong.</p><h3 id="config-serializer" tabindex="-1">Serializer <a class="header-anchor" href="#config-serializer" aria-label="Permalink to &quot;Serializer {#config-serializer}&quot;">​</a></h3><p>The default <code>spark.serializer</code> is <code>org.apache.spark.serializer.JavaSerializer</code>, which works with any <code>Serializable</code> Java object but is &quot;quite slow&quot;; Spark&#39;s configuration reference recommends switching to <code>KryoSerializer</code> &quot;when speed is necessary&quot;<sup class="footnote-ref"><a href="#fn4" id="fnref4:15">[4:15]</a></sup>. The tuning guide is more specific: Kryo is &quot;significantly faster and more compact than Java serialization (often as much as 10x),&quot; though it doesn&#39;t support all <code>Serializable</code> types and needs its classes registered in advance for best performance<sup class="footnote-ref"><a href="#fn6" id="fnref6:4">[6:4]</a></sup>. That registration requirement is the stated reason Kryo isn&#39;t the default: &quot;The only reason Kryo is not the default is because of the custom registration requirement,&quot; and Spark recommends trying it for any network-intensive application<sup class="footnote-ref"><a href="#fn6" id="fnref6:5">[6:5]</a></sup>. Since Spark 2.0.0, Spark internally uses Kryo regardless of this setting when shuffling RDDs of simple types, arrays of simple types, or strings, and auto-registers common Scala classes via Twitter chill&#39;s <code>AllScalaRegistrar</code><sup class="footnote-ref"><a href="#fn6" id="fnref6:6">[6:6]</a></sup>. A practitioner summary puts Java serialization at &quot;2-10x slower and larger on the wire&quot; on every shuffle, with the caveat that classes left unregistered silently fall back to Java serialization unless registered via <code>spark.kryo.classesToRegister</code> or <code>spark.kryo.registrator</code><sup class="footnote-ref"><a href="#fn13" id="fnref13">[13]</a></sup>. Persisting RDDs in serialized form is another case where Kryo can be more space-efficient than Java serialization<sup class="footnote-ref"><a href="#fn10" id="fnref10:1">[10:1]</a></sup>.</p><p><strong>Limitations / false-positive risk:</strong> <code>JavaSerializer</code> may be kept on purpose when an application depends on <code>Serializable</code> types Kryo cannot handle, so a non-Kryo setting is not necessarily a mistake.</p><h3 id="config-memory-overhead" tabindex="-1">Memory overhead <a class="header-anchor" href="#config-memory-overhead" aria-label="Permalink to &quot;Memory overhead {#config-memory-overhead}&quot;">​</a></h3><p><code>spark.executor.memoryOverhead</code> covers everything an executor needs outside the JVM heap sized by <code>spark.executor.memory</code>: thread stacks, JIT buffers, metaspace, JNI, native libraries, and, if <code>spark.executor.pyspark.memory</code> isn&#39;t set separately, the memory used by PySpark&#39;s per-task Python worker processes<sup class="footnote-ref"><a href="#fn7" id="fnref7:4">[7:4]</a></sup>. Left unset, Spark defaults it to the greater of 10% of executor memory or 384MB<sup class="footnote-ref"><a href="#fn7" id="fnref7:5">[7:5]</a></sup>; the configuration reference formalizes that floor as <code>spark.executor.minMemoryOverhead</code> (default <code>384m</code>, since Spark 4.0) and the percentage as <code>spark.executor.memoryOverheadFactor</code> (default 0.10, or 0.40 for non-JVM jobs on Kubernetes, since those need more non-JVM heap space)<sup class="footnote-ref"><a href="#fn4" id="fnref4:16">[4:16]</a></sup>. The resource manager (YARN&#39;s NodeManager, or the Kubernetes kubelet) sizes the container/pod to the sum of <code>spark.executor.memoryOverhead</code>, <code>spark.executor.memory</code>, <code>spark.memory.offHeap.size</code>, and <code>spark.executor.pyspark.memory</code><sup class="footnote-ref"><a href="#fn4" id="fnref4:17">[4:17]</a></sup>. If off-heap memory or PySpark worker processes consume memory that isn&#39;t reflected in a correspondingly larger <code>memoryOverhead</code>, that extra usage still shows up in the process&#39;s actual RSS, which the resource manager does track. Once real usage exceeds the container&#39;s allocated size, the executor is killed by YARN or OOMKilled by the kubelet, with no Spark-level error, just an abrupt process death<sup class="footnote-ref"><a href="#fn7" id="fnref7:6">[7:6]</a></sup>.</p><p><strong>Limitations / false-positive risk:</strong> the default overhead is adequate for many JVM-only jobs, so a value left at the default only signals trouble on off-heap-heavy or PySpark workloads that need more non-heap room.</p><h2 id="jvm-gc-flags" tabindex="-1">JVM GC flags <a class="header-anchor" href="#jvm-gc-flags" aria-label="Permalink to &quot;JVM GC flags&quot;">​</a></h2><table tabindex="0"><thead><tr><th>Flag</th><th>Effect</th></tr></thead><tbody><tr><td><code>-XX:+UseG1GC</code></td><td>Default collector since Spark 4.0.0, which defaults to JDK 17<sup class="footnote-ref"><a href="#fn6" id="fnref6:7">[6:7]</a></sup>.</td></tr><tr><td><code>-XX:G1HeapRegionSize</code></td><td>May need raising alongside large executor heaps<sup class="footnote-ref"><a href="#fn6" id="fnref6:8">[6:8]</a></sup>.</td></tr><tr><td><code>-XX:InitiatingHeapOccupancyPercent</code></td><td>Tuned (with <code>-XX:ConcGCThreads</code> and RSet-update settings) to fix a documented ~100-second G1 full-GC pause on an 88GB executor heap<sup class="footnote-ref"><a href="#fn14" id="fnref14">[14]</a></sup>.</td></tr><tr><td><code>-XX:ConcGCThreads</code></td><td>See <code>-XX:InitiatingHeapOccupancyPercent</code> above<sup class="footnote-ref"><a href="#fn14" id="fnref14:1">[14:1]</a></sup>.</td></tr><tr><td><code>-XX:+UseZGC</code></td><td>Concurrent, low-latency collector; sub-millisecond pause target independent of heap size, from a few hundred megabytes up to 16TB<sup class="footnote-ref"><a href="#fn15" id="fnref15">[15]</a></sup>.</td></tr></tbody></table><h3 id="g1-vs-zgc" tabindex="-1">G1 vs. ZGC <a class="header-anchor" href="#g1-vs-zgc" aria-label="Permalink to &quot;G1 vs. ZGC&quot;">​</a></h3><p>G1 was designed as a CMS replacement aiming at both throughput and low latency: it partitions the heap into equal-sized regions and copies out only the live objects from collected regions rather than compacting the whole heap<sup class="footnote-ref"><a href="#fn14" id="fnref14:2">[14:2]</a></sup>. Even so, a documented 88GB-heap Spark benchmark hit &quot;unacceptable full GC,&quot; with one job pausing nearly 100 seconds under default G1 settings. Only after tuning <code>InitiatingHeapOccupancyPercent</code>, <code>ConcGCThreads</code>, and RSet-update parameters did G1 beat Parallel/CMS GC on both throughput and latency<sup class="footnote-ref"><a href="#fn14" id="fnref14:3">[14:3]</a></sup>. ZGC, per the OpenJDK project page, &quot;performs all expensive work concurrently, without stopping the execution of application threads for more than a millisecond,&quot; with pause times that stay flat as heap size scales from a few hundred megabytes up to 16TB<sup class="footnote-ref"><a href="#fn15" id="fnref15:1">[15:1]</a></sup>.</p><h2 id="sources" tabindex="-1">Sources <a class="header-anchor" href="#sources" aria-label="Permalink to &quot;Sources&quot;">​</a></h2><hr class="footnotes-sep"><section class="footnotes"><ol class="footnotes-list"><li id="fn1" class="footnote-item"><p><a href="https://spark.apache.org/docs/latest/sql-performance-tuning.html" target="_blank" rel="noreferrer">Performance Tuning — Spark SQL, DataFrames and Datasets Guide</a> <a href="#fnref1" class="footnote-backref">↩︎</a> <a href="#fnref1:1" class="footnote-backref">↩︎</a> <a href="#fnref1:2" class="footnote-backref">↩︎</a> <a href="#fnref1:3" class="footnote-backref">↩︎</a> <a href="#fnref1:4" class="footnote-backref">↩︎</a> <a href="#fnref1:5" class="footnote-backref">↩︎</a> <a href="#fnref1:6" class="footnote-backref">↩︎</a> <a href="#fnref1:7" class="footnote-backref">↩︎</a> <a href="#fnref1:8" class="footnote-backref">↩︎</a> <a href="#fnref1:9" class="footnote-backref">↩︎</a> <a href="#fnref1:10" class="footnote-backref">↩︎</a></p></li><li id="fn2" class="footnote-item"><p><em>Spark: The Definitive Guide</em>, Chambers &amp; Zaharia, ch. 4, ch. 10 <a href="#fnref2" class="footnote-backref">↩︎</a> <a href="#fnref2:1" class="footnote-backref">↩︎</a> <a href="#fnref2:2" class="footnote-backref">↩︎</a> <a href="#fnref2:3" class="footnote-backref">↩︎</a></p></li><li id="fn3" class="footnote-item"><p><em>Learning Spark, 2nd Edition</em>, Damji, Wenig, Das &amp; Lee, ch. 7 <a href="#fnref3" class="footnote-backref">↩︎</a> <a href="#fnref3:1" class="footnote-backref">↩︎</a> <a href="#fnref3:2" class="footnote-backref">↩︎</a></p></li><li id="fn4" class="footnote-item"><p><a href="https://spark.apache.org/docs/latest/configuration.html" target="_blank" rel="noreferrer">Configuration — Spark</a> <a href="#fnref4" class="footnote-backref">↩︎</a> <a href="#fnref4:1" class="footnote-backref">↩︎</a> <a href="#fnref4:2" class="footnote-backref">↩︎</a> <a href="#fnref4:3" class="footnote-backref">↩︎</a> <a href="#fnref4:4" class="footnote-backref">↩︎</a> <a href="#fnref4:5" class="footnote-backref">↩︎</a> <a href="#fnref4:6" class="footnote-backref">↩︎</a> <a href="#fnref4:7" class="footnote-backref">↩︎</a> <a href="#fnref4:8" class="footnote-backref">↩︎</a> <a href="#fnref4:9" class="footnote-backref">↩︎</a> <a href="#fnref4:10" class="footnote-backref">↩︎</a> <a href="#fnref4:11" class="footnote-backref">↩︎</a> <a href="#fnref4:12" class="footnote-backref">↩︎</a> <a href="#fnref4:13" class="footnote-backref">↩︎</a> <a href="#fnref4:14" class="footnote-backref">↩︎</a> <a href="#fnref4:15" class="footnote-backref">↩︎</a> <a href="#fnref4:16" class="footnote-backref">↩︎</a> <a href="#fnref4:17" class="footnote-backref">↩︎</a></p></li><li id="fn5" class="footnote-item"><p><a href="https://raw.githubusercontent.com/apache/spark/v3.5.0/sql/catalyst/src/main/scala/org/apache/spark/sql/internal/SQLConf.scala" target="_blank" rel="noreferrer">SQLConf.scala (Spark 3.5.0)</a> <a href="#fnref5" class="footnote-backref">↩︎</a> <a href="#fnref5:1" class="footnote-backref">↩︎</a> <a href="#fnref5:2" class="footnote-backref">↩︎</a> <a href="#fnref5:3" class="footnote-backref">↩︎</a> <a href="#fnref5:4" class="footnote-backref">↩︎</a> <a href="#fnref5:5" class="footnote-backref">↩︎</a> <a href="#fnref5:6" class="footnote-backref">↩︎</a></p></li><li id="fn6" class="footnote-item"><p><a href="https://spark.apache.org/docs/latest/tuning.html" target="_blank" rel="noreferrer">Tuning Spark</a> <a href="#fnref6" class="footnote-backref">↩︎</a> <a href="#fnref6:1" class="footnote-backref">↩︎</a> <a href="#fnref6:2" class="footnote-backref">↩︎</a> <a href="#fnref6:3" class="footnote-backref">↩︎</a> <a href="#fnref6:4" class="footnote-backref">↩︎</a> <a href="#fnref6:5" class="footnote-backref">↩︎</a> <a href="#fnref6:6" class="footnote-backref">↩︎</a> <a href="#fnref6:7" class="footnote-backref">↩︎</a> <a href="#fnref6:8" class="footnote-backref">↩︎</a></p></li><li id="fn7" class="footnote-item"><p><a href="https://luminousmen.com/post/dive-into-spark-memory" target="_blank" rel="noreferrer">Dive Into Spark Memory Management</a> <a href="#fnref7" class="footnote-backref">↩︎</a> <a href="#fnref7:1" class="footnote-backref">↩︎</a> <a href="#fnref7:2" class="footnote-backref">↩︎</a> <a href="#fnref7:3" class="footnote-backref">↩︎</a> <a href="#fnref7:4" class="footnote-backref">↩︎</a> <a href="#fnref7:5" class="footnote-backref">↩︎</a> <a href="#fnref7:6" class="footnote-backref">↩︎</a></p></li><li id="fn8" class="footnote-item"><p><a href="https://spark.apache.org/docs/3.5.8/api/python/user_guide/sql/arrow_pandas.html" target="_blank" rel="noreferrer">Apache Arrow in PySpark</a> <a href="#fnref8" class="footnote-backref">↩︎</a> <a href="#fnref8:1" class="footnote-backref">↩︎</a> <a href="#fnref8:2" class="footnote-backref">↩︎</a> <a href="#fnref8:3" class="footnote-backref">↩︎</a> <a href="#fnref8:4" class="footnote-backref">↩︎</a> <a href="#fnref8:5" class="footnote-backref">↩︎</a></p></li><li id="fn9" class="footnote-item"><p><a href="https://www.databricks.com/blog/2020/05/29/adaptive-query-execution-speeding-up-spark-sql-at-runtime.html" target="_blank" rel="noreferrer">Adaptive Query Execution: Speeding Up Spark SQL at Runtime</a> <a href="#fnref9" class="footnote-backref">↩︎</a></p></li><li id="fn10" class="footnote-item"><p><em>High Performance Spark, 2nd Edition</em>, Karau, Polak &amp; Warren, ch. 5 <a href="#fnref10" class="footnote-backref">↩︎</a> <a href="#fnref10:1" class="footnote-backref">↩︎</a></p></li><li id="fn11" class="footnote-item"><p><a href="https://raw.githubusercontent.com/spoddutur/spark-notes/master/task_memory_management_in_spark.md" target="_blank" rel="noreferrer">task_memory_management_in_spark.md</a> <a href="#fnref11" class="footnote-backref">↩︎</a></p></li><li id="fn12" class="footnote-item"><p><a href="https://spark.apache.org/docs/latest/job-scheduling.html" target="_blank" rel="noreferrer">Job Scheduling — Spark</a> <a href="#fnref12" class="footnote-backref">↩︎</a> <a href="#fnref12:1" class="footnote-backref">↩︎</a> <a href="#fnref12:2" class="footnote-backref">↩︎</a></p></li><li id="fn13" class="footnote-item"><p><a href="https://luminousmen.com/post/the-apache-spark-optimization-checklist" target="_blank" rel="noreferrer">The Apache Spark Optimization Checklist</a> <a href="#fnref13" class="footnote-backref">↩︎</a></p></li><li id="fn14" class="footnote-item"><p><a href="https://www.databricks.com/blog/2015/05/28/tuning-java-garbage-collection-for-spark-applications.html" target="_blank" rel="noreferrer">Tuning Java Garbage Collection for Spark Applications</a> <a href="#fnref14" class="footnote-backref">↩︎</a> <a href="#fnref14:1" class="footnote-backref">↩︎</a> <a href="#fnref14:2" class="footnote-backref">↩︎</a> <a href="#fnref14:3" class="footnote-backref">↩︎</a></p></li><li id="fn15" class="footnote-item"><p><a href="https://wiki.openjdk.org/display/zgc" target="_blank" rel="noreferrer">ZGC — The Z Garbage Collector</a> <a href="#fnref15" class="footnote-backref">↩︎</a> <a href="#fnref15:1" class="footnote-backref">↩︎</a></p></li></ol></section>',29)])])}const h=a(s,[["render",f]]);export{u as __pageData,h as default};
@@ -0,0 +1 @@
1
+ import{_ as a,o,c as t,a5 as r}from"./chunks/framework.DSg0KOwT.js";const u=JSON.parse('{"title":"Spark Config Quick-Reference","description":"","frontmatter":{"title":"Spark Config Quick-Reference"},"headers":[],"relativePath":"tuning-reference/config.md","filePath":"tuning-reference/config.md"}'),s={name:"tuning-reference/config.md"};function f(n,e,c,i,d,l){return o(),t("div",null,[...e[0]||(e[0]=[r("",29)])])}const h=a(s,[["render",f]]);export{u as __pageData,h as default};
@@ -0,0 +1 @@
1
+ import{_ as a,o as t,c as o,a5 as r}from"./chunks/framework.DSg0KOwT.js";const s="../assets/columnar-layout.PghGeOEA.svg",n="../assets/columnar-layout.dark.BVNlz0ff.svg",m=JSON.parse('{"title":"Data Formats","description":"","frontmatter":{"title":"Data Formats"},"headers":[],"relativePath":"tuning-reference/data-formats.md","filePath":"tuning-reference/data-formats.md"}'),f={name:"tuning-reference/data-formats.md"};function i(l,e,c,d,p,h){return t(),o("div",null,[...e[0]||(e[0]=[r('<h1 id="data-formats" tabindex="-1">Data Formats <a class="header-anchor" href="#data-formats" aria-label="Permalink to &quot;Data Formats {#data-formats}&quot;">​</a></h1><h2 id="how-parquet-and-orc-lay-out-data" tabindex="-1">How Parquet and ORC lay out data <a class="header-anchor" href="#how-parquet-and-orc-lay-out-data" aria-label="Permalink to &quot;How Parquet and ORC lay out data&quot;">​</a></h2><p>Parquet and ORC are both self-describing, columnar file formats for storing Spark&#39;s structured data<sup class="footnote-ref"><a href="#fn1" id="fnref1">[1]</a></sup>. They&#39;re closer than they are different: <em>Spark: The Definitive Guide</em> frames the choice as &quot;for the most part, they&#39;re quite similar; the fundamental difference is that Parquet is further optimized for use with Spark, whereas ORC is further optimized for Hive&quot;<sup class="footnote-ref"><a href="#fn1" id="fnref1:1">[1:1]</a></sup>, and notes that ORC &quot;has no options for reading in data because Spark understands the file format quite well&quot;<sup class="footnote-ref"><a href="#fn1" id="fnref1:2">[1:2]</a></sup>: Spark treats it as a format it understands natively rather than one with extra read-side knobs.</p><p>Both formats group rows into chunks (row groups in Parquet, stripes in ORC), each holding every column&#39;s data for that slice, with a footer or file-level metadata section recording where each chunk lives on disk. Parquet&#39;s <code>FileMetaData</code> footer records the offset and size of every row group and column chunk, and &quot;readers are expected to first read the file metadata to find all the column chunks they are interested in. The column chunks should then be read sequentially&quot;<sup class="footnote-ref"><a href="#fn2" id="fnref2">[2]</a></sup>. Because compression is applied per column chunk rather than as one continuous stream over the whole file, a reader can seek straight to a row group&#39;s footer-recorded offset and decode just that chunk, which is exactly why Parquet files compressed with Snappy stay splittable even though the Snappy stream format itself provides no split points: it has &quot;no entropy encoder backend nor framing layer -- the latter is assumed to be handled by other parts of the system&quot;<sup class="footnote-ref"><a href="#fn3" id="fnref3">[3]</a></sup>.</p><img class="light-only" src="'+s+'" alt="A columnar file&#39;s footer holds row-group offsets and per-column min/max statistics that let a reader skip row groups whose stats exclude the predicate."><img class="dark-only" src="'+n+'" alt="A columnar file&#39;s footer holds row-group offsets and per-column min/max statistics that let a reader skip row groups whose stats exclude the predicate."><p>Both formats also carry the statistics that later drive predicate pushdown. Parquet records min/max values per column chunk, plus an optional page-level <code>ColumnIndex</code> that lets a reader binary-search ordered columns for matching pages<sup class="footnote-ref"><a href="#fn4" id="fnref4">[4]</a></sup>, and optional Bloom filters for columns whose cardinality is too high for a dictionary to be practical<sup class="footnote-ref"><a href="#fn5" id="fnref5">[5]</a></sup>. ORC&#39;s <code>ColumnStatistics</code> protobuf records row count and, for most primitive types, min/max (plus sum for numeric types); from Hive 1.1.0 it also records a <code>hasNull</code> flag used specifically by ORC&#39;s predicate pushdown to answer <code>IS NULL</code> queries<sup class="footnote-ref"><a href="#fn6" id="fnref6">[6]</a></sup>. ORC layers this at two granularities: a <code>RowIndexEntry</code> per row group (10,000 rows by default), kept at the front of each stripe so it&#39;s read only when pushdown or seeking is actually needed<sup class="footnote-ref"><a href="#fn7" id="fnref7">[7]</a></sup>, and file-level <code>StripeStatistics</code> that let whole stripes be skipped by predicate pushdown<sup class="footnote-ref"><a href="#fn8" id="fnref8">[8]</a></sup>. ORC also supports Bloom filters (Hive 1.2.0+), but only evaluates them against row groups that already passed the min/max row-index check first: a second-stage filter, not an independent one<sup class="footnote-ref"><a href="#fn7" id="fnref7:1">[7:1]</a></sup>.</p><p>On compression: Parquet defaults to Snappy, while ORC&#39;s default has been Zstd since Spark 2.3.0<sup class="footnote-ref"><a href="#fn9" id="fnref9">[9]</a></sup>. Zstandard&#39;s own manual describes it as &quot;a fast lossless compression algorithm, targeting real-time compression scenarios at zlib-level and better compression ratios,&quot; offering regular levels 1–22 plus negative levels that trade ratio for speed<sup class="footnote-ref"><a href="#fn10" id="fnref10">[10]</a></sup>. Spark exposes the level directly through <code>spark.io.compression.zstd.level</code> (default <code>1</code>)<sup class="footnote-ref"><a href="#fn9" id="fnref9:1">[9:1]</a></sup>, along with <code>spark.io.compression.zstd.workers</code> for parallel compression threads (default <code>0</code>, since 4.0.0) and <code>spark.io.compression.zstd.bufferSize</code> (default 32k)<sup class="footnote-ref"><a href="#fn9" id="fnref9:2">[9:2]</a></sup>.</p><p>Parquet has one more structural knob: <code>parquet.writer.version</code>. Version 2 changes the on-disk page-header encoding, and the spec is explicit that this is a forward-<em>incompatible</em> change: &quot;a reader that only understands <code>DataPageHeader</code> cannot parse <code>DataPageHeaderV2</code> pages&quot;<sup class="footnote-ref"><a href="#fn11" id="fnref11">[11]</a></sup>. The spec also flags that the file&#39;s own version marker &quot;has historically been used inconsistently: writers populate 1 or 2 without a consistent relationship to the features actually used&quot;<sup class="footnote-ref"><a href="#fn11" id="fnref11:1">[11:1]</a></sup>, so the marker itself isn&#39;t a reliable way to tell which page format a file actually contains.</p><h2 id="what-shows-up-at-read-time" tabindex="-1">What shows up at read time <a class="header-anchor" href="#what-shows-up-at-read-time" aria-label="Permalink to &quot;What shows up at read time&quot;">​</a></h2><p>A file&#39;s layout shapes the job the moment Spark reads it. Spark&#39;s file readers derive partition counts from file metadata rather than from a fixed rule: &quot;for structured formats like Parquet, ORC, or Avro, Spark actually reads the metadata (footers, row groups, that kind of thing) and tries to slice the file in a way that makes sense. Often you&#39;ll see one partition per row group, though Spark may merge or split depending on file sizes and configs&quot;<sup class="footnote-ref"><a href="#fn12" id="fnref12">[12]</a></sup>. A concrete case: a single 30 GB Parquet file with 300 row groups gets split into exactly 300 partitions<sup class="footnote-ref"><a href="#fn12" id="fnref12:1">[12:1]</a></sup>. So a surprising partition count for a given file is usually explained by its row-group layout, separately from <code>spark.sql.files.maxPartitionBytes</code> (128 MB by default) and its <code>maxPartitionNum</code>/<code>minPartitionNum</code> companions, which govern splitting for file-based sources generally<sup class="footnote-ref"><a href="#fn13" id="fnref13">[13]</a></sup>.</p><p>The <a href="./bottleneck-small-files.html">small-files problem</a> shows up as metadata overhead rather than raw I/O cost: &quot;when you&#39;re writing lots of small files, there&#39;s a significant metadata overhead that you incur managing all of those files. Spark especially does not do well with small files&quot;<sup class="footnote-ref"><a href="#fn1" id="fnref1:3">[1:3]</a></sup>. At the other extreme, oversized or misaligned row groups show up as lost locality: the Parquet spec&#39;s rationale for matching block size to row-group size is that &quot;since an entire row group might need to be read, we want it to completely fit on one HDFS block&quot;<sup class="footnote-ref"><a href="#fn14" id="fnref14">[14]</a></sup>. A row group straddling a block boundary can&#39;t get that benefit.</p><p>Raw (non-Parquet) gzip or zip source files reveal themselves as a single-executor bottleneck at read time, visible as one task doing dramatically more work than the rest of its stage: Spark &quot;needs to download the whole file on one executor, unpack it on just one core, and then redistribute the partitions to the cluster nodes&quot;<sup class="footnote-ref"><a href="#fn15" id="fnref15">[15]</a></sup>.</p><p>Version and schema mismatches surface as read failures rather than silent corruption. A file written with <code>parquet.writer.version=2</code> can&#39;t be parsed by an older Spark version, or a non-Spark engine such as Hive or Impala, if that reader only implements the original <code>DataPageHeader</code><sup class="footnote-ref"><a href="#fn11" id="fnref11:2">[11:2]</a></sup>. <em>Spark: The Definitive Guide</em> frames this as a general risk: &quot;you can still encounter problems if you&#39;re working with incompatible Parquet files. Be careful when you write out Parquet files with different versions of Spark (especially older ones) because this can cause significant headache&quot;<sup class="footnote-ref"><a href="#fn1" id="fnref1:4">[1:4]</a></sup>.</p><h2 id="what-the-layout-buys-you" tabindex="-1">What the layout buys you <a class="header-anchor" href="#what-the-layout-buys-you" aria-label="Permalink to &quot;What the layout buys you&quot;">​</a></h2><p>Those read-time symptoms trace back to specific tradeoffs in the layout. Splittability is what decides whether a file can be processed in parallel at all. A non-splittable raw compressed source file forces Spark onto a single core for the whole download-and-unpack step before it can redistribute anything: &quot;as you can imagine, this becomes a huge bottleneck in your distributed processing&quot;<sup class="footnote-ref"><a href="#fn15" id="fnref15:1">[15:1]</a></sup>. Parquet sidesteps this at the container level: because its footer records row-group and column-chunk offsets independently of whichever codec compressed the bytes inside them, a task can seek and decode a row group on its own, so even a non-splittable codec like Snappy doesn&#39;t cost the file its parallelism<sup class="footnote-ref"><a href="#fn2" id="fnref2:1">[2:1]</a></sup><sup class="footnote-ref"><a href="#fn3" id="fnref3:1">[3:1]</a></sup>.</p><p>Row-group and block-size alignment governs the same kind of locality at a coarser grain. The spec recommends large row groups, 512 MB–1 GB, &quot;because larger row groups allow for larger column chunks which makes it possible to do larger sequential IO,&quot; at the cost of more write-side buffering, paired with a block size sized the same way; its own worked example is &quot;1GB row groups, 1GB HDFS block size, 1 HDFS block per HDFS file&quot;<sup class="footnote-ref"><a href="#fn14" id="fnref14:1">[14:1]</a></sup>.</p><p>The statistics both formats carry are what predicate pushdown runs against. Spark plans queries directly off &quot;statistics that Spark reads directly from the underlying data source, like the counts and min/max values in the metadata of Parquet files&quot;<sup class="footnote-ref"><a href="#fn16" id="fnref16">[16]</a></sup>, gated by <code>spark.sql.parquet.filterPushdown</code> (default <code>true</code> since Spark 1.2.0) and, for ORC, <code>spark.sql.orc.filterPushdown</code> (default <code>true</code> since 1.4.0)<sup class="footnote-ref"><a href="#fn17" id="fnref17">[17]</a></sup><sup class="footnote-ref"><a href="#fn18" id="fnref18">[18]</a></sup>. A more aggressive option, <code>spark.sql.parquet.aggregatePushdown</code> (default <code>false</code>, since 3.3.0), pushes <code>MIN</code>, <code>MAX</code>, and <code>COUNT</code> down to Parquet&#39;s footer statistics directly, and throws if the needed statistic is missing from a file&#39;s footer<sup class="footnote-ref"><a href="#fn17" id="fnref17:1">[17:1]</a></sup>. Skipping a row group or stripe this way means its bytes are never read off disk, which beats any amount of post-read filtering.</p><p>Small files cost more in metadata management than their data volume would suggest, and Spark &quot;especially does not do well&quot; with them<sup class="footnote-ref"><a href="#fn1" id="fnref1:5">[1:5]</a></sup>. On the scheduling side, Spark&#39;s tuning guide notes it &quot;can efficiently support tasks as short as 200 ms&quot; because it reuses one executor JVM across many tasks and has low task-launch cost, while separately recommending &quot;2-3 tasks per CPU core in your cluster&quot;<sup class="footnote-ref"><a href="#fn19" id="fnref19">[19]</a></sup>, which implies <a href="./bottleneck-tiny-tasks.html">scheduling overhead</a> only becomes proportionally significant once tasks (and the files behind them) shrink below roughly that 200 ms floor.</p><p>Compression choice trades CPU against I/O and storage. Once I/O stops being the bottleneck, paying a codec&#39;s decompression cost is a net loss: &quot;uncompressed files are clearly outperforming compressed files. This is because uncompressed files are I/O bound, and compressed files are CPU bound, but I/O is good enough here&quot;<sup class="footnote-ref"><a href="#fn15" id="fnref15:2">[15:2]</a></sup>. Bzip2 illustrates the opposite failure mode: it&#39;s splittable, but compresses so aggressively that &quot;you get very few partitions and therefore they can be poorly distributed&quot;<sup class="footnote-ref"><a href="#fn15" id="fnref15:3">[15:3]</a></sup>.</p><p>Writer-version and schema changes matter because they fail at read time, for whoever reads the data next (not at write time, for whoever wrote it). A <code>parquet.writer.version=2</code> file is simply unreadable by any engine that only understands the original page header<sup class="footnote-ref"><a href="#fn11" id="fnref11:3">[11:3]</a></sup>. Schema drift has its own correctness angle: for <a href="./table-formats.html">Delta Lake</a> tables, adding a column via <code>mergeSchema</code> causes existing rows, when read back, to have that new column&#39;s value read as <code>NULL</code><sup class="footnote-ref"><a href="#fn20" id="fnref20">[20]</a></sup>, a defined outcome, but one that changes what a downstream query sees for rows written before the schema changed.</p><h2 id="choosing-and-tuning-the-format" tabindex="-1">Choosing and tuning the format <a class="header-anchor" href="#choosing-and-tuning-the-format" aria-label="Permalink to &quot;Choosing and tuning the format&quot;">​</a></h2><p>A few defaults cover most cases despite those tradeoffs. Default to Parquet. Reach for ORC specifically when the same data also has to serve Hive consumers or existing Hive ORC tables, where ORC is the better-optimized target<sup class="footnote-ref"><a href="#fn1" id="fnref1:6">[1:6]</a></sup>.</p><p>Size row groups and the underlying filesystem block together, not independently. The Parquet spec&#39;s own recommended setup is 1 GB row groups paired with a 1 GB HDFS block size, one block per file<sup class="footnote-ref"><a href="#fn14" id="fnref14:2">[14:2]</a></sup>. Leave <code>spark.sql.files.maxPartitionBytes</code> (128 MB default) and its <code>maxPartitionNum</code>/<code>minPartitionNum</code> companions for the general file-splitting case, but expect row-group boundaries (not these settings) to be what actually decides partition count for row-group-oriented formats<sup class="footnote-ref"><a href="#fn13" id="fnref13:1">[13:1]</a></sup><sup class="footnote-ref"><a href="#fn12" id="fnref12:2">[12:2]</a></sup>.</p><p>Leave predicate pushdown on: <code>spark.sql.parquet.filterPushdown</code> and <code>spark.sql.orc.filterPushdown</code> both default to <code>true</code> already<sup class="footnote-ref"><a href="#fn17" id="fnref17:2">[17:2]</a></sup><sup class="footnote-ref"><a href="#fn18" id="fnref18:1">[18:1]</a></sup>, so the main action is not disabling them, plus considering <code>spark.sql.parquet.aggregatePushdown</code> when a workload is dominated by <code>MIN</code>/<code>MAX</code>/<code>COUNT</code> over Parquet sources with complete footer statistics<sup class="footnote-ref"><a href="#fn17" id="fnref17:3">[17:3]</a></sup>.</p><p>Cap output file size directly instead of letting the small-files problem accumulate: <code>maxRecordsPerFile</code>, introduced in Spark 2.2, targets an optimum file size by capping the number of records written per file, e.g. <code>df.write.option(&quot;maxRecordsPerFile&quot;, 5000)</code><sup class="footnote-ref"><a href="#fn1" id="fnref1:7">[1:7]</a></sup>.</p><p>Pick a compression codec based on the actual bottleneck. Parquet&#39;s default, Snappy, favors speed; ORC has already defaulted to Zstd since Spark 2.3.0<sup class="footnote-ref"><a href="#fn9" id="fnref9:3">[9:3]</a></sup>, and at least one optimization checklist recommends overriding Parquet&#39;s default to Zstd as well<sup class="footnote-ref"><a href="#fn21" id="fnref21">[21]</a></sup>. Zstd&#39;s level is tunable through <code>spark.io.compression.zstd.level</code>: higher levels buy better compression &quot;at the expense of more CPU and memory&quot;<sup class="footnote-ref"><a href="#fn9" id="fnref9:4">[9:4]</a></sup>. Avoid feeding Spark large raw <code>.gz</code>/<code>.zip</code> source files directly (unpack them before loading), since the non-splittability lives in the raw container format rather than in gzip itself; the same <em>Definitive Guide</em> that warns about raw gzip files elsewhere still recommends Parquet with gzip compression once gzip is wrapped inside Parquet&#39;s row-group container<sup class="footnote-ref"><a href="#fn1" id="fnref1:8">[1:8]</a></sup>.</p><p>Treat <code>spark.sql.parquet.mergeSchema</code> (default <code>false</code>) as opt-in rather than default-on: enabling it &quot;merges schemas collected from all data files,&quot; instead of trusting a single summary file or a random file&#39;s schema<sup class="footnote-ref"><a href="#fn22" id="fnref22">[22]</a></sup>, useful for evolving schemas, but it means every file in the dataset gets scanned for its schema. For Delta tables specifically, remember that <code>mergeSchema</code>-added columns read back as <code>NULL</code> on pre-existing rows<sup class="footnote-ref"><a href="#fn20" id="fnref20:1">[20:1]</a></sup> before relying on it for a backfill.</p><p>Don&#39;t flip <code>parquet.writer.version</code> to <code>2</code> without confirming every downstream reader of that data (an older Spark version, Hive, Impala, or anything else touching the files) actually supports <code>DataPageHeaderV2</code> first; it&#39;s a forward-incompatible page-format change, not an additive one<sup class="footnote-ref"><a href="#fn11" id="fnref11:4">[11:4]</a></sup><sup class="footnote-ref"><a href="#fn1" id="fnref1:9">[1:9]</a></sup>.</p><h2 id="sources" tabindex="-1">Sources <a class="header-anchor" href="#sources" aria-label="Permalink to &quot;Sources&quot;">​</a></h2><hr class="footnotes-sep"><section class="footnotes"><ol class="footnotes-list"><li id="fn1" class="footnote-item"><p><em>Spark: The Definitive Guide</em>, Chambers &amp; Zaharia, ch. 9 <a href="#fnref1" class="footnote-backref">↩︎</a> <a href="#fnref1:1" class="footnote-backref">↩︎</a> <a href="#fnref1:2" class="footnote-backref">↩︎</a> <a href="#fnref1:3" class="footnote-backref">↩︎</a> <a href="#fnref1:4" class="footnote-backref">↩︎</a> <a href="#fnref1:5" class="footnote-backref">↩︎</a> <a href="#fnref1:6" class="footnote-backref">↩︎</a> <a href="#fnref1:7" class="footnote-backref">↩︎</a> <a href="#fnref1:8" class="footnote-backref">↩︎</a> <a href="#fnref1:9" class="footnote-backref">↩︎</a></p></li><li id="fn2" class="footnote-item"><p><a href="https://parquet.apache.org/_print/docs/file-format/" target="_blank" rel="noreferrer">Parquet File Format: File Metadata and Row Groups</a> <a href="#fnref2" class="footnote-backref">↩︎</a> <a href="#fnref2:1" class="footnote-backref">↩︎</a></p></li><li id="fn3" class="footnote-item"><p><a href="https://github.com/google/snappy/blob/main/format_description.txt" target="_blank" rel="noreferrer">Snappy Compressed Format Description</a> <a href="#fnref3" class="footnote-backref">↩︎</a> <a href="#fnref3:1" class="footnote-backref">↩︎</a></p></li><li id="fn4" class="footnote-item"><p><a href="https://parquet.apache.org/_print/docs/file-format/" target="_blank" rel="noreferrer">Parquet File Format: Column Index</a> <a href="#fnref4" class="footnote-backref">↩︎</a></p></li><li id="fn5" class="footnote-item"><p><a href="https://parquet.apache.org/docs/file-format/bloomfilter/" target="_blank" rel="noreferrer">Parquet File Format: Bloom Filter</a> <a href="#fnref5" class="footnote-backref">↩︎</a></p></li><li id="fn6" class="footnote-item"><p><a href="https://orc.apache.org/specification/ORCv1/" target="_blank" rel="noreferrer">ORC Specification v1: Column Statistics</a> <a href="#fnref6" class="footnote-backref">↩︎</a></p></li><li id="fn7" class="footnote-item"><p><a href="https://orc.apache.org/specification/ORCv1/" target="_blank" rel="noreferrer">ORC Specification v1: Row Index and Bloom Filters</a> <a href="#fnref7" class="footnote-backref">↩︎</a> <a href="#fnref7:1" class="footnote-backref">↩︎</a></p></li><li id="fn8" class="footnote-item"><p><a href="https://orc.apache.org/specification/ORCv1/" target="_blank" rel="noreferrer">ORC Specification v1: Stripe Statistics</a> <a href="#fnref8" class="footnote-backref">↩︎</a></p></li><li id="fn9" class="footnote-item"><p><a href="https://spark.apache.org/docs/latest/configuration.html" target="_blank" rel="noreferrer">Configuration: Spark</a> <a href="#fnref9" class="footnote-backref">↩︎</a> <a href="#fnref9:1" class="footnote-backref">↩︎</a> <a href="#fnref9:2" class="footnote-backref">↩︎</a> <a href="#fnref9:3" class="footnote-backref">↩︎</a> <a href="#fnref9:4" class="footnote-backref">↩︎</a></p></li><li id="fn10" class="footnote-item"><p><a href="https://facebook.github.io/zstd/zstd_manual.html" target="_blank" rel="noreferrer">Zstandard Manual</a> <a href="#fnref10" class="footnote-backref">↩︎</a></p></li><li id="fn11" class="footnote-item"><p><a href="https://parquet.apache.org/_print/docs/file-format/" target="_blank" rel="noreferrer">Parquet File Format: Data Pages (V1/V2)</a> <a href="#fnref11" class="footnote-backref">↩︎</a> <a href="#fnref11:1" class="footnote-backref">↩︎</a> <a href="#fnref11:2" class="footnote-backref">↩︎</a> <a href="#fnref11:3" class="footnote-backref">↩︎</a> <a href="#fnref11:4" class="footnote-backref">↩︎</a></p></li><li id="fn12" class="footnote-item"><p><a href="https://luminousmen.com/post/spark-partitions" target="_blank" rel="noreferrer">How Spark Determines Partitions for a File</a> <a href="#fnref12" class="footnote-backref">↩︎</a> <a href="#fnref12:1" class="footnote-backref">↩︎</a> <a href="#fnref12:2" class="footnote-backref">↩︎</a></p></li><li id="fn13" class="footnote-item"><p><a href="https://spark.apache.org/docs/latest/sql-performance-tuning.html" target="_blank" rel="noreferrer">Performance Tuning: Spark SQL, DataFrames and Datasets Guide</a> <a href="#fnref13" class="footnote-backref">↩︎</a> <a href="#fnref13:1" class="footnote-backref">↩︎</a></p></li><li id="fn14" class="footnote-item"><p><a href="https://parquet.apache.org/_print/docs/file-format/" target="_blank" rel="noreferrer">Parquet File Format: Row Group Size</a> <a href="#fnref14" class="footnote-backref">↩︎</a> <a href="#fnref14:1" class="footnote-backref">↩︎</a> <a href="#fnref14:2" class="footnote-backref">↩︎</a></p></li><li id="fn15" class="footnote-item"><p><a href="https://luminousmen.com/post/spark-tips-dont-collect-data-on-driver" target="_blank" rel="noreferrer">Spark Tips: Don&#39;t Collect Data on Driver</a> <a href="#fnref15" class="footnote-backref">↩︎</a> <a href="#fnref15:1" class="footnote-backref">↩︎</a> <a href="#fnref15:2" class="footnote-backref">↩︎</a> <a href="#fnref15:3" class="footnote-backref">↩︎</a></p></li><li id="fn16" class="footnote-item"><p><a href="https://spark.apache.org/docs/latest/sql-performance-tuning.html" target="_blank" rel="noreferrer">Performance Tuning: Spark SQL, DataFrames and Datasets Guide</a> <a href="#fnref16" class="footnote-backref">↩︎</a></p></li><li id="fn17" class="footnote-item"><p><a href="https://spark.apache.org/docs/latest/configuration.html" target="_blank" rel="noreferrer">Configuration: Spark</a> <a href="#fnref17" class="footnote-backref">↩︎</a> <a href="#fnref17:1" class="footnote-backref">↩︎</a> <a href="#fnref17:2" class="footnote-backref">↩︎</a> <a href="#fnref17:3" class="footnote-backref">↩︎</a></p></li><li id="fn18" class="footnote-item"><p><a href="https://spark.apache.org/docs/latest/configuration.html" target="_blank" rel="noreferrer">Configuration: Spark</a> <a href="#fnref18" class="footnote-backref">↩︎</a> <a href="#fnref18:1" class="footnote-backref">↩︎</a></p></li><li id="fn19" class="footnote-item"><p><a href="https://spark.apache.org/docs/latest/tuning.html" target="_blank" rel="noreferrer">Spark Tuning Guide: Level of Parallelism</a> <a href="#fnref19" class="footnote-backref">↩︎</a></p></li><li id="fn20" class="footnote-item"><p><em>Learning Spark, 2nd Edition</em>, Damji, Wenig, Das &amp; Lee, ch. 9 <a href="#fnref20" class="footnote-backref">↩︎</a> <a href="#fnref20:1" class="footnote-backref">↩︎</a></p></li><li id="fn21" class="footnote-item"><p><a href="https://luminousmen.com/post/the-apache-spark-optimization-checklist" target="_blank" rel="noreferrer">The Apache Spark Optimization Checklist</a> <a href="#fnref21" class="footnote-backref">↩︎</a></p></li><li id="fn22" class="footnote-item"><p><a href="https://raw.githubusercontent.com/apache/spark/v3.5.0/sql/catalyst/src/main/scala/org/apache/spark/sql/internal/SQLConf.scala" target="_blank" rel="noreferrer">SQLConf.scala</a> <a href="#fnref22" class="footnote-backref">↩︎</a></p></li></ol></section>',32)])])}const g=a(f,[["render",i]]);export{m as __pageData,g as default};
@@ -0,0 +1 @@
1
+ import{_ as a,o as t,c as o,a5 as r}from"./chunks/framework.DSg0KOwT.js";const s="../assets/columnar-layout.PghGeOEA.svg",n="../assets/columnar-layout.dark.BVNlz0ff.svg",m=JSON.parse('{"title":"Data Formats","description":"","frontmatter":{"title":"Data Formats"},"headers":[],"relativePath":"tuning-reference/data-formats.md","filePath":"tuning-reference/data-formats.md"}'),f={name:"tuning-reference/data-formats.md"};function i(l,e,c,d,p,h){return t(),o("div",null,[...e[0]||(e[0]=[r("",32)])])}const g=a(f,[["render",i]]);export{m as __pageData,g as default};
@@ -0,0 +1 @@
1
+ import{_ as n,C as t,o as a,c as r,E as i}from"./chunks/framework.DSg0KOwT.js";const g=JSON.parse('{"title":"Spark Tuning Reference","description":"","frontmatter":{"layout":"page","title":"Spark Tuning Reference","sidebar":false},"headers":[],"relativePath":"tuning-reference/index.md","filePath":"tuning-reference/index.md"}'),o={name:"tuning-reference/index.md"};function c(s,d,p,f,l,u){const e=t("TuningLanding");return a(),r("div",null,[i(e)])}const m=n(o,[["render",c]]);export{g as __pageData,m as default};
@@ -0,0 +1 @@
1
+ import{_ as n,C as t,o as a,c as r,E as i}from"./chunks/framework.DSg0KOwT.js";const g=JSON.parse('{"title":"Spark Tuning Reference","description":"","frontmatter":{"layout":"page","title":"Spark Tuning Reference","sidebar":false},"headers":[],"relativePath":"tuning-reference/index.md","filePath":"tuning-reference/index.md"}'),o={name:"tuning-reference/index.md"};function c(s,d,p,f,l,u){const e=t("TuningLanding");return a(),r("div",null,[i(e)])}const m=n(o,[["render",c]]);export{g as __pageData,m as default};
@@ -0,0 +1 @@
1
+ import{_ as t,o as e,c as s,a5 as n}from"./chunks/framework.DSg0KOwT.js";const f=JSON.parse('{"title":"Introduction","description":"","frontmatter":{"title":"Introduction"},"headers":[],"relativePath":"tuning-reference/intro.md","filePath":"tuning-reference/intro.md"}'),r={name:"tuning-reference/intro.md"};function o(i,a,l,c,p,h){return e(),s("div",null,[...a[0]||(a[0]=[n('<h1 id="intro" tabindex="-1">Introduction <a class="header-anchor" href="#intro" aria-label="Permalink to &quot;Introduction {#intro}&quot;">​</a></h1><p>This guide is a hands-on reference for optimizing Spark and PySpark jobs. It covers <a href="./spark-architecture.html">Spark internals</a>, <a href="./memory-model.html">memory management</a>, <a href="./joins.html">joins</a>, <a href="./shuffle.html">shuffle</a>, <a href="./data-formats.html">data formats</a>, <a href="./pyspark.html">PySpark-specific patterns</a>, <a href="./aqe.html">Adaptive Query Execution</a> (AQE), and <a href="./cluster-config.html">cluster tuning</a>, followed by a bottleneck diagnostic reference.</p><p>Read it top to bottom for a full grounding in Spark performance, or jump straight to the <strong>Bottleneck Reference</strong> section below when you already know which symptom you&#39;re chasing.</p><h2 id="severity-dots" tabindex="-1">Severity dots <a class="header-anchor" href="#severity-dots" aria-label="Permalink to &quot;Severity dots&quot;">​</a></h2><p>Each finding is marked with a severity dot:</p><ul><li><span class="severity-dot info"></span> <strong>Info</strong>: worth knowing, not yet a problem</li><li><span class="severity-dot warning"></span> <strong>Warning</strong>: likely hurting job performance</li><li><span class="severity-dot critical"></span> <strong>Critical</strong>: actively bottlenecking the job</li></ul><p>Each bottleneck section below documents the exact thresholds behind these dots.</p><h2 id="tag-system" tabindex="-1">Tag system <a class="header-anchor" href="#tag-system" aria-label="Permalink to &quot;Tag system&quot;">​</a></h2><p>Every bottleneck has a short tag used throughout this reference:</p><p><span class="tag">SKEW</span> <span class="tag">SHFL</span> <span class="tag">SPILL</span><span class="tag">GC</span> <span class="tag">COLD</span> <span class="tag">UTIL</span><span class="tag">HOST</span> <span class="tag">FAIL</span> <span class="tag">STRAG</span><span class="tag">RETRY</span> <span class="tag">TINY</span> <span class="tag">FAIL-RATE</span></p><p>See the <a href="./bottleneck-skew.html">Bottleneck Reference</a> for what each one means, how it&#39;s detected, and how to fix it.</p>',11)])])}const g=t(r,[["render",o]]);export{f as __pageData,g as default};
@@ -0,0 +1 @@
1
+ import{_ as t,o as e,c as s,a5 as n}from"./chunks/framework.DSg0KOwT.js";const f=JSON.parse('{"title":"Introduction","description":"","frontmatter":{"title":"Introduction"},"headers":[],"relativePath":"tuning-reference/intro.md","filePath":"tuning-reference/intro.md"}'),r={name:"tuning-reference/intro.md"};function o(i,a,l,c,p,h){return e(),s("div",null,[...a[0]||(a[0]=[n("",11)])])}const g=t(r,[["render",o]]);export{f as __pageData,g as default};
@@ -0,0 +1 @@
1
+ import{_ as a,o as t,c as o,a5 as s}from"./chunks/framework.DSg0KOwT.js";const r="../assets/join-strategy.C_FvrCEo.svg",n="../assets/join-strategy.dark.ChMLnNII.svg",b=JSON.parse('{"title":"Join Optimization","description":"","frontmatter":{"title":"Join Optimization"},"headers":[],"relativePath":"tuning-reference/joins.md","filePath":"tuning-reference/joins.md"}'),i={name:"tuning-reference/joins.md"};function f(c,e,l,h,d,p){return t(),o("div",null,[...e[0]||(e[0]=[s('<h1 id="joins" tabindex="-1">Join Optimization <a class="header-anchor" href="#joins" aria-label="Permalink to &quot;Join Optimization {#joins}&quot;">​</a></h1><h2 id="join-strategy-selection" tabindex="-1">Join strategy selection <a class="header-anchor" href="#join-strategy-selection" aria-label="Permalink to &quot;Join strategy selection&quot;">​</a></h2><p>Spark SQL&#39;s Catalyst optimizer picks from five physical join operators: broadcast hash join, broadcast nested loop join, shuffle hash join, shuffle sort-merge join (SMJ), and shuffle-and-replicate nested loop (cartesian) join<sup class="footnote-ref"><a href="#fn1" id="fnref1">[1]</a></sup>. Broadcast hash join avoids a shuffle entirely by sending one small side to every executor; it requires an equi-join condition and supports every join type except full outer<sup class="footnote-ref"><a href="#fn1" id="fnref1:1">[1:1]</a></sup>. Broadcast nested loop join relaxes the equi-join requirement (it supports non-equi conditions and every join type) at the cost of scanning one side repeatedly, so it&#39;s normally a fallback rather than a first choice<sup class="footnote-ref"><a href="#fn1" id="fnref1:2">[1:2]</a></sup>. Unlike hand-tuned RDD joins, where the partitioner is chosen explicitly, Spark SQL&#39;s optimizer can also push down or reorder other operators automatically to make the eventual join cheaper<sup class="footnote-ref"><a href="#fn1" id="fnref1:3">[1:3]</a></sup>.</p><p>At the communication level, Spark&#39;s choice is binary: an all-to-all <a href="./shuffle.html">shuffle</a> join, or a broadcast join that replicates the small side once so every node can then join locally with no further network traffic<sup class="footnote-ref"><a href="#fn2" id="fnref2">[2]</a></sup>. Once a join has been routed to a shuffle-based strategy (broadcast wasn&#39;t used), Spark defaults to preferring sort-merge join over shuffle hash join, a preference controlled by <code>spark.sql.join.preferSortMergeJoin</code><sup class="footnote-ref"><a href="#fn3" id="fnref3">[3]</a></sup>. Since Spark 3.2, <a href="./aqe.html">Adaptive Query Execution</a> (AQE), on by default, re-optimizes the physical plan mid-execution using runtime statistics gathered after the shuffle actually runs, rather than relying only on pre-execution estimates<sup class="footnote-ref"><a href="#fn4" id="fnref4">[4]</a></sup>.</p><img class="light-only" src="'+r+'" alt="Decision tree for choosing a physical join operator: the equi-join test splits off the nested-loop strategies, then the broadcast-threshold and preferSortMergeJoin checks select broadcast hash, sort-merge, or shuffle hash join, with AQE able to promote sort-merge to broadcast at runtime."><img class="dark-only" src="'+n+'" alt="Decision tree for choosing a physical join operator: the equi-join test splits off the nested-loop strategies, then the broadcast-threshold and preferSortMergeJoin checks select broadcast hash, sort-merge, or shuffle hash join, with AQE able to promote sort-merge to broadcast at runtime."><h2 id="reading-it-in-the-plan" tabindex="-1">Reading it in the plan <a class="header-anchor" href="#reading-it-in-the-plan" aria-label="Permalink to &quot;Reading it in the plan&quot;">​</a></h2><p>That choice is legible before a job even runs. The deciding signal is the relation&#39;s size relative to <code>spark.sql.autoBroadcastJoinThreshold</code>, which defaults to <code>10485760</code> bytes (10 MB), unchanged since it was introduced in Spark 1.1.0 and still the documented default in Spark 3.5<sup class="footnote-ref"><a href="#fn4" id="fnref4:1">[4:1]</a></sup><sup class="footnote-ref"><a href="#fn5" id="fnref5">[5]</a></sup>. That comparison is driven by table/plan size statistics rather than a fresh scan of the data: Spark consults catalog statistics (collected via <code>ANALYZE TABLE</code>, inspectable through <code>DESCRIBE EXTENDED</code>) and the cost estimates shown in <code>EXPLAIN COST</code><sup class="footnote-ref"><a href="#fn4" id="fnref4:2">[4:2]</a></sup>. When those statistics are missing, <code>spark.sql.statistics.fallBackToHdfs</code> (default <code>false</code>) controls whether Spark falls back to on-disk file size to judge broadcast eligibility, and for partitioned tables without statistics Spark instead uses the <code>spark.sql.defaultSizeInBytes</code> placeholder<sup class="footnote-ref"><a href="#fn5" id="fnref5:1">[5:1]</a></sup>.</p><blockquote><p><strong>PySpark:</strong> don&#39;t guess whether a join will broadcast. Call <code>df.explain(mode=&quot;cost&quot;)</code> to see the same size estimates Catalyst used to pick the strategy.</p></blockquote><p>Because AQE re-optimizes at runtime, the physical join operator visible in a plan can change after execution starts. Two runtime-driven overrides are worth checking for in the Spark UI or an <code>EXPLAIN</code> plan: <code>spark.sql.adaptive.maxShuffledHashJoinLocalMapThreshold</code> (default <code>0</code>, disabled) can make AQE prefer shuffled hash join over sort-merge &quot;regardless of the value of <code>spark.sql.join.preferSortMergeJoin</code>&quot; whenever every post-shuffle partition stays under that threshold and above <code>spark.sql.adaptive.advisoryPartitionSizeInBytes</code><sup class="footnote-ref"><a href="#fn4" id="fnref4:3">[4:3]</a></sup>; separately, AQE can convert an already-planned sort-merge join into a broadcast hash join mid-execution when the runtime statistics of either join side turn out smaller than the adaptive broadcast threshold<sup class="footnote-ref"><a href="#fn4" id="fnref4:4">[4:4]</a></sup>.</p><p>Bucketing status is also visible directly in the physical plan: a shuffle-free bucketed join only appears once both sides are bucketed to the same count, at which point the <code>Exchange</code> nodes disappear entirely<sup class="footnote-ref"><a href="#fn6" id="fnref6">[6]</a></sup>. This is demonstrated with 4 buckets on each side<sup class="footnote-ref"><a href="#fn7" id="fnref7">[7]</a></sup> and with 16 buckets on each side, where the plan shows <code>SelectedBucketsCount: 16 out of 16</code> on both branches of the <code>SortMergeJoin</code><sup class="footnote-ref"><a href="#fn8" id="fnref8">[8]</a></sup>. A separate real-world account of eliminating a bucketed shuffle frames the same signal just as plainly: the giveaway is that &quot;the right branch is missing an Exchange (i.e. shuffle)&quot;<sup class="footnote-ref"><a href="#fn6" id="fnref6:1">[6:1]</a></sup>. Removing the shuffle this way does not necessarily remove the sort phase: in the 16-bucket demo, a <code>Sort</code> operator still appears on each branch of the <code>SortMergeJoin</code> even with the <code>Exchange</code> gone<sup class="footnote-ref"><a href="#fn8" id="fnref8:1">[8:1]</a></sup>.</p><p>Whether Catalyst is reordering joins (rather than just individual operators) is also a config check: <code>spark.sql.cbo.enabled</code> and the separate <code>spark.sql.cbo.joinReorder.enabled</code> flag both default to <code>false</code>, so multi-way join reordering is off unless both are explicitly enabled<sup class="footnote-ref"><a href="#fn5" id="fnref5:2">[5:2]</a></sup>. That&#39;s distinct from Catalyst&#39;s regular rule-based optimizations, which run regardless of those flags. For example, a filter written after a join in the DataFrame API was observed moved before the join (and pushed into the JDBC source) automatically, visible in the physical plan<sup class="footnote-ref"><a href="#fn9" id="fnref9">[9]</a></sup>.</p><h2 id="costs-the-plan-doesn-t-show" tabindex="-1">Costs the plan doesn&#39;t show <a class="header-anchor" href="#costs-the-plan-doesn-t-show" aria-label="Permalink to &quot;Costs the plan doesn&#39;t show&quot;">​</a></h2><p>Picking a strategy is one thing; paying for it is another. Broadcasting isn&#39;t free on the driver side. Building a broadcast join replicates the small-side DataFrame to every worker, but that replication is preceded by collecting the DataFrame back to the driver first: an <a href="./bottleneck-broadcast-sizing.html">oversized broadcast</a>, whether chosen automatically or forced via a hint or <code>broadcast()</code> call, &quot;can crash your driver node (because that collect is expensive)&quot;<sup class="footnote-ref"><a href="#fn2" id="fnref2:1">[2:1]</a></sup>. The shuffle-and-replicate nested loop (cartesian-style) strategy carries a related but distinct risk: because every partition is joined against every other partition, it has a high chance of data explosion<sup class="footnote-ref"><a href="#fn1" id="fnref1:4">[1:4]</a></sup>.</p><p>Bucketing&#39;s shuffle-free path also has a real tradeoff once bucket counts don&#39;t match exactly. Since Spark 3.1, <code>spark.sql.bucketing.coalesceBucketsInJoin.enabled</code> lets Spark coalesce the side with more buckets down to the smaller count, but only within a ratio bounded by <code>spark.sql.bucketing.coalesceBucketsInJoin.maxBucketRatio</code> (default <code>4</code>). Enabling it can still remove the shuffle, but it does so by cutting parallelism on the finer-grained side, and Spark&#39;s own docs note it &quot;could possibly cause OOM for shuffled hash join&quot; as a result<sup class="footnote-ref"><a href="#fn5" id="fnref5:3">[5:3]</a></sup>. Outside that ratio, or with coalescing disabled, mismatched bucket counts get no free win at all: the join simply falls back to a normal shuffle.</p><h2 id="forcing-a-strategy" tabindex="-1">Forcing a strategy <a class="header-anchor" href="#forcing-a-strategy" aria-label="Permalink to &quot;Forcing a strategy&quot;">​</a></h2><p>When the automatic choice isn&#39;t the right one, Spark still lets you override it directly. Join hints force a specific strategy per relation, and Spark honors them even against its own size-based defaults:</p><ul><li><strong>BROADCAST</strong> (also accepted as <code>BROADCASTJOIN</code>/<code>MAPJOIN</code>) forces a broadcast join with the hinted relation as the build side (broadcast hash if there&#39;s an equi-join key, broadcast nested loop otherwise), and this is honored &quot;even if the size of table &#39;t1&#39; suggested by the statistics is above the configuration <code>spark.sql.autoBroadcastJoinThreshold</code>&quot;<sup class="footnote-ref"><a href="#fn4" id="fnref4:5">[4:5]</a></sup>.</li><li><strong>MERGE</strong> forces a shuffle sort-merge join; it needs an equi-join on sortable keys and high cardinality to pay off, and is less prone to OOM than the hash-based options since it never builds an in-memory hash table<sup class="footnote-ref"><a href="#fn1" id="fnref1:5">[1:5]</a></sup>.</li><li><strong>SHUFFLE_HASH</strong> forces a shuffle hash join, supporting all join types on an equi-join key but, like MERGE, wanting high cardinality; it builds a per-partition hash table after the shuffle<sup class="footnote-ref"><a href="#fn1" id="fnref1:6">[1:6]</a></sup>.</li><li><strong>SHUFFLE_REPLICATE_NL</strong> forces the shuffle-and-replicate nested loop join, supporting inner and cartesian joins with equi or non-equi conditions<sup class="footnote-ref"><a href="#fn1" id="fnref1:7">[1:7]</a></sup>.</li></ul><p>When both sides of a join carry conflicting hints, Spark resolves them by a fixed priority: BROADCAST over MERGE over SHUFFLE_HASH over SHUFFLE_REPLICATE_NL. If both sides carry the <em>same</em> BROADCAST or SHUFFLE_HASH hint, Spark still picks a build side based on join type and relation sizes<sup class="footnote-ref"><a href="#fn4" id="fnref4:6">[4:6]</a></sup>. As with any hint, none of this is guaranteed: Spark won&#39;t honor a hinted strategy that structurally can&#39;t support the actual join type being performed<sup class="footnote-ref"><a href="#fn4" id="fnref4:7">[4:7]</a></sup>. The threshold itself can also be turned off outright: setting <code>spark.sql.autoBroadcastJoinThreshold</code> to <code>-1</code> forces every join to fall back to shuffle sort-merge instead of ever considering broadcast<sup class="footnote-ref"><a href="#fn3" id="fnref3:1">[3:1]</a></sup>.</p><p>For repeated joins on the same key, bucketing both tables to the <em>same</em> bucket count removes the shuffle for free. To also remove the sort phase (not just the shuffle), the bucketed tables additionally need to be written pre-sorted on the join key, using <code>sortBy</code> alongside <code>bucketBy</code>; done that way, the merge phase has nothing left to do, since &quot;the joined output is sorted ... because we saved the tables sorted in ascending order ... there&#39;s no need to sort during the <code>SortMergeJoin</code>,&quot; and the Spark UI shows the query going straight to <code>WholeStageCodegen</code> with no <code>Exchange</code> at all<sup class="footnote-ref"><a href="#fn3" id="fnref3:2">[3:2]</a></sup>.</p><blockquote><p><strong>PySpark:</strong> <code>df.write.bucketBy(n, &quot;key&quot;).saveAsTable(...)</code> drops the shuffle; add <code>.sortBy(&quot;key&quot;)</code> on both tables to drop the sort phase too.</p></blockquote><p>Multi-way join reordering is a deliberate opt-in: enable <code>spark.sql.cbo.enabled</code> for cost-based statistics, then <code>spark.sql.cbo.joinReorder.enabled</code> for the reordering rule itself. <code>spark.sql.cbo.joinReorder.dp.threshold</code> caps the dynamic-programming enumeration at 12 joined nodes by default, and <code>spark.sql.cbo.joinReorder.dp.star.filter</code> applies star-join filter heuristics on top of it<sup class="footnote-ref"><a href="#fn5" id="fnref5:4">[5:4]</a></sup>. A lighter-weight alternative that skips full CBO is <code>spark.sql.cbo.starSchemaDetection</code> (default <code>false</code>), which enables join reordering based on star-schema detection alone<sup class="footnote-ref"><a href="#fn5" id="fnref5:5">[5:5]</a></sup>.</p><p>For <a href="./bottleneck-skew.html">skewed join keys</a>, Spark offers two built-in alternatives to hand-rolled salting: AQE&#39;s skew-join handling (<code>spark.sql.adaptive.skewJoin.enabled</code>), which detects oversized shuffle partitions at runtime and splits them automatically, replicating if needed<sup class="footnote-ref"><a href="#fn10" id="fnref10">[10]</a></sup><sup class="footnote-ref"><a href="#fn5" id="fnref5:6">[5:6]</a></sup>, and Databricks&#39; declarative <code>SKEW</code> hint, which builds a skew-aware plan without any manual salting<sup class="footnote-ref"><a href="#fn11" id="fnref11">[11]</a></sup>. Manual salting remains the fallback where neither is available: add a random salt column to the join key on both sides so a hot key spreads across many partitions (exploding the dimension side into one row per salt value and assigning a random salt on the fact side), then join on the composite <code>(key, salt)</code> pair<sup class="footnote-ref"><a href="#fn12" id="fnref12">[12]</a></sup>.</p><h2 id="sources" tabindex="-1">Sources <a class="header-anchor" href="#sources" aria-label="Permalink to &quot;Sources&quot;">​</a></h2><hr class="footnotes-sep"><section class="footnotes"><ol class="footnotes-list"><li id="fn1" class="footnote-item"><p><em>High Performance Spark, 2nd Edition</em>, Karau, Polak &amp; Warren, ch. 6 <a href="#fnref1" class="footnote-backref">↩︎</a> <a href="#fnref1:1" class="footnote-backref">↩︎</a> <a href="#fnref1:2" class="footnote-backref">↩︎</a> <a href="#fnref1:3" class="footnote-backref">↩︎</a> <a href="#fnref1:4" class="footnote-backref">↩︎</a> <a href="#fnref1:5" class="footnote-backref">↩︎</a> <a href="#fnref1:6" class="footnote-backref">↩︎</a> <a href="#fnref1:7" class="footnote-backref">↩︎</a></p></li><li id="fn2" class="footnote-item"><p><em>Spark: The Definitive Guide</em>, Chambers &amp; Zaharia, ch. 8 <a href="#fnref2" class="footnote-backref">↩︎</a> <a href="#fnref2:1" class="footnote-backref">↩︎</a></p></li><li id="fn3" class="footnote-item"><p><em>Learning Spark, 2nd Edition</em>, Damji, Wenig, Das, Lee, ch. 7 <a href="#fnref3" class="footnote-backref">↩︎</a> <a href="#fnref3:1" class="footnote-backref">↩︎</a> <a href="#fnref3:2" class="footnote-backref">↩︎</a></p></li><li id="fn4" class="footnote-item"><p><a href="https://spark.apache.org/docs/latest/sql-performance-tuning.html" target="_blank" rel="noreferrer">Performance Tuning — Spark SQL, DataFrames and Datasets Guide</a> <a href="#fnref4" class="footnote-backref">↩︎</a> <a href="#fnref4:1" class="footnote-backref">↩︎</a> <a href="#fnref4:2" class="footnote-backref">↩︎</a> <a href="#fnref4:3" class="footnote-backref">↩︎</a> <a href="#fnref4:4" class="footnote-backref">↩︎</a> <a href="#fnref4:5" class="footnote-backref">↩︎</a> <a href="#fnref4:6" class="footnote-backref">↩︎</a> <a href="#fnref4:7" class="footnote-backref">↩︎</a></p></li><li id="fn5" class="footnote-item"><p><a href="https://spark.apache.org/docs/latest/configuration.html" target="_blank" rel="noreferrer">Configuration — Spark</a> <a href="#fnref5" class="footnote-backref">↩︎</a> <a href="#fnref5:1" class="footnote-backref">↩︎</a> <a href="#fnref5:2" class="footnote-backref">↩︎</a> <a href="#fnref5:3" class="footnote-backref">↩︎</a> <a href="#fnref5:4" class="footnote-backref">↩︎</a> <a href="#fnref5:5" class="footnote-backref">↩︎</a> <a href="#fnref5:6" class="footnote-backref">↩︎</a></p></li><li id="fn6" class="footnote-item"><p><a href="https://www.taboola.com/engineering/bucket-the-shuffle-out-of-here/" target="_blank" rel="noreferrer">Bucket the Shuffle Out of Here</a> <a href="#fnref6" class="footnote-backref">↩︎</a> <a href="#fnref6:1" class="footnote-backref">↩︎</a></p></li><li id="fn7" class="footnote-item"><p><a href="https://books.japila.pl/spark-sql-internals/bucketing/" target="_blank" rel="noreferrer">Bucketing — The Internals of Spark SQL</a> <a href="#fnref7" class="footnote-backref">↩︎</a></p></li><li id="fn8" class="footnote-item"><p><a href="https://luminousmen.com/post/the-5-minute-guide-to-using-bucketing-in-pyspark" target="_blank" rel="noreferrer">The 5-Minute Guide to Using Bucketing in PySpark</a> <a href="#fnref8" class="footnote-backref">↩︎</a> <a href="#fnref8:1" class="footnote-backref">↩︎</a></p></li><li id="fn9" class="footnote-item"><p><a href="https://luminousmen.com/post/spark-tips-dataframe-api" target="_blank" rel="noreferrer">Spark Tips: DataFrame API</a> <a href="#fnref9" class="footnote-backref">↩︎</a></p></li><li id="fn10" class="footnote-item"><p><a href="https://issues.apache.org/jira/browse/SPARK-29544" target="_blank" rel="noreferrer">SPARK-29544 — Optimize Skewed Join at Runtime</a> <a href="#fnref10" class="footnote-backref">↩︎</a></p></li><li id="fn11" class="footnote-item"><p><a href="https://docs.databricks.com/aws/en/archive/legacy/skew-join" target="_blank" rel="noreferrer">Skew Join Hint</a> <a href="#fnref11" class="footnote-backref">↩︎</a></p></li><li id="fn12" class="footnote-item"><p><a href="https://luminousmen.com/post/spark-tips-partition-tuning" target="_blank" rel="noreferrer">Spark Tips: Partition Tuning</a> <a href="#fnref12" class="footnote-backref">↩︎</a></p></li></ol></section>',26)])])}const g=a(i,[["render",f]]);export{b as __pageData,g as default};
@@ -0,0 +1 @@
1
+ import{_ as a,o as t,c as o,a5 as s}from"./chunks/framework.DSg0KOwT.js";const r="../assets/join-strategy.C_FvrCEo.svg",n="../assets/join-strategy.dark.ChMLnNII.svg",b=JSON.parse('{"title":"Join Optimization","description":"","frontmatter":{"title":"Join Optimization"},"headers":[],"relativePath":"tuning-reference/joins.md","filePath":"tuning-reference/joins.md"}'),i={name:"tuning-reference/joins.md"};function f(c,e,l,h,d,p){return t(),o("div",null,[...e[0]||(e[0]=[s("",26)])])}const g=a(i,[["render",f]]);export{b as __pageData,g as default};
@@ -0,0 +1 @@
1
+ import{_ as o,o as a,c as r,a5 as t}from"./chunks/framework.DSg0KOwT.js";const s="../assets/memory-regions.XHvO7jHG.svg",n="../assets/memory-regions.dark.D4TP9_08.svg",f="../assets/memory-borrowing.BqQRJg0u.svg",i="../assets/memory-borrowing.dark.Yhh20O9C.svg",y=JSON.parse('{"title":"Memory Management","description":"","frontmatter":{"title":"Memory Management"},"headers":[],"relativePath":"tuning-reference/memory-model.md","filePath":"tuning-reference/memory-model.md"}'),c={name:"tuning-reference/memory-model.md"};function l(h,e,d,p,u,m){return a(),r("div",null,[...e[0]||(e[0]=[t('<h1 id="memory-model" tabindex="-1">Memory Management <a class="header-anchor" href="#memory-model" aria-label="Permalink to &quot;Memory Management {#memory-model}&quot;">​</a></h1><h2 id="the-unified-memory-model" tabindex="-1">The unified memory model <a class="header-anchor" href="#the-unified-memory-model" aria-label="Permalink to &quot;The unified memory model&quot;">​</a></h2><p>Since Spark 1.6, the <code>UnifiedMemoryManager</code> has replaced the earlier static split between execution and storage memory with a single shared region, referred to as M<sup class="footnote-ref"><a href="#fn1" id="fnref1">[1]</a></sup><sup class="footnote-ref"><a href="#fn2" id="fnref2">[2]</a></sup>. M is carved out of the JVM heap after first setting aside 300 MiB of reserved system memory: <code>M = (JVM heap − 300 MiB) × spark.memory.fraction</code>, with <code>spark.memory.fraction</code> defaulting to 0.6<sup class="footnote-ref"><a href="#fn2" id="fnref2:1">[2:1]</a></sup><sup class="footnote-ref"><a href="#fn3" id="fnref3">[3]</a></sup>. Within M, <code>spark.memory.storageFraction</code> (default 0.5) marks off a subregion reserved for cached blocks that are immune to eviction; with defaults, this works out to roughly a 30%/30% on-heap split between execution and storage out of total executor memory<sup class="footnote-ref"><a href="#fn3" id="fnref3:1">[3:1]</a></sup>.</p><img class="light-only" src="'+s+'" alt="The executor memory layout showing reserved memory, the unified region split into execution and storage, and user memory inside the JVM heap, with the off-heap pool and memory overhead outside it."><img class="dark-only" src="'+n+'" alt="The executor memory layout showing reserved memory, the unified region split into execution and storage, and user memory inside the JVM heap, with the off-heap pool and memory overhead outside it."><p>Execution and storage borrow from each other dynamically rather than sitting behind a hard wall: when execution memory is unused, storage can acquire all of it, and vice versa<sup class="footnote-ref"><a href="#fn2" id="fnref2:2">[2:2]</a></sup><sup class="footnote-ref"><a href="#fn1" id="fnref1:1">[1:1]</a></sup>. The two are not equal peers, though. Execution always has priority, taking memory immediately and evicting cached storage blocks if necessary<sup class="footnote-ref"><a href="#fn3" id="fnref3:2">[3:2]</a></sup>.</p><p>The reserved 300 MB is a flat constant (<code>RESERVED_SYSTEM_MEMORY_BYTES</code> in the Spark source), not a formula, and it&#39;s hardcoded: there is no supported production configuration to change it. The only override is the internal <code>spark.testing.reservedMemory</code> property, which exists for Spark&#39;s own test suite rather than production tuning<sup class="footnote-ref"><a href="#fn3" id="fnref3:3">[3:3]</a></sup>. <code>spark.memory.fraction</code> is applied against heap minus that reserved 300 MB, not against total heap. The tuning guide is explicit that it &quot;expresses the size of M as a fraction of the (JVM heap space − 300MiB)&quot;<sup class="footnote-ref"><a href="#fn2" id="fnref2:3">[2:3]</a></sup>.</p><p>Off-heap memory extends the same model outside the JVM heap. When <code>spark.memory.offHeap.enabled=true</code>, Spark allocates raw off-heap buffers via <code>sun.misc.Unsafe</code> (routed through <code>jdk.internal.misc.Unsafe</code> on JDK 17+, which is why some builds need <code>--add-opens</code> flags), and that off-heap region is split into its own execution and storage pools, following the same borrowing rules as on-heap memory: execution has priority, storage gets evicted first<sup class="footnote-ref"><a href="#fn3" id="fnref3:4">[3:4]</a></sup>. Enabling it requires <code>spark.memory.offHeap.size</code> to be set to a positive value; leaving it enabled with the default size of <code>0</code> is a documented-invalid combination, not a supported way to run with off-heap &quot;on but empty&quot;<sup class="footnote-ref"><a href="#fn4" id="fnref4">[4]</a></sup>. The setting also &quot;has no impact on heap memory usage&quot;. Turning on off-heap memory does not shrink the JVM <code>-Xmx</code> for you, so the on-heap size has to be reduced manually to keep total footprint constant<sup class="footnote-ref"><a href="#fn4" id="fnref4:1">[4:1]</a></sup>. This off-heap execution memory underlies Project Tungsten: its compact binary row encoding, its explicit-memory-managed hash map for aggregations, and its cache-aware sort/join algorithms are all built to run against off-heap, GC-invisible memory<sup class="footnote-ref"><a href="#fn5" id="fnref5">[5]</a></sup><sup class="footnote-ref"><a href="#fn6" id="fnref6">[6]</a></sup>.</p><p>Executor memory overhead is a separate pool layered on top of M. <code>spark.executor.memoryOverhead</code> defaults to <code>executorMemory × spark.executor.memoryOverheadFactor</code>, floored at <code>spark.executor.minMemoryOverhead</code> (384 MiB)<sup class="footnote-ref"><a href="#fn4" id="fnref4:2">[4:2]</a></sup>. <code>spark.executor.memoryOverheadFactor</code> itself defaults to 0.10 for ordinary JVM executors, but Spark bumps that default to 0.40 specifically for Kubernetes non-JVM jobs, since those workloads tend to need more non-JVM heap space<sup class="footnote-ref"><a href="#fn4" id="fnref4:3">[4:3]</a></sup>. The legacy, YARN-only <code>spark.yarn.executor.memoryOverhead</code> property was removed in Spark 3.0 in favor of the cluster-manager-agnostic <code>spark.executor.memoryOverhead</code><sup class="footnote-ref"><a href="#fn3" id="fnref3:5">[3:5]</a></sup>.</p><blockquote><p><strong>PySpark:</strong> <code>spark.python.worker.memory</code> (default <code>512m</code>) is a soft spill threshold for a Python worker&#39;s own aggregation buffering, not a JVM-enforced cap. A PySpark worker runs as a separate OS process outside the JVM heap, so the JVM cannot police it directly<sup class="footnote-ref"><a href="#fn3" id="fnref3:6">[3:6]</a></sup>. The only setting that actually tries to bound PySpark memory per executor is <code>spark.executor.pyspark.memory</code>, and even that depends on Python&#39;s <code>resource</code> module, which isn&#39;t supported on Windows and doesn&#39;t actually limit anything on macOS<sup class="footnote-ref"><a href="#fn4" id="fnref4:4">[4:4]</a></sup>. Left unset, PySpark memory usage is folded into <code>spark.executor.memoryOverhead</code> by default<sup class="footnote-ref"><a href="#fn4" id="fnref4:5">[4:5]</a></sup>.</p></blockquote><h2 id="watching-eviction-and-spill" tabindex="-1">Watching eviction and spill <a class="header-anchor" href="#watching-eviction-and-spill" aria-label="Permalink to &quot;Watching eviction and spill&quot;">​</a></h2><p>That borrowing leaves a trace once eviction actually happens. Storage eviction under memory pressure follows an LRU (least recently used) policy<sup class="footnote-ref"><a href="#fn1" id="fnref1:2">[1:2]</a></sup><sup class="footnote-ref"><a href="#fn7" id="fnref7">[7]</a></sup>. The granularity is the block: the <code>BlockManager</code> tracks cached data as per-partition blocks (one block per RDD/DataFrame partition) and LRU operates at that level rather than evicting an entire RDD or cached table in one shot<sup class="footnote-ref"><a href="#fn8" id="fnref8">[8]</a></sup><sup class="footnote-ref"><a href="#fn9" id="fnref9">[9]</a></sup>. In practice this tends to evict the oldest partitions first, those materialized in the earliest job or stage, though lazy evaluation makes it hard to predict exactly which partitions will go ahead of time<sup class="footnote-ref"><a href="#fn9" id="fnref9:1">[9:1]</a></sup>.</p><p>What happens to an evicted block is visible in its downstream effect and depends on its <code>StorageLevel</code>. For <code>MEMORY_ONLY</code>, the block is simply dropped and recomputed from the RDD&#39;s lineage the next time it&#39;s needed. For a disk-backed level like <code>MEMORY_AND_DISK</code>, the evicted block is written to disk and read back from there instead of being recomputed<sup class="footnote-ref"><a href="#fn9" id="fnref9:2">[9:2]</a></sup><sup class="footnote-ref"><a href="#fn7" id="fnref7:1">[7:1]</a></sup>. Blocks can also be evicted well before you ever try to reuse them: <code>.cache()</code> does not guarantee a block stays resident, especially on busy clusters or long pipelines<sup class="footnote-ref"><a href="#fn7" id="fnref7:2">[7:2]</a></sup>.</p><p>On the execution side, each task gets its own <code>TaskMemoryManager</code>, which enforces a soft per-task cap: with <code>n</code> tasks running concurrently, each task is allowed to allocate somewhere between <code>1/(2n)</code> and <code>1/n</code> of total execution memory, and the first task to arrive on an idle executor typically grabs more than its later-arriving neighbors<sup class="footnote-ref"><a href="#fn3" id="fnref3:7">[3:7]</a></sup>.</p><p>When a sort or hash-aggregation operator keeps requesting execution-memory pages and can&#39;t get more, Spark doesn&#39;t fail immediately. It blocks the requesting task, spills that task&#39;s in-memory data structure to disk, or, in the worst case, throws an <code>OutOfMemoryError</code><sup class="footnote-ref"><a href="#fn3" id="fnref3:8">[3:8]</a></sup>. The spill path runs through Spark&#39;s map/shuffle I/O machinery: shuffle partitions created by wide transformations like <code>groupBy()</code> or <code>join()</code> <a href="./bottleneck-spill.html">spill</a> to the executors&#39; local disks at the location set by <code>spark.local.directory</code><sup class="footnote-ref"><a href="#fn6" id="fnref6:1">[6:1]</a></sup>. SQL physical operators apply the same idea with their own row-count guardrails: <code>sortMergeJoinExec</code>&#39;s in-memory buffer and the cartesian-product operator&#39;s buffer both spill once they cross a configured row threshold, which by default is set to the value of <code>spark.shuffle.spill.numElementsForceSpillThreshold</code><sup class="footnote-ref"><a href="#fn10" id="fnref10">[10]</a></sup>.</p><p>Whatever the trigger, the resulting spilled bytes are exposed on the task metrics as <code>memoryBytesSpilled</code>, visible in the Spark UI and event log. This is the concrete signal to watch for execution memory pressure<sup class="footnote-ref"><a href="#fn11" id="fnref11">[11]</a></sup>.</p><p>Memory-overhead misconfiguration shows up differently: it manifests as the container or pod being OOMKilled with a vague container-memory error rather than a Spark-level exception, since the enforcement happens at the YARN NodeManager or Kubernetes kubelet/cgroup layer, not inside the JVM<sup class="footnote-ref"><a href="#fn3" id="fnref3:9">[3:9]</a></sup>. Under the old default overhead factor of 0.10, non-JVM Kubernetes jobs commonly failed with &quot;Memory Overhead Exceeded&quot; errors, which is exactly why Spark bumped the Kubernetes non-JVM default to 0.40<sup class="footnote-ref"><a href="#fn4" id="fnref4:6">[4:6]</a></sup>.</p><h2 id="what-the-borrowing-costs-you" tabindex="-1">What the borrowing costs you <a class="header-anchor" href="#what-the-borrowing-costs-you" aria-label="Permalink to &quot;What the borrowing costs you&quot;">​</a></h2><p>None of this is free for whoever&#39;s counting on cached data staying put. Because execution always wins the tug-of-war over storage, <a href="./caching.html">caching</a> is never a guarantee; it&#39;s a best effort. As luminousmen-memory-management puts it: &quot;If Execution needs memory, it takes it. If Storage is using that space (cached RDDs, broadcasts), Spark starts evicting blocks. If Execution is idle, Storage can grow into that space, until Execution comes back.&quot;<sup class="footnote-ref"><a href="#fn3" id="fnref3:10">[3:10]</a></sup> That growth-then-shrink-back behavior is dynamic borrowing, not storage evicting execution; eviction is strictly one-directional. As spark-notes-task-memory-management summarizes the agreement between the two regions: &quot;keep acquiring execution memory and evict storage as you need more execution memory&quot;<sup class="footnote-ref"><a href="#fn1" id="fnref1:3">[1:3]</a></sup>, never the reverse. The same asymmetric rule carries over when off-heap memory is enabled: execution still has priority and storage still gets evicted first<sup class="footnote-ref"><a href="#fn3" id="fnref3:11">[3:11]</a></sup>. Practically, this means a cached DataFrame can silently lose blocks under memory pressure, forcing an expensive lineage recompute (for <code>MEMORY_ONLY</code>) or an extra disk round-trip (for <code>MEMORY_AND_DISK</code>) the next time it&#39;s touched.</p><img class="light-only" src="'+f+'" alt="How storage memory borrows idle execution space while execution reclaims its own space by evicting storage in one direction."><img class="dark-only" src="'+i+'" alt="How storage memory borrows idle execution space while execution reclaims its own space by evicting storage in one direction."><p>The <code>TaskMemoryManager</code>&#39;s soft per-task cap explains why spill behavior is workload-shape-dependent rather than a fixed threshold: with more tasks packed onto an executor, each one&#39;s guaranteed share of execution memory shrinks toward <code>1/(2n)</code>, making spills more likely under high task concurrency even when total execution memory hasn&#39;t changed<sup class="footnote-ref"><a href="#fn3" id="fnref3:12">[3:12]</a></sup>.</p><p>Off-heap memory&#39;s payoff is specifically about <a href="./bottleneck-gc.html">garbage collection</a>: because off-heap buffers sit outside the JVM heap, they are invisible to the garbage collector, so fewer and smaller live objects need to be tracked, scanned, and copied, which reduces both the frequency and duration of GC pauses<sup class="footnote-ref"><a href="#fn3" id="fnref3:13">[3:13]</a></sup>. Project Tungsten&#39;s off-heap hash map for aggregations was benchmarked at over 1 million operations per second in a single thread, with &quot;almost no performance degradation as memory utilization increases,&quot; unlike the JVM default <code>java.util.HashMap</code>, which eventually thrashes on GC<sup class="footnote-ref"><a href="#fn5" id="fnref5:1">[5:1]</a></sup><sup class="footnote-ref"><a href="#fn6" id="fnref6:2">[6:2]</a></sup>. The tradeoff is that off-heap memory removes the GC safety net along with the GC overhead: there&#39;s no garbage collector cleaning up if something goes wrong<sup class="footnote-ref"><a href="#fn3" id="fnref3:14">[3:14]</a></sup>.</p><p>Memory overhead matters because off-heap memory and (if unset) PySpark memory are not automatically folded into the overhead calculation the way heap memory is. Enabling <code>spark.memory.offHeap.size</code> without also raising <code>spark.executor.memoryOverhead</code> (or explicitly setting <code>spark.executor.pyspark.memory</code>) can push the executor&#39;s real footprint past what the cluster manager granted, and the process gets killed with a vague container-memory error instead of a Spark-level exception<sup class="footnote-ref"><a href="#fn3" id="fnref3:15">[3:15]</a></sup>. Worked example: with <code>--executor-memory=8G</code>, the default 10% overhead gives <code>max(0.1 × 8192 MB, 384 MB) = 819 MB</code>, so the total memory requested from the cluster manager is <code>8192 + 819 = 9011 MB</code><sup class="footnote-ref"><a href="#fn3" id="fnref3:16">[3:16]</a></sup>, useful to keep in mind when <a href="./cluster-config.html">sizing containers or pods</a> against a cluster&#39;s available capacity.</p><h2 id="tuning-the-memory-pools" tabindex="-1">Tuning the memory pools <a class="header-anchor" href="#tuning-the-memory-pools" aria-label="Permalink to &quot;Tuning the memory pools&quot;">​</a></h2><p>Most of this is adjustable, within limits. Tune <code>spark.memory.fraction</code> and <code>spark.memory.storageFraction</code> only if your workload&#39;s execution/storage balance genuinely needs to shift away from the ~30/30 default. But don&#39;t try to reclaim the 300 MB reserved region; it&#39;s a fixed, non-tunable constant in production<sup class="footnote-ref"><a href="#fn3" id="fnref3:17">[3:17]</a></sup>.</p><p>If losing cached partitions to eviction is costly, prefer a disk-backed <code>StorageLevel</code> such as <code>MEMORY_AND_DISK</code> over <code>MEMORY_ONLY</code>: an evicted block gets written to disk and read back rather than triggering a full lineage recompute<sup class="footnote-ref"><a href="#fn9" id="fnref9:3">[9:3]</a></sup><sup class="footnote-ref"><a href="#fn7" id="fnref7:3">[7:3]</a></sup>. Keep in mind <code>.cache()</code> alone is not a residency guarantee even with this choice<sup class="footnote-ref"><a href="#fn7" id="fnref7:4">[7:4]</a></sup>.</p><p>Watch <code>memoryBytesSpilled</code> in the Spark UI or history server to catch execution-memory pressure early<sup class="footnote-ref"><a href="#fn11" id="fnref11:1">[11:1]</a></sup>. If sort or hash-aggregation spills are heavy, consider giving the executor more memory or reducing the number of concurrently running tasks per executor, since the <code>TaskMemoryManager</code>&#39;s per-task guarantee shrinks as task concurrency <code>n</code> grows<sup class="footnote-ref"><a href="#fn3" id="fnref3:18">[3:18]</a></sup>. For SQL joins and cartesian products, <code>spark.shuffle.spill.numElementsForceSpillThreshold</code> governs when <code>sortMergeJoinExec</code> and the cartesian-product operator spill their buffers<sup class="footnote-ref"><a href="#fn10" id="fnref10:1">[10:1]</a></sup>.</p><p>Set <code>spark.executor.memoryOverhead</code> explicitly rather than relying on the default, especially on Kubernetes for non-JVM workloads (where the default factor is already bumped to 0.40) or whenever off-heap memory or heavy PySpark usage is in play: those aren&#39;t automatically folded into the overhead calculation, so an unadjusted overhead can lead to an OOMKilled container instead of a clean Spark-level error<sup class="footnote-ref"><a href="#fn3" id="fnref3:19">[3:19]</a></sup><sup class="footnote-ref"><a href="#fn4" id="fnref4:7">[4:7]</a></sup>. Use <code>spark.executor.memoryOverhead</code>, not the removed <code>spark.yarn.executor.memoryOverhead</code><sup class="footnote-ref"><a href="#fn3" id="fnref3:20">[3:20]</a></sup>. For capacity planning, remember the total requested from the cluster manager is executor memory plus overhead, e.g., <code>8192 + 819 = 9011 MB</code> for an 8 GB executor at the default 10% factor<sup class="footnote-ref"><a href="#fn3" id="fnref3:21">[3:21]</a></sup>.</p><p>When enabling off-heap memory, always pair <code>spark.memory.offHeap.enabled=true</code> with an explicit positive <code>spark.memory.offHeap.size</code> (never leave it enabled at the default size of <code>0</code><sup class="footnote-ref"><a href="#fn4" id="fnref4:8">[4:8]</a></sup>) and manually reduce the JVM <code>-Xmx</code> to compensate, since enabling off-heap memory does not shrink the heap for you<sup class="footnote-ref"><a href="#fn4" id="fnref4:9">[4:9]</a></sup>.</p><blockquote><p><strong>PySpark:</strong> if you need PySpark&#39;s own memory bounded rather than folded silently into the overhead, set <code>spark.executor.pyspark.memory</code> explicitly, though its enforcement relies on Python&#39;s <code>resource</code> module and won&#39;t work on Windows and won&#39;t actually limit anything on macOS<sup class="footnote-ref"><a href="#fn4" id="fnref4:10">[4:10]</a></sup>.</p></blockquote><h2 id="sources" tabindex="-1">Sources <a class="header-anchor" href="#sources" aria-label="Permalink to &quot;Sources&quot;">​</a></h2><hr class="footnotes-sep"><section class="footnotes"><ol class="footnotes-list"><li id="fn1" class="footnote-item"><p><a href="https://raw.githubusercontent.com/spoddutur/spark-notes/master/task_memory_management_in_spark.md" target="_blank" rel="noreferrer">Task Memory Management in Spark</a> <a href="#fnref1" class="footnote-backref">↩︎</a> <a href="#fnref1:1" class="footnote-backref">↩︎</a> <a href="#fnref1:2" class="footnote-backref">↩︎</a> <a href="#fnref1:3" class="footnote-backref">↩︎</a></p></li><li id="fn2" class="footnote-item"><p><a href="https://spark.apache.org/docs/latest/tuning.html" target="_blank" rel="noreferrer">Tuning Spark</a> <a href="#fnref2" class="footnote-backref">↩︎</a> <a href="#fnref2:1" class="footnote-backref">↩︎</a> <a href="#fnref2:2" class="footnote-backref">↩︎</a> <a href="#fnref2:3" class="footnote-backref">↩︎</a></p></li><li id="fn3" class="footnote-item"><p><a href="https://luminousmen.com/post/dive-into-spark-memory" target="_blank" rel="noreferrer">Dive into Spark Memory</a> <a href="#fnref3" class="footnote-backref">↩︎</a> <a href="#fnref3:1" class="footnote-backref">↩︎</a> <a href="#fnref3:2" class="footnote-backref">↩︎</a> <a href="#fnref3:3" class="footnote-backref">↩︎</a> <a href="#fnref3:4" class="footnote-backref">↩︎</a> <a href="#fnref3:5" class="footnote-backref">↩︎</a> <a href="#fnref3:6" class="footnote-backref">↩︎</a> <a href="#fnref3:7" class="footnote-backref">↩︎</a> <a href="#fnref3:8" class="footnote-backref">↩︎</a> <a href="#fnref3:9" class="footnote-backref">↩︎</a> <a href="#fnref3:10" class="footnote-backref">↩︎</a> <a href="#fnref3:11" class="footnote-backref">↩︎</a> <a href="#fnref3:12" class="footnote-backref">↩︎</a> <a href="#fnref3:13" class="footnote-backref">↩︎</a> <a href="#fnref3:14" class="footnote-backref">↩︎</a> <a href="#fnref3:15" class="footnote-backref">↩︎</a> <a href="#fnref3:16" class="footnote-backref">↩︎</a> <a href="#fnref3:17" class="footnote-backref">↩︎</a> <a href="#fnref3:18" class="footnote-backref">↩︎</a> <a href="#fnref3:19" class="footnote-backref">↩︎</a> <a href="#fnref3:20" class="footnote-backref">↩︎</a> <a href="#fnref3:21" class="footnote-backref">↩︎</a></p></li><li id="fn4" class="footnote-item"><p><a href="https://spark.apache.org/docs/latest/configuration.html" target="_blank" rel="noreferrer">Configuration — Spark</a> <a href="#fnref4" class="footnote-backref">↩︎</a> <a href="#fnref4:1" class="footnote-backref">↩︎</a> <a href="#fnref4:2" class="footnote-backref">↩︎</a> <a href="#fnref4:3" class="footnote-backref">↩︎</a> <a href="#fnref4:4" class="footnote-backref">↩︎</a> <a href="#fnref4:5" class="footnote-backref">↩︎</a> <a href="#fnref4:6" class="footnote-backref">↩︎</a> <a href="#fnref4:7" class="footnote-backref">↩︎</a> <a href="#fnref4:8" class="footnote-backref">↩︎</a> <a href="#fnref4:9" class="footnote-backref">↩︎</a> <a href="#fnref4:10" class="footnote-backref">↩︎</a></p></li><li id="fn5" class="footnote-item"><p><a href="https://www.databricks.com/blog/2015/04/28/project-tungsten-bringing-spark-closer-to-bare-metal.html" target="_blank" rel="noreferrer">Project Tungsten: Bringing Spark Closer to Bare Metal</a> <a href="#fnref5" class="footnote-backref">↩︎</a> <a href="#fnref5:1" class="footnote-backref">↩︎</a></p></li><li id="fn6" class="footnote-item"><p><em>Learning Spark, 2nd Edition</em>, Damji, Wenig, Das, Lee, ch. 6–7 <a href="#fnref6" class="footnote-backref">↩︎</a> <a href="#fnref6:1" class="footnote-backref">↩︎</a> <a href="#fnref6:2" class="footnote-backref">↩︎</a></p></li><li id="fn7" class="footnote-item"><p><a href="https://luminousmen.com/post/explaining-the-mechanics-of-spark-caching" target="_blank" rel="noreferrer">Explaining the Mechanics of Spark Caching</a> <a href="#fnref7" class="footnote-backref">↩︎</a> <a href="#fnref7:1" class="footnote-backref">↩︎</a> <a href="#fnref7:2" class="footnote-backref">↩︎</a> <a href="#fnref7:3" class="footnote-backref">↩︎</a> <a href="#fnref7:4" class="footnote-backref">↩︎</a></p></li><li id="fn8" class="footnote-item"><p><a href="https://spark.apache.org/docs/latest/rdd-programming-guide.html" target="_blank" rel="noreferrer">RDD Programming Guide</a> <a href="#fnref8" class="footnote-backref">↩︎</a></p></li><li id="fn9" class="footnote-item"><p><em>High Performance Spark, 2nd Edition</em>, Karau, Polak &amp; Warren, ch. 7 <a href="#fnref9" class="footnote-backref">↩︎</a> <a href="#fnref9:1" class="footnote-backref">↩︎</a> <a href="#fnref9:2" class="footnote-backref">↩︎</a> <a href="#fnref9:3" class="footnote-backref">↩︎</a></p></li><li id="fn10" class="footnote-item"><p><a href="https://raw.githubusercontent.com/apache/spark/v3.5.0/sql/catalyst/src/main/scala/org/apache/spark/sql/internal/SQLConf.scala" target="_blank" rel="noreferrer">SQLConf.scala</a> <a href="#fnref10" class="footnote-backref">↩︎</a> <a href="#fnref10:1" class="footnote-backref">↩︎</a></p></li><li id="fn11" class="footnote-item"><p><a href="https://spark.apache.org/docs/latest/monitoring.html#spark-history-server" target="_blank" rel="noreferrer">Monitoring and Instrumentation</a> <a href="#fnref11" class="footnote-backref">↩︎</a> <a href="#fnref11:1" class="footnote-backref">↩︎</a></p></li></ol></section>',34)])])}const k=o(c,[["render",l]]);export{y as __pageData,k as default};
@@ -0,0 +1 @@
1
+ import{_ as o,o as a,c as r,a5 as t}from"./chunks/framework.DSg0KOwT.js";const s="../assets/memory-regions.XHvO7jHG.svg",n="../assets/memory-regions.dark.D4TP9_08.svg",f="../assets/memory-borrowing.BqQRJg0u.svg",i="../assets/memory-borrowing.dark.Yhh20O9C.svg",y=JSON.parse('{"title":"Memory Management","description":"","frontmatter":{"title":"Memory Management"},"headers":[],"relativePath":"tuning-reference/memory-model.md","filePath":"tuning-reference/memory-model.md"}'),c={name:"tuning-reference/memory-model.md"};function l(h,e,d,p,u,m){return a(),r("div",null,[...e[0]||(e[0]=[t("",34)])])}const k=o(c,[["render",l]]);export{y as __pageData,k as default};
@@ -0,0 +1 @@
1
+ import{_ as t,o as a,c as r,a5 as s}from"./chunks/framework.DSg0KOwT.js";const u=JSON.parse('{"title":"Metrics Glossary","description":"","frontmatter":{"title":"Metrics Glossary"},"headers":[],"relativePath":"tuning-reference/metrics.md","filePath":"tuning-reference/metrics.md"}'),o={name:"tuning-reference/metrics.md"};function i(c,e,n,l,d,f){return a(),r("div",null,[...e[0]||(e[0]=[s('<h1 id="metrics" tabindex="-1">Metrics Glossary <a class="header-anchor" href="#metrics" aria-label="Permalink to &quot;Metrics Glossary {#metrics}&quot;">​</a></h1><p>Reference for every metric surfaced elsewhere in this guide: what it measures, where Spark records it in the event log, and what a problematic value looks like.</p><h2 id="metric-task-duration" tabindex="-1">Task duration (P50/P95/max) <a class="header-anchor" href="#metric-task-duration" aria-label="Permalink to &quot;Task duration (P50/P95/max) {#metric-task-duration}&quot;">​</a></h2><p>Per-task wall-clock duration, aggregated across a stage&#39;s tasks into percentiles (P50, P95, max) to surface skew and stragglers. It is computed from <code>SparkListenerTaskEnd</code> events rather than read off a single field: each task end carries its own duration, and the percentiles are derived by aggregating those values across the stage.</p><p>A problematic value looks like a small fraction of tasks running far past the rest: a stage is classified as a straggler when more than 5% of tasks run at least 4x the median duration (with at least 10 tasks in the stage), or when speculative tasks were fired.</p><h2 id="metric-shuffle-read-bytes" tabindex="-1">Shuffle read bytes <a class="header-anchor" href="#metric-shuffle-read-bytes" aria-label="Permalink to &quot;Shuffle read bytes {#metric-shuffle-read-bytes}&quot;">​</a></h2><p>The volume of shuffle data a task reads from remote executors, recorded in <code>taskMetrics.shuffleReadMetrics.remoteBytesRead</code>.</p><h2 id="metric-shuffle-write-bytes" tabindex="-1">Shuffle write bytes <a class="header-anchor" href="#metric-shuffle-write-bytes" aria-label="Permalink to &quot;Shuffle write bytes {#metric-shuffle-write-bytes}&quot;">​</a></h2><p>The size of the shuffle output a task writes, recorded in <code>taskMetrics.shuffleWriteMetrics.bytesWritten</code>. Spark&#39;s own metric description calls it simply the &quot;Number of bytes written in shuffle operations,&quot; without stating compression state on its own<sup class="footnote-ref"><a href="#fn1" id="fnref1">[1]</a></sup>.</p><p>The sort-based shuffle writer fills in the mechanics: incoming records are serialized as soon as they reach the shuffle writer and buffered in serialized form while sorting<sup class="footnote-ref"><a href="#fn2" id="fnref2">[2]</a></sup>. When the spill compression codec supports concatenating compressed data, the final merge step concatenates the already-compressed spill partitions directly into the output file, using <code>transferTo</code> rather than decompressing and recompressing<sup class="footnote-ref"><a href="#fn2" id="fnref2:1">[2:1]</a></sup>. So <code>bytesWritten</code> counts compressed, post-serialization bytes, and in this common fast-merge path the final written output is built directly from data that had already been spilled to disk during sorting, rather than written fresh at merge time.</p><h2 id="metric-memory-bytes-spilled" tabindex="-1">Memory bytes spilled <a class="header-anchor" href="#metric-memory-bytes-spilled" aria-label="Permalink to &quot;Memory bytes spilled {#metric-memory-bytes-spilled}&quot;">​</a></h2><p>Bytes a task spilled from in-memory structures, recorded in <code>taskMetrics.memoryBytesSpilled</code>. Any value above zero is a spill warning, with the skew-vs-volume classification (based on what fraction of a stage&#39;s tasks show zero spill) as the actionable signal.</p><h2 id="metric-disk-bytes-spilled" tabindex="-1">Disk bytes spilled <a class="header-anchor" href="#metric-disk-bytes-spilled" aria-label="Permalink to &quot;Disk bytes spilled {#metric-disk-bytes-spilled}&quot;">​</a></h2><p>The on-disk counterpart to memory spill: bytes a task spilled to disk, recorded in <code>taskMetrics.diskBytesSpilled</code>. It feeds the same spill classification as memory bytes spilled.</p><h2 id="metric-jvm-gc-time" tabindex="-1">JVM GC time <a class="header-anchor" href="#metric-jvm-gc-time" aria-label="Permalink to &quot;JVM GC time {#metric-jvm-gc-time}&quot;">​</a></h2><p>Elapsed time the JVM spent in garbage collection while a task executed, recorded in <code>taskMetrics.jvmGCTime</code> and expressed in milliseconds<sup class="footnote-ref"><a href="#fn1" id="fnref1:1">[1:1]</a></sup>. The value is cumulative across the task&#39;s full execution window: the sum of every GC pause that occurred during that task&#39;s run, not just the most recent one<sup class="footnote-ref"><a href="#fn1" id="fnref1:2">[1:2]</a></sup>. This matches how it&#39;s serialized: as a single scalar long, consistent with an accumulator rather than a per-GC-event log entry<sup class="footnote-ref"><a href="#fn3" id="fnref3">[3]</a></sup>.</p><h2 id="metric-gcpct" tabindex="-1">gcPct <a class="header-anchor" href="#metric-gcpct" aria-label="Permalink to &quot;gcPct {#metric-gcpct}&quot;">​</a></h2><p>A synthetic ratio, <code>jvmGCTime / executorRunTime</code>, not a raw Spark field. It drives GC-bottleneck classification: above 10% is a warning, above 20% is critical.</p><h2 id="metric-executor-run-time" tabindex="-1">Executor run time <a class="header-anchor" href="#metric-executor-run-time" aria-label="Permalink to &quot;Executor run time {#metric-executor-run-time}&quot;">​</a></h2><p>Elapsed time the executor spent running a task, recorded in <code>taskMetrics.executorRunTime</code> and expressed in milliseconds<sup class="footnote-ref"><a href="#fn1" id="fnref1:3">[1:3]</a></sup>. <code>TaskMetrics</code> (and therefore <code>executorRunTime</code>) is serialized as an optional part of the <code>SparkListenerTaskEnd</code> payload, not guaranteed on every task end<sup class="footnote-ref"><a href="#fn3" id="fnref3:1">[3:1]</a></sup>. In practice it is available for failed tasks that got far enough to actually run, such as <code>ExceptionFailure</code> or <code>TaskKilled</code>. The event-log deserialization code even falls back to reading accumulator updates out of the embedded <code>TaskMetrics</code> for old, Spark-1.x-era logs, which only makes sense if the metrics object is normally populated for that failure type<sup class="footnote-ref"><a href="#fn3" id="fnref3:2">[3:2]</a></sup>. It can be absent for reasons like <code>Resubmitted</code>, where the task attempt never completed on that executor<sup class="footnote-ref"><a href="#fn3" id="fnref3:3">[3:3]</a></sup>.</p><h2 id="metric-fetch-wait-time-ratio" tabindex="-1">Fetch wait time ratio <a class="header-anchor" href="#metric-fetch-wait-time-ratio" aria-label="Permalink to &quot;Fetch wait time ratio {#metric-fetch-wait-time-ratio}&quot;">​</a></h2><p>A synthetic ratio, <code>fetchWaitTime / taskDuration</code>, not a raw Spark field: the time a task spent blocked waiting on remote shuffle blocks, relative to its total duration.</p><h2 id="metric-input-bytes" tabindex="-1">Input bytes <a class="header-anchor" href="#metric-input-bytes" aria-label="Permalink to &quot;Input bytes {#metric-input-bytes}&quot;">​</a></h2><p>Bytes a task read as input, recorded in <code>taskMetrics.inputMetrics.bytesRead</code>.</p><h2 id="metric-output-bytes" tabindex="-1">Output bytes <a class="header-anchor" href="#metric-output-bytes" aria-label="Permalink to &quot;Output bytes {#metric-output-bytes}&quot;">​</a></h2><p>Bytes a task wrote as output, recorded in <code>taskMetrics.outputMetrics.bytesWritten</code>.</p><h2 id="metric-io-ratio" tabindex="-1">I/O ratio <a class="header-anchor" href="#metric-io-ratio" aria-label="Permalink to &quot;I/O ratio {#metric-io-ratio}&quot;">​</a></h2><p>A synthetic ratio, <code>outputBytes / inputBytes</code>, not a raw Spark field.</p><h2 id="metric-peak-execution-memory" tabindex="-1">Peak execution memory <a class="header-anchor" href="#metric-peak-execution-memory" aria-label="Permalink to &quot;Peak execution memory {#metric-peak-execution-memory}&quot;">​</a></h2><p>Peak memory recorded in <code>taskMetrics.peakExecutionMemory</code>. This is a task-level accumulator, distinct from the separate executor-level <code>peakMemoryMetrics.OnHeapExecutionMemory</code> and <code>.OffHeapExecutionMemory</code> gauges, which report the on-heap and off-heap execution pools as two separate numbers at the executor level rather than as a single per-task figure<sup class="footnote-ref"><a href="#fn1" id="fnref1:4">[1:4]</a></sup>.</p><h2 id="metric-failed-tasks" tabindex="-1">Failed tasks / failure rate <a class="header-anchor" href="#metric-failed-tasks" aria-label="Permalink to &quot;Failed tasks / failure rate {#metric-failed-tasks}&quot;">​</a></h2><p>Tasks whose <code>SparkListenerTaskEnd</code> reason is not <code>Success</code>. The <code>Reason</code> field holds the formatted class name of whichever <code>TaskEndReason</code> was assigned to that task end<sup class="footnote-ref"><a href="#fn3" id="fnref3:4">[3:4]</a></sup>, and the canonical set of reasons includes <code>FetchFailed</code>, <code>ExceptionFailure</code>, <code>TaskResultLost</code>, <code>TaskKilled</code>, <code>TaskCommitDenied</code>, <code>ExecutorLostFailure</code>, and <code>UnknownReason</code><sup class="footnote-ref"><a href="#fn3" id="fnref3:5">[3:5]</a></sup>. Because <code>TaskMetrics</code> is only an optional part of the task-end payload, it can be absent for some of these reasons: for example <code>Resubmitted</code>, where the task attempt never actually completed on that executor<sup class="footnote-ref"><a href="#fn3" id="fnref3:6">[3:6]</a></sup>.</p><h2 id="metric-speculative-tasks" tabindex="-1">Speculative tasks / straggler count <a class="header-anchor" href="#metric-speculative-tasks" aria-label="Permalink to &quot;Speculative tasks / straggler count {#metric-speculative-tasks}&quot;">​</a></h2><p>Tasks launched as speculative retries of a slow-running task, recorded via <code>SparkListenerTaskStart</code> where <code>speculative = true</code>. Any speculative task firing (or more than 5% of a stage&#39;s tasks running at least 4x the median duration, with at least 10 tasks in the stage) is a straggler signal.</p><h2 id="metric-executor-count" tabindex="-1">Executor count (added/removed/concurrent) <a class="header-anchor" href="#metric-executor-count" aria-label="Permalink to &quot;Executor count (added/removed/concurrent) {#metric-executor-count}&quot;">​</a></h2><p>The number of executors added, removed, or concurrently running, recorded via <code>SparkListenerExecutorAdded</code> / <code>SparkListenerExecutorRemoved</code> events.</p><h2 id="metric-stage-duration" tabindex="-1">Stage wall-clock duration <a class="header-anchor" href="#metric-stage-duration" aria-label="Permalink to &quot;Stage wall-clock duration {#metric-stage-duration}&quot;">​</a></h2><p>A stage&#39;s total elapsed time, recorded from <code>SparkListenerStageCompleted</code>: <code>completionTime − submissionTime</code>.</p><h2 id="metric-first-stage-submitted-at" tabindex="-1">firstStageSubmittedAt <a class="header-anchor" href="#metric-first-stage-submitted-at" aria-label="Permalink to &quot;firstStageSubmittedAt {#metric-first-stage-submitted-at}&quot;">​</a></h2><p>A synthetic field: the timestamp of the first <code>SparkListenerStageSubmitted</code> event in the event log. The gap between this timestamp and the application&#39;s start time drives cold-start classification: more than 30 seconds is a warning.</p><h2 id="sources" tabindex="-1">Sources <a class="header-anchor" href="#sources" aria-label="Permalink to &quot;Sources&quot;">​</a></h2><hr class="footnotes-sep"><section class="footnotes"><ol class="footnotes-list"><li id="fn1" class="footnote-item"><p><a href="https://spark.apache.org/docs/latest/monitoring.html#spark-history-server" target="_blank" rel="noreferrer">Monitoring and Instrumentation</a> <a href="#fnref1" class="footnote-backref">↩︎</a> <a href="#fnref1:1" class="footnote-backref">↩︎</a> <a href="#fnref1:2" class="footnote-backref">↩︎</a> <a href="#fnref1:3" class="footnote-backref">↩︎</a> <a href="#fnref1:4" class="footnote-backref">↩︎</a></p></li><li id="fn2" class="footnote-item"><p><a href="https://raw.githubusercontent.com/apache/spark/v3.5.0/core/src/main/scala/org/apache/spark/shuffle/sort/SortShuffleManager.scala" target="_blank" rel="noreferrer">SortShuffleManager.scala</a> <a href="#fnref2" class="footnote-backref">↩︎</a> <a href="#fnref2:1" class="footnote-backref">↩︎</a></p></li><li id="fn3" class="footnote-item"><p><a href="https://raw.githubusercontent.com/apache/spark/v3.5.0/core/src/main/scala/org/apache/spark/util/JsonProtocol.scala" target="_blank" rel="noreferrer">JsonProtocol.scala</a> <a href="#fnref3" class="footnote-backref">↩︎</a> <a href="#fnref3:1" class="footnote-backref">↩︎</a> <a href="#fnref3:2" class="footnote-backref">↩︎</a> <a href="#fnref3:3" class="footnote-backref">↩︎</a> <a href="#fnref3:4" class="footnote-backref">↩︎</a> <a href="#fnref3:5" class="footnote-backref">↩︎</a> <a href="#fnref3:6" class="footnote-backref">↩︎</a></p></li></ol></section>',43)])])}const m=t(o,[["render",i]]);export{u as __pageData,m as default};
@@ -0,0 +1 @@
1
+ import{_ as t,o as a,c as r,a5 as s}from"./chunks/framework.DSg0KOwT.js";const u=JSON.parse('{"title":"Metrics Glossary","description":"","frontmatter":{"title":"Metrics Glossary"},"headers":[],"relativePath":"tuning-reference/metrics.md","filePath":"tuning-reference/metrics.md"}'),o={name:"tuning-reference/metrics.md"};function i(c,e,n,l,d,f){return a(),r("div",null,[...e[0]||(e[0]=[s("",43)])])}const m=t(o,[["render",i]]);export{u as __pageData,m as default};
@@ -0,0 +1 @@
1
+ import{_ as a,o as t,c as o,a5 as r}from"./chunks/framework.DSg0KOwT.js";const s="../assets/repartition-vs-coalesce.BovLRrpj.svg",n="../assets/repartition-vs-coalesce.dark.BhAczKZQ.svg",g=JSON.parse('{"title":"Partitioning","description":"","frontmatter":{"title":"Partitioning"},"headers":[],"relativePath":"tuning-reference/partitioning.md","filePath":"tuning-reference/partitioning.md"}'),i={name:"tuning-reference/partitioning.md"};function f(l,e,c,h,d,p){return t(),o("div",null,[...e[0]||(e[0]=[r('<h1 id="partitioning" tabindex="-1">Partitioning <a class="header-anchor" href="#partitioning" aria-label="Permalink to &quot;Partitioning {#partitioning}&quot;">​</a></h1><h2 id="how-partitions-get-sized-and-shuffled" tabindex="-1">How partitions get sized and shuffled <a class="header-anchor" href="#how-partitions-get-sized-and-shuffled" aria-label="Permalink to &quot;How partitions get sized and shuffled&quot;">​</a></h2><p>Every partition Spark creates maps to exactly one task and one thread, which is why partition count and size drive so much of a job&#39;s performance<sup class="footnote-ref"><a href="#fn1" id="fnref1">[1]</a></sup>. The <a href="./shuffle.html">shuffle</a> side of that is governed by a single default: <code>spark.sql.shuffle.partitions</code>, which defaults to 200 and applies to every <code>join()</code>, <code>groupBy()</code>, and aggregation regardless of how much data is actually moving<sup class="footnote-ref"><a href="#fn2" id="fnref2">[2]</a></sup><sup class="footnote-ref"><a href="#fn3" id="fnref3">[3]</a></sup>.</p><p>Two operations reshape that layout, and they are not interchangeable. <code>repartition(numPartitions)</code> &quot;return[s] a new RDD that has exactly numPartitions partitions&quot;<sup class="footnote-ref"><a href="#fn4" id="fnref4">[4]</a></sup>, and it earns that guarantee by always running a full hash shuffle: records are first spread across a temporary key space starting from a randomized position, seeded per-partition via <code>XORShiftRandom</code>, so upstream data ends up distributed evenly rather than clustered by its original layout. That feeds into a <code>ShuffledRDD</code> keyed by a <code>HashPartitioner(numPartitions)</code>, then a <code>CoalescedRDD</code> lands it on exactly the requested count<sup class="footnote-ref"><a href="#fn4" id="fnref4:1">[4:1]</a></sup>. High Performance Spark puts the same mechanics more plainly: &quot;repartition shuffles the RDD with a hash partitioner and the given number of partitions&quot;<sup class="footnote-ref"><a href="#fn5" id="fnref5">[5]</a></sup>. That holds whether the new count is bigger or smaller than the current one. Repartition always performs the full shuffle just described, unlike coalesce<sup class="footnote-ref"><a href="#fn6" id="fnref6">[6]</a></sup>.</p><p><code>coalesce</code>, left at its default, does not shuffle at all: it&#39;s a narrow transformation where each output partition is simply the union of a fixed set of parent partitions decided at plan time, not routed by data values, so the stage&#39;s task count just drops to the coalesced number<sup class="footnote-ref"><a href="#fn5" id="fnref5:1">[5:1]</a></sup>.</p><p>Neither <code>repartition</code> nor <code>coalesce</code> leaves behind a &quot;known partitioner&quot; the way <code>partitionBy</code> does<sup class="footnote-ref"><a href="#fn5" id="fnref5:2">[5:2]</a></sup>. That distinction is what Spark&#39;s join planner checks before it will skip a shuffle: it only does so when both sides already carry partitioner objects it can prove equal: matching partition counts for a <code>HashPartitioner</code>, matching range bounds for a <code>RangePartitioner</code><sup class="footnote-ref"><a href="#fn5" id="fnref5:3">[5:3]</a></sup>. Its default shuffle hash join path makes the same point from the other side: it partitions the second dataset using the same partitioner as the first specifically so matching keys land together, which only lets a later shuffle be skipped when a shared, recognized partitioner is already in place beforehand<sup class="footnote-ref"><a href="#fn5" id="fnref5:4">[5:4]</a></sup>.</p><p>There&#39;s also a hard ceiling on shuffle output, though it isn&#39;t something to design around: Spark&#39;s sort-based shuffle manager caps output partitions in serialized mode at <code>PackedRecordPointer.MAXIMUM_PARTITION_ID + 1</code>, roughly 16.8 million. The source code itself calls this &quot;an extreme defensive programming measure,&quot; since no real shuffle comes remotely close to it<sup class="footnote-ref"><a href="#fn7" id="fnref7">[7]</a></sup>.</p><h2 id="spotting-a-bad-layout" tabindex="-1">Spotting a bad layout <a class="header-anchor" href="#spotting-a-bad-layout" aria-label="Permalink to &quot;Spotting a bad layout&quot;">​</a></h2><p>That layout choice shows up as overhead in both directions once it&#39;s wrong: partitions that are too small flood the cluster with per-task scheduling overhead, while partitions that are too large create memory pressure and straggler tasks<sup class="footnote-ref"><a href="#fn1" id="fnref1:1">[1:1]</a></sup>. Learning Spark 2nd Edition frames the healthy target from the parallelism side rather than an absolute number: at least as many partitions as there are cores across the executors, so no core sits idle; more partitions than cores is fine as long as it doesn&#39;t drift into the small-partition overhead regime above<sup class="footnote-ref"><a href="#fn3" id="fnref3:1">[3:1]</a></sup>.</p><p>A heavy filter is an easy-to-miss cause of imbalance: Spark doesn&#39;t shrink partition count when rows are filtered out, so 2,000 partitions holding 5% of the original rows just become 2,000 mostly-empty partitions<sup class="footnote-ref"><a href="#fn1" id="fnref1:2">[1:2]</a></sup>.</p><p>To size shuffle partitions from a job&#39;s actual behavior rather than a guess, Cloudera&#39;s tuning guide takes a stage that already ran, computes the ratio between its Shuffle Spill (Memory) and Shuffle Spill (Disk) metrics, and multiplies total shuffle write by that ratio to estimate in-memory shuffle size, then rounds the resulting partition count up rather than down<sup class="footnote-ref"><a href="#fn8" id="fnref8">[8]</a></sup>.</p><p><a href="./bottleneck-skew.html">Skew</a> specifically has documented, numeric detection thresholds under <a href="./aqe.html">Adaptive Query Execution</a>: a partition counts as skewed if it&#39;s larger than <code>spark.sql.adaptive.skewJoin.skewedPartitionFactor</code> (default 5.0) times the median partition size, and also larger than <code>spark.sql.adaptive.skewJoin.skewedPartitionThresholdInBytes</code> (default 256 MB)<sup class="footnote-ref"><a href="#fn2" id="fnref2:1">[2:1]</a></sup><sup class="footnote-ref"><a href="#fn9" id="fnref9">[9]</a></sup>.</p><p>On the input side, <code>spark.sql.files.maxPartitionBytes</code> (defaulting to 128 MB) governs the target chunk size when Spark splits splittable file-based sources (CSV, JSON, line-delimited text) into input partitions, based on file size<sup class="footnote-ref"><a href="#fn1" id="fnref1:3">[1:3]</a></sup>.</p><h2 id="where-a-bad-layout-costs-you" tabindex="-1">Where a bad layout costs you <a class="header-anchor" href="#where-a-bad-layout-costs-you" aria-label="Permalink to &quot;Where a bad layout costs you&quot;">​</a></h2><p>Getting the partition count wrong costs more than the immediate slowdown suggests. Cloudera&#39;s tuning guide argues it&#39;s safer to over-provision partitions than to under-provision them, because Spark, unlike MapReduce, has low per-task startup overhead: &quot;when in doubt, it&#39;s almost always better to err on the side of a larger number of tasks&quot;<sup class="footnote-ref"><a href="#fn8" id="fnref8:1">[8:1]</a></sup>.</p><p>Getting <code>coalesce</code> wrong costs more than the coalesce step itself: because it&#39;s narrow, it forces the <em>entire</em> upstream stage to run at the reduced parallelism, not just the final step<sup class="footnote-ref"><a href="#fn5" id="fnref5:5">[5:5]</a></sup>. Pushed too far (<code>coalesce(1)</code>), it kills parallelism outright, because coalesce doesn&#39;t rebalance data, it just stacks existing partitions together, so partitions that were uneven going in are still uneven coming out<sup class="footnote-ref"><a href="#fn1" id="fnref1:4">[1:4]</a></sup>.</p><p>Once a shuffle stage has run, its output file count is locked in: you can&#39;t change it after the fact without inserting a stage barrier, such as writing to temporary storage or calling <code>localCheckpoint()</code>, between the shuffle and the write<sup class="footnote-ref"><a href="#fn10" id="fnref10">[10]</a></sup>.</p><p>Join alignment has its own gotcha: calling <code>repartition(n, col)</code> with the same column and count on two separate DataFrames produces data that&#39;s plausibly laid out the same way, but it doesn&#39;t leave a trackable partitioner object behind. The planner has no recorded partitioner to compare, so it has no basis for treating the two sides as co-partitioned and skipping the shuffle at join time, even though the call sites look identical<sup class="footnote-ref"><a href="#fn5" id="fnref5:6">[5:6]</a></sup>.</p><h2 id="fixing-the-layout" tabindex="-1">Fixing the layout <a class="header-anchor" href="#fixing-the-layout" aria-label="Permalink to &quot;Fixing the layout&quot;">​</a></h2><p>Fixing those costs starts with accepting there&#39;s no formula to look up. There&#39;s no single formula for the right <code>spark.sql.shuffle.partitions</code> value: it depends on data set size, core count, and executor memory, and comes down to trial and error<sup class="footnote-ref"><a href="#fn3" id="fnref3:2">[3:2]</a></sup>. As a starting point, Learning Spark notes the default of 200 is usually too high for small or streaming workloads, where it should be pulled down toward the executor core count<sup class="footnote-ref"><a href="#fn3" id="fnref3:3">[3:3]</a></sup>.</p><p>For the repartition-vs-coalesce decision itself: reach for <code>coalesce</code> first whenever you&#39;re only reducing partition count, since it merges partitions already on the same node without a shuffle<sup class="footnote-ref"><a href="#fn6" id="fnref6:1">[6:1]</a></sup>. Reach for <code>repartition</code> when you actually need the shuffle: increasing partition count, fixing a lopsided distribution, or preparing a DataFrame ahead of a join or a <code>cache()</code> call, where even parallelism is worth more than the shuffle cost<sup class="footnote-ref"><a href="#fn6" id="fnref6:2">[6:2]</a></sup>. High Performance Spark reduces this to a readability rule: use <code>repartition</code> when you want a shuffle, <code>coalesce</code> when you don&#39;t, rather than leaning on coalesce&#39;s shuffle toggle to blur the line<sup class="footnote-ref"><a href="#fn5" id="fnref5:7">[5:7]</a></sup>. There is a middle option, <code>coalesce(n, shuffle=True)</code>, which behaves more like repartition, paying for a shuffle but getting real rebalancing on the way down<sup class="footnote-ref"><a href="#fn1" id="fnref1:5">[1:5]</a></sup>. <code>coalesce()</code> itself belongs at the tail of a pipeline, right before a write, purely to cut down <a href="./bottleneck-small-files.html">output file count</a><sup class="footnote-ref"><a href="#fn1" id="fnref1:6">[1:6]</a></sup>; after a heavy filter has left partitions mostly empty, following up with <code>repartition(100)</code> (or similar) buys back real parallelism<sup class="footnote-ref"><a href="#fn1" id="fnref1:7">[1:7]</a></sup>. At the SQL layer, both directions have dedicated hints for controlling output partitioning directly: <code>/*+ REPARTITION(n) */</code>, <code>/*+ REPARTITION(cols) */</code>, <code>/*+ REPARTITION_BY_RANGE(cols) */</code>, and <code>/*+ COALESCE(n) */</code><sup class="footnote-ref"><a href="#fn9" id="fnref9:1">[9:1]</a></sup>.</p><img class="light-only" src="'+s+'" alt="A decision flowchart for choosing repartition versus coalesce based on whether a shuffle and a rebalance are needed."><img class="dark-only" src="'+n+'" alt="A decision flowchart for choosing repartition versus coalesce based on whether a shuffle and a rebalance are needed."><p>Separate from the DataFrame <code>.coalesce()</code> call, Adaptive Query Execution has its own runtime coalescing behavior: <code>spark.sql.adaptive.coalescePartitions.enabled</code> (default true) merges small, contiguous post-shuffle partitions toward a target size at runtime, correcting over-partitioning without a manual <code>.coalesce()</code> call<sup class="footnote-ref"><a href="#fn9" id="fnref9:2">[9:2]</a></sup><sup class="footnote-ref"><a href="#fn1" id="fnref1:8">[1:8]</a></sup>.</p><p>For skewed joins specifically, salting (adding a prefix to skewed keys so the same key is treated as several different keys, then adjusting the data distribution accordingly) was one of three manual approaches used before adaptive execution existed, alongside raising <code>spark.sql.shuffle.partitions</code> and raising the broadcast hash join threshold to push a sort-merge join toward a broadcast hash join instead. All three carry &quot;lots of limitations&quot; and require manual processing, which is exactly the gap AQE&#39;s skew-join optimization was built to close<sup class="footnote-ref"><a href="#fn11" id="fnref11">[11]</a></sup>. Where automatic detection isn&#39;t precise enough, Databricks&#39; <code>/*+ SKEW(...) */</code> hint names the skewed relation and column(s), and optionally the specific skewed key values, letting the planner target just those keys directly instead of relying on automatic detection<sup class="footnote-ref"><a href="#fn12" id="fnref12">[12]</a></sup>.</p><p>For the join-alignment gotcha above, the mechanism that does reliably guarantee a shuffle can be skipped is storage-partitioned joins: both tables are physically bucketed identically at the catalog level, for example <a href="./table-formats.html">Iceberg</a> tables created with matching <code>PARTITIONED BY (bucket(...))</code> clauses. When Spark recognizes both sides report the same partitioning through <code>SupportsReportPartitioning</code>, it can drop the Exchange (shuffle) node entirely, or shuffle only one side. A plain DataFrame-level <code>repartition(col)</code> doesn&#39;t offer that guarantee, because it isn&#39;t backed by catalog-level partitioning metadata<sup class="footnote-ref"><a href="#fn9" id="fnref9:3">[9:3]</a></sup><sup class="footnote-ref"><a href="#fn2" id="fnref2:2">[2:2]</a></sup>.</p><h2 id="sources" tabindex="-1">Sources <a class="header-anchor" href="#sources" aria-label="Permalink to &quot;Sources&quot;">​</a></h2><hr class="footnotes-sep"><section class="footnotes"><ol class="footnotes-list"><li id="fn1" class="footnote-item"><p><a href="https://luminousmen.com/post/spark-partitions" target="_blank" rel="noreferrer">Spark Partitions</a> <a href="#fnref1" class="footnote-backref">↩︎</a> <a href="#fnref1:1" class="footnote-backref">↩︎</a> <a href="#fnref1:2" class="footnote-backref">↩︎</a> <a href="#fnref1:3" class="footnote-backref">↩︎</a> <a href="#fnref1:4" class="footnote-backref">↩︎</a> <a href="#fnref1:5" class="footnote-backref">↩︎</a> <a href="#fnref1:6" class="footnote-backref">↩︎</a> <a href="#fnref1:7" class="footnote-backref">↩︎</a> <a href="#fnref1:8" class="footnote-backref">↩︎</a></p></li><li id="fn2" class="footnote-item"><p><a href="https://spark.apache.org/docs/latest/configuration.html" target="_blank" rel="noreferrer">Configuration — Spark</a> <a href="#fnref2" class="footnote-backref">↩︎</a> <a href="#fnref2:1" class="footnote-backref">↩︎</a> <a href="#fnref2:2" class="footnote-backref">↩︎</a></p></li><li id="fn3" class="footnote-item"><p><em>Learning Spark, 2nd Edition</em>, Damji, Wenig, Das &amp; Lee, ch. 7 <a href="#fnref3" class="footnote-backref">↩︎</a> <a href="#fnref3:1" class="footnote-backref">↩︎</a> <a href="#fnref3:2" class="footnote-backref">↩︎</a> <a href="#fnref3:3" class="footnote-backref">↩︎</a></p></li><li id="fn4" class="footnote-item"><p><a href="https://raw.githubusercontent.com/apache/spark/v3.5.0/core/src/main/scala/org/apache/spark/rdd/RDD.scala" target="_blank" rel="noreferrer">RDD.scala</a> <a href="#fnref4" class="footnote-backref">↩︎</a> <a href="#fnref4:1" class="footnote-backref">↩︎</a></p></li><li id="fn5" class="footnote-item"><p><em>High Performance Spark, 2nd Edition</em>, Karau, Polak &amp; Warren, ch. 8 <a href="#fnref5" class="footnote-backref">↩︎</a> <a href="#fnref5:1" class="footnote-backref">↩︎</a> <a href="#fnref5:2" class="footnote-backref">↩︎</a> <a href="#fnref5:3" class="footnote-backref">↩︎</a> <a href="#fnref5:4" class="footnote-backref">↩︎</a> <a href="#fnref5:5" class="footnote-backref">↩︎</a> <a href="#fnref5:6" class="footnote-backref">↩︎</a> <a href="#fnref5:7" class="footnote-backref">↩︎</a></p></li><li id="fn6" class="footnote-item"><p><em>Spark: The Definitive Guide</em>, Chambers &amp; Zaharia, ch. 19 <a href="#fnref6" class="footnote-backref">↩︎</a> <a href="#fnref6:1" class="footnote-backref">↩︎</a> <a href="#fnref6:2" class="footnote-backref">↩︎</a></p></li><li id="fn7" class="footnote-item"><p><a href="https://raw.githubusercontent.com/apache/spark/v3.5.0/core/src/main/scala/org/apache/spark/shuffle/sort/SortShuffleManager.scala" target="_blank" rel="noreferrer">SortShuffleManager.scala</a> <a href="#fnref7" class="footnote-backref">↩︎</a></p></li><li id="fn8" class="footnote-item"><p><a href="https://blog.cloudera.com/how-to-tune-your-apache-spark-jobs-part-2/" target="_blank" rel="noreferrer">How to Tune Your Apache Spark Jobs (Part 2)</a> <a href="#fnref8" class="footnote-backref">↩︎</a> <a href="#fnref8:1" class="footnote-backref">↩︎</a></p></li><li id="fn9" class="footnote-item"><p><a href="https://spark.apache.org/docs/latest/sql-performance-tuning.html" target="_blank" rel="noreferrer">Performance Tuning — Spark SQL, DataFrames and Datasets Guide</a> <a href="#fnref9" class="footnote-backref">↩︎</a> <a href="#fnref9:1" class="footnote-backref">↩︎</a> <a href="#fnref9:2" class="footnote-backref">↩︎</a> <a href="#fnref9:3" class="footnote-backref">↩︎</a></p></li><li id="fn10" class="footnote-item"><p><a href="https://luminousmen.com/post/spark-tips-partition-tuning" target="_blank" rel="noreferrer">Spark Tips: Partition Tuning</a> <a href="#fnref10" class="footnote-backref">↩︎</a></p></li><li id="fn11" class="footnote-item"><p><a href="https://issues.apache.org/jira/browse/SPARK-29544" target="_blank" rel="noreferrer">SPARK-29544 — Optimize skewed join at runtime</a> <a href="#fnref11" class="footnote-backref">↩︎</a></p></li><li id="fn12" class="footnote-item"><p><a href="https://docs.databricks.com/aws/en/archive/legacy/skew-join" target="_blank" rel="noreferrer">Skew Join (legacy)</a> <a href="#fnref12" class="footnote-backref">↩︎</a></p></li></ol></section>',29)])])}const m=a(i,[["render",f]]);export{g as __pageData,m as default};
@@ -0,0 +1 @@
1
+ import{_ as a,o as t,c as o,a5 as r}from"./chunks/framework.DSg0KOwT.js";const s="../assets/repartition-vs-coalesce.BovLRrpj.svg",n="../assets/repartition-vs-coalesce.dark.BhAczKZQ.svg",g=JSON.parse('{"title":"Partitioning","description":"","frontmatter":{"title":"Partitioning"},"headers":[],"relativePath":"tuning-reference/partitioning.md","filePath":"tuning-reference/partitioning.md"}'),i={name:"tuning-reference/partitioning.md"};function f(l,e,c,h,d,p){return t(),o("div",null,[...e[0]||(e[0]=[r("",29)])])}const m=a(i,[["render",f]]);export{g as __pageData,m as default};