sparkforensics-cli 0.1.0 → 0.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (360) hide show
  1. package/README.md +6 -0
  2. package/bin/sparkforensics-analyze.mjs +113 -48
  3. package/export-template/docs/404.html +25 -0
  4. package/export-template/docs/assets/app.CndaAS6v.js +1 -0
  5. package/export-template/docs/assets/aqe-loop.IwQSATHw.svg +1 -0
  6. package/export-template/docs/assets/aqe-loop.dark.DGbaxqJE.svg +1 -0
  7. package/export-template/docs/assets/broadcast-vs-shuffle.Db4WY1XK.svg +1 -0
  8. package/export-template/docs/assets/broadcast-vs-shuffle.dark.C7Bxs0mG.svg +1 -0
  9. package/export-template/docs/assets/cache-lifecycle.dark.B-hS7AgU.svg +1 -0
  10. package/export-template/docs/assets/cache-lifecycle.rEOVYQNU.svg +1 -0
  11. package/export-template/docs/assets/chunks/@localSearchIndexroot.DNY8bVcl.js +1 -0
  12. package/export-template/docs/assets/chunks/VPLocalSearchBox.yJbZbsEo.js +9 -0
  13. package/export-template/docs/assets/chunks/duplicate-plan-subtree.dark.Cdp70QhV.js +1 -0
  14. package/export-template/docs/assets/chunks/framework.DSg0KOwT.js +20 -0
  15. package/export-template/docs/assets/chunks/retry-escalation-ladder.dark.DHipdJgZ.js +1 -0
  16. package/export-template/docs/assets/chunks/theme.Df2VAG9w.js +2 -0
  17. package/export-template/docs/assets/cold-start-timeline.DxC_Sc7w.svg +1 -0
  18. package/export-template/docs/assets/cold-start-timeline.dark.CZ17YcAG.svg +1 -0
  19. package/export-template/docs/assets/columnar-layout.PghGeOEA.svg +1 -0
  20. package/export-template/docs/assets/columnar-layout.dark.BVNlz0ff.svg +1 -0
  21. package/export-template/docs/assets/container-memory.DIO0AnIm.svg +1 -0
  22. package/export-template/docs/assets/container-memory.dark.CP-5zuCl.svg +1 -0
  23. package/export-template/docs/assets/contributor-guide_architecture_board-widgets.md.B-OsL91z.js +1 -0
  24. package/export-template/docs/assets/contributor-guide_architecture_board-widgets.md.B-OsL91z.lean.js +1 -0
  25. package/export-template/docs/assets/contributor-guide_architecture_detector-contract.md.BOeH4d1J.js +1 -0
  26. package/export-template/docs/assets/contributor-guide_architecture_detector-contract.md.BOeH4d1J.lean.js +1 -0
  27. package/export-template/docs/assets/contributor-guide_architecture_drill-down.md.BtPdlM7r.js +1 -0
  28. package/export-template/docs/assets/contributor-guide_architecture_drill-down.md.BtPdlM7r.lean.js +1 -0
  29. package/export-template/docs/assets/contributor-guide_architecture_impact-estimation.md.DYCDPgkh.js +1 -0
  30. package/export-template/docs/assets/contributor-guide_architecture_impact-estimation.md.DYCDPgkh.lean.js +1 -0
  31. package/export-template/docs/assets/contributor-guide_architecture_index.md.3TO9ic6w.js +1 -0
  32. package/export-template/docs/assets/contributor-guide_architecture_index.md.3TO9ic6w.lean.js +1 -0
  33. package/export-template/docs/assets/contributor-guide_architecture_overview.md.CehiRmGn.js +1 -0
  34. package/export-template/docs/assets/contributor-guide_architecture_overview.md.CehiRmGn.lean.js +1 -0
  35. package/export-template/docs/assets/contributor-guide_architecture_state-and-history.md.m3S3UdMk.js +1 -0
  36. package/export-template/docs/assets/contributor-guide_architecture_state-and-history.md.m3S3UdMk.lean.js +1 -0
  37. package/export-template/docs/assets/contributor-guide_architecture_widget-rendering.md.DbqPf2OT.js +1 -0
  38. package/export-template/docs/assets/contributor-guide_architecture_widget-rendering.md.DbqPf2OT.lean.js +1 -0
  39. package/export-template/docs/assets/contributor-guide_architecture_worker-protocol.md.B93qJ_tT.js +6 -0
  40. package/export-template/docs/assets/contributor-guide_architecture_worker-protocol.md.B93qJ_tT.lean.js +1 -0
  41. package/export-template/docs/assets/contributor-guide_contributing.md.CvRsdr6J.js +1 -0
  42. package/export-template/docs/assets/contributor-guide_contributing.md.CvRsdr6J.lean.js +1 -0
  43. package/export-template/docs/assets/contributor-guide_development-setup.md.DvAN_9mK.js +12 -0
  44. package/export-template/docs/assets/contributor-guide_development-setup.md.DvAN_9mK.lean.js +1 -0
  45. package/export-template/docs/assets/contributor-guide_testing.md.6rIKqSyY.js +1 -0
  46. package/export-template/docs/assets/contributor-guide_testing.md.6rIKqSyY.lean.js +1 -0
  47. package/export-template/docs/assets/dag-stages.DSz_S937.svg +1 -0
  48. package/export-template/docs/assets/dag-stages.dark.F72UzxH4.svg +1 -0
  49. package/export-template/docs/assets/driver-executor.D5pQ7YN1.svg +1 -0
  50. package/export-template/docs/assets/driver-executor.dark.BmX9cPvh.svg +1 -0
  51. package/export-template/docs/assets/duplicate-plan-subtree.B4cvN6fj.svg +1 -0
  52. package/export-template/docs/assets/duplicate-plan-subtree.dark.Dw8wS0Ag.svg +1 -0
  53. package/export-template/docs/assets/index.md.CHJVslga.js +1 -0
  54. package/export-template/docs/assets/index.md.CHJVslga.lean.js +1 -0
  55. package/export-template/docs/assets/inter-italic-cyrillic-ext.r48I6akx.woff2 +0 -0
  56. package/export-template/docs/assets/inter-italic-cyrillic.By2_1cv3.woff2 +0 -0
  57. package/export-template/docs/assets/inter-italic-greek-ext.1u6EdAuj.woff2 +0 -0
  58. package/export-template/docs/assets/inter-italic-greek.DJ8dCoTZ.woff2 +0 -0
  59. package/export-template/docs/assets/inter-italic-latin-ext.CN1xVJS-.woff2 +0 -0
  60. package/export-template/docs/assets/inter-italic-latin.C2AdPX0b.woff2 +0 -0
  61. package/export-template/docs/assets/inter-italic-vietnamese.BSbpV94h.woff2 +0 -0
  62. package/export-template/docs/assets/inter-roman-cyrillic-ext.BBPuwvHQ.woff2 +0 -0
  63. package/export-template/docs/assets/inter-roman-cyrillic.C5lxZ8CY.woff2 +0 -0
  64. package/export-template/docs/assets/inter-roman-greek-ext.CqjqNYQ-.woff2 +0 -0
  65. package/export-template/docs/assets/inter-roman-greek.BBVDIX6e.woff2 +0 -0
  66. package/export-template/docs/assets/inter-roman-latin-ext.4ZJIpNVo.woff2 +0 -0
  67. package/export-template/docs/assets/inter-roman-latin.Di8DUHzh.woff2 +0 -0
  68. package/export-template/docs/assets/inter-roman-vietnamese.BjW4sHH5.woff2 +0 -0
  69. package/export-template/docs/assets/join-strategy.C_FvrCEo.svg +1 -0
  70. package/export-template/docs/assets/join-strategy.dark.ChMLnNII.svg +1 -0
  71. package/export-template/docs/assets/memory-borrowing.BqQRJg0u.svg +1 -0
  72. package/export-template/docs/assets/memory-borrowing.dark.Yhh20O9C.svg +1 -0
  73. package/export-template/docs/assets/memory-regions.XHvO7jHG.svg +1 -0
  74. package/export-template/docs/assets/memory-regions.dark.D4TP9_08.svg +1 -0
  75. package/export-template/docs/assets/repartition-vs-coalesce.BovLRrpj.svg +1 -0
  76. package/export-template/docs/assets/repartition-vs-coalesce.dark.BhAczKZQ.svg +1 -0
  77. package/export-template/docs/assets/retry-escalation-ladder.DyTKJJmZ.svg +1 -0
  78. package/export-template/docs/assets/retry-escalation-ladder.dark.BdsabtU3.svg +1 -0
  79. package/export-template/docs/assets/shuffle-map-reduce.KuOEZVmg.svg +1 -0
  80. package/export-template/docs/assets/shuffle-map-reduce.dark.BgQZnFSb.svg +1 -0
  81. package/export-template/docs/assets/spill-classification.BU2euYDO.svg +1 -0
  82. package/export-template/docs/assets/spill-classification.dark.D7i1M40d.svg +1 -0
  83. package/export-template/docs/assets/style.DXOMCXxn.css +1 -0
  84. package/export-template/docs/assets/tuning-reference_anti-patterns.md.Df1YMIHu.js +1 -0
  85. package/export-template/docs/assets/tuning-reference_anti-patterns.md.Df1YMIHu.lean.js +1 -0
  86. package/export-template/docs/assets/tuning-reference_aqe.md.BIsCtLzm.js +1 -0
  87. package/export-template/docs/assets/tuning-reference_aqe.md.BIsCtLzm.lean.js +1 -0
  88. package/export-template/docs/assets/tuning-reference_bottleneck-broadcast-sizing.md.CEstB3Ia.js +1 -0
  89. package/export-template/docs/assets/tuning-reference_bottleneck-broadcast-sizing.md.CEstB3Ia.lean.js +1 -0
  90. package/export-template/docs/assets/tuning-reference_bottleneck-cold-start.md.CEuy-72y.js +7 -0
  91. package/export-template/docs/assets/tuning-reference_bottleneck-cold-start.md.CEuy-72y.lean.js +1 -0
  92. package/export-template/docs/assets/tuning-reference_bottleneck-duplicate-plan-subtree.md.CIohQDfn.js +1 -0
  93. package/export-template/docs/assets/tuning-reference_bottleneck-duplicate-plan-subtree.md.CIohQDfn.lean.js +1 -0
  94. package/export-template/docs/assets/tuning-reference_bottleneck-failures.md.4z5BXGJ2.js +6 -0
  95. package/export-template/docs/assets/tuning-reference_bottleneck-failures.md.4z5BXGJ2.lean.js +1 -0
  96. package/export-template/docs/assets/tuning-reference_bottleneck-gc.md.DSxzZRK7.js +6 -0
  97. package/export-template/docs/assets/tuning-reference_bottleneck-gc.md.DSxzZRK7.lean.js +1 -0
  98. package/export-template/docs/assets/tuning-reference_bottleneck-job-failure-rate.md.BaJl__1W.js +8 -0
  99. package/export-template/docs/assets/tuning-reference_bottleneck-job-failure-rate.md.BaJl__1W.lean.js +1 -0
  100. package/export-template/docs/assets/tuning-reference_bottleneck-memory-utilization.md.DbP-SJZc.js +7 -0
  101. package/export-template/docs/assets/tuning-reference_bottleneck-memory-utilization.md.DbP-SJZc.lean.js +1 -0
  102. package/export-template/docs/assets/tuning-reference_bottleneck-retry-waste.md.D5JMjVOt.js +1 -0
  103. package/export-template/docs/assets/tuning-reference_bottleneck-retry-waste.md.D5JMjVOt.lean.js +1 -0
  104. package/export-template/docs/assets/tuning-reference_bottleneck-shuffle.md.CM-nTmIH.js +12 -0
  105. package/export-template/docs/assets/tuning-reference_bottleneck-shuffle.md.CM-nTmIH.lean.js +1 -0
  106. package/export-template/docs/assets/tuning-reference_bottleneck-skew.md.BdUwiDhn.js +14 -0
  107. package/export-template/docs/assets/tuning-reference_bottleneck-skew.md.BdUwiDhn.lean.js +1 -0
  108. package/export-template/docs/assets/tuning-reference_bottleneck-slow-host.md.BlIo6UDW.js +7 -0
  109. package/export-template/docs/assets/tuning-reference_bottleneck-slow-host.md.BlIo6UDW.lean.js +1 -0
  110. package/export-template/docs/assets/tuning-reference_bottleneck-small-files.md.B8kloyx8.js +5 -0
  111. package/export-template/docs/assets/tuning-reference_bottleneck-small-files.md.B8kloyx8.lean.js +1 -0
  112. package/export-template/docs/assets/tuning-reference_bottleneck-spill.md.PNH7mITt.js +6 -0
  113. package/export-template/docs/assets/tuning-reference_bottleneck-spill.md.PNH7mITt.lean.js +1 -0
  114. package/export-template/docs/assets/tuning-reference_bottleneck-straggler.md.DY36fHN5.js +7 -0
  115. package/export-template/docs/assets/tuning-reference_bottleneck-straggler.md.DY36fHN5.lean.js +1 -0
  116. package/export-template/docs/assets/tuning-reference_bottleneck-tiny-tasks.md.QTV7O8kU.js +8 -0
  117. package/export-template/docs/assets/tuning-reference_bottleneck-tiny-tasks.md.QTV7O8kU.lean.js +1 -0
  118. package/export-template/docs/assets/tuning-reference_bottleneck-utilization.md.DTiueZC3.js +5 -0
  119. package/export-template/docs/assets/tuning-reference_bottleneck-utilization.md.DTiueZC3.lean.js +1 -0
  120. package/export-template/docs/assets/tuning-reference_caching.md.B7aQ8asB.js +1 -0
  121. package/export-template/docs/assets/tuning-reference_caching.md.B7aQ8asB.lean.js +1 -0
  122. package/export-template/docs/assets/tuning-reference_cluster-config.md.ZVmDGsQ3.js +1 -0
  123. package/export-template/docs/assets/tuning-reference_cluster-config.md.ZVmDGsQ3.lean.js +1 -0
  124. package/export-template/docs/assets/tuning-reference_config.md.UvveiWG3.js +1 -0
  125. package/export-template/docs/assets/tuning-reference_config.md.UvveiWG3.lean.js +1 -0
  126. package/export-template/docs/assets/tuning-reference_data-formats.md.bjCAWH3N.js +1 -0
  127. package/export-template/docs/assets/tuning-reference_data-formats.md.bjCAWH3N.lean.js +1 -0
  128. package/export-template/docs/assets/tuning-reference_index.md.BQ_NooMV.js +1 -0
  129. package/export-template/docs/assets/tuning-reference_index.md.BQ_NooMV.lean.js +1 -0
  130. package/export-template/docs/assets/tuning-reference_intro.md.CobD-lGB.js +1 -0
  131. package/export-template/docs/assets/tuning-reference_intro.md.CobD-lGB.lean.js +1 -0
  132. package/export-template/docs/assets/tuning-reference_joins.md.BtKs_CuW.js +1 -0
  133. package/export-template/docs/assets/tuning-reference_joins.md.BtKs_CuW.lean.js +1 -0
  134. package/export-template/docs/assets/tuning-reference_memory-model.md.DhT-n4y3.js +1 -0
  135. package/export-template/docs/assets/tuning-reference_memory-model.md.DhT-n4y3.lean.js +1 -0
  136. package/export-template/docs/assets/tuning-reference_metrics.md.mLOh7Apj.js +1 -0
  137. package/export-template/docs/assets/tuning-reference_metrics.md.mLOh7Apj.lean.js +1 -0
  138. package/export-template/docs/assets/tuning-reference_partitioning.md.q0zKF_8X.js +1 -0
  139. package/export-template/docs/assets/tuning-reference_partitioning.md.q0zKF_8X.lean.js +1 -0
  140. package/export-template/docs/assets/tuning-reference_pyspark.md.DDCfvN9t.js +6 -0
  141. package/export-template/docs/assets/tuning-reference_pyspark.md.DDCfvN9t.lean.js +1 -0
  142. package/export-template/docs/assets/tuning-reference_shuffle.md.BZZ7R4Ix.js +1 -0
  143. package/export-template/docs/assets/tuning-reference_shuffle.md.BZZ7R4Ix.lean.js +1 -0
  144. package/export-template/docs/assets/tuning-reference_spark-architecture.md.Dwzm5avO.js +1 -0
  145. package/export-template/docs/assets/tuning-reference_spark-architecture.md.Dwzm5avO.lean.js +1 -0
  146. package/export-template/docs/assets/tuning-reference_table-formats.md.D6wj-2dX.js +1 -0
  147. package/export-template/docs/assets/tuning-reference_table-formats.md.D6wj-2dX.lean.js +1 -0
  148. package/export-template/docs/assets/udf-execution-models.BUFDICuG.svg +1 -0
  149. package/export-template/docs/assets/udf-execution-models.dark.YTNS6GDq.svg +1 -0
  150. package/export-template/docs/assets/user-guide_alternative-log-retrieval.md.sU3KGarf.js +1 -0
  151. package/export-template/docs/assets/user-guide_alternative-log-retrieval.md.sU3KGarf.lean.js +1 -0
  152. package/export-template/docs/assets/user-guide_getting-started.md.DtEM37MK.js +3 -0
  153. package/export-template/docs/assets/user-guide_getting-started.md.DtEM37MK.lean.js +1 -0
  154. package/export-template/docs/assets/user-guide_mcp-tools.md.C8MiIu7F.js +125 -0
  155. package/export-template/docs/assets/user-guide_mcp-tools.md.C8MiIu7F.lean.js +1 -0
  156. package/export-template/docs/assets/user-guide_run-comparison.md.S0TWWmLY.js +1 -0
  157. package/export-template/docs/assets/user-guide_run-comparison.md.S0TWWmLY.lean.js +1 -0
  158. package/export-template/docs/assets/user-guide_understanding-findings.md.D0R_Y-R2.js +1 -0
  159. package/export-template/docs/assets/user-guide_understanding-findings.md.D0R_Y-R2.lean.js +1 -0
  160. package/export-template/docs/contributor-guide/architecture/board-widgets.html +25 -0
  161. package/export-template/docs/contributor-guide/architecture/detector-contract.html +25 -0
  162. package/export-template/docs/contributor-guide/architecture/drill-down.html +25 -0
  163. package/export-template/docs/contributor-guide/architecture/impact-estimation.html +25 -0
  164. package/export-template/docs/contributor-guide/architecture/index.html +25 -0
  165. package/export-template/docs/contributor-guide/architecture/overview.html +25 -0
  166. package/export-template/docs/contributor-guide/architecture/state-and-history.html +25 -0
  167. package/export-template/docs/contributor-guide/architecture/widget-rendering.html +25 -0
  168. package/export-template/docs/contributor-guide/architecture/worker-protocol.html +30 -0
  169. package/export-template/docs/contributor-guide/contributing.html +25 -0
  170. package/export-template/docs/contributor-guide/development-setup.html +36 -0
  171. package/export-template/docs/contributor-guide/testing.html +25 -0
  172. package/export-template/docs/favicon.svg +4 -0
  173. package/export-template/docs/hashmap.json +1 -0
  174. package/export-template/docs/index.html +25 -0
  175. package/export-template/docs/package.json +1 -0
  176. package/export-template/docs/tuning-reference/anti-patterns.html +25 -0
  177. package/export-template/docs/tuning-reference/aqe.html +25 -0
  178. package/export-template/docs/tuning-reference/bottleneck-broadcast-sizing.html +25 -0
  179. package/export-template/docs/tuning-reference/bottleneck-cold-start.html +31 -0
  180. package/export-template/docs/tuning-reference/bottleneck-duplicate-plan-subtree.html +25 -0
  181. package/export-template/docs/tuning-reference/bottleneck-failures.html +30 -0
  182. package/export-template/docs/tuning-reference/bottleneck-gc.html +30 -0
  183. package/export-template/docs/tuning-reference/bottleneck-job-failure-rate.html +32 -0
  184. package/export-template/docs/tuning-reference/bottleneck-memory-utilization.html +31 -0
  185. package/export-template/docs/tuning-reference/bottleneck-retry-waste.html +25 -0
  186. package/export-template/docs/tuning-reference/bottleneck-shuffle.html +36 -0
  187. package/export-template/docs/tuning-reference/bottleneck-skew.html +38 -0
  188. package/export-template/docs/tuning-reference/bottleneck-slow-host.html +31 -0
  189. package/export-template/docs/tuning-reference/bottleneck-small-files.html +29 -0
  190. package/export-template/docs/tuning-reference/bottleneck-spill.html +30 -0
  191. package/export-template/docs/tuning-reference/bottleneck-straggler.html +31 -0
  192. package/export-template/docs/tuning-reference/bottleneck-tiny-tasks.html +32 -0
  193. package/export-template/docs/tuning-reference/bottleneck-utilization.html +29 -0
  194. package/export-template/docs/tuning-reference/caching.html +25 -0
  195. package/export-template/docs/tuning-reference/cluster-config.html +25 -0
  196. package/export-template/docs/tuning-reference/config.html +25 -0
  197. package/export-template/docs/tuning-reference/data-formats.html +25 -0
  198. package/export-template/docs/tuning-reference/index.html +25 -0
  199. package/export-template/docs/tuning-reference/intro.html +25 -0
  200. package/export-template/docs/tuning-reference/joins.html +25 -0
  201. package/export-template/docs/tuning-reference/memory-model.html +25 -0
  202. package/export-template/docs/tuning-reference/metrics.html +25 -0
  203. package/export-template/docs/tuning-reference/partitioning.html +25 -0
  204. package/export-template/docs/tuning-reference/pyspark.html +30 -0
  205. package/export-template/docs/tuning-reference/shuffle.html +25 -0
  206. package/export-template/docs/tuning-reference/spark-architecture.html +25 -0
  207. package/export-template/docs/tuning-reference/table-formats.html +25 -0
  208. package/export-template/docs/user-guide/alternative-log-retrieval.html +25 -0
  209. package/export-template/docs/user-guide/getting-started.html +27 -0
  210. package/export-template/docs/user-guide/mcp-tools.html +149 -0
  211. package/export-template/docs/user-guide/run-comparison.html +25 -0
  212. package/export-template/docs/user-guide/understanding-findings.html +25 -0
  213. package/export-template/docs/vp-icons.css +0 -0
  214. package/export-template/favicon.svg +4 -0
  215. package/export-template/index.html +115 -0
  216. package/export-template/parser-worker-QqyEE4m9.js +64 -0
  217. package/package.json +16 -3
  218. package/vendor-core/analyzer.js +74 -74
  219. package/vendor-core/cli/budgets.js +13 -27
  220. package/vendor-core/cli/collect-run.js +43 -19
  221. package/vendor-core/core-count.js +25 -27
  222. package/vendor-core/core-locality-ratio.js +4 -11
  223. package/vendor-core/core-time-series.js +6 -12
  224. package/vendor-core/core-usage-locality.js +3 -4
  225. package/vendor-core/detectors.js +256 -375
  226. package/vendor-core/docs-config.js +69 -21
  227. package/vendor-core/docs-content/chapters/01-intro.md +32 -0
  228. package/vendor-core/docs-content/chapters/02-spark-architecture.md +76 -0
  229. package/vendor-core/docs-content/chapters/03-memory-model.md +73 -0
  230. package/vendor-core/docs-content/chapters/04-partitioning.md +65 -0
  231. package/vendor-core/docs-content/chapters/05-joins.md +62 -0
  232. package/vendor-core/docs-content/chapters/06-shuffle.md +59 -0
  233. package/vendor-core/docs-content/chapters/07-data-formats.md +81 -0
  234. package/vendor-core/docs-content/chapters/07b-table-formats.md +56 -0
  235. package/vendor-core/docs-content/chapters/08-caching.md +58 -0
  236. package/vendor-core/docs-content/chapters/09-pyspark.md +78 -0
  237. package/vendor-core/docs-content/chapters/10-aqe.md +167 -0
  238. package/vendor-core/docs-content/chapters/11-cluster-config.md +170 -0
  239. package/vendor-core/docs-content/chapters/12-anti-patterns.md +171 -0
  240. package/vendor-core/docs-content/chapters/14-metrics.md +87 -0
  241. package/vendor-core/docs-content/chapters/15-config.md +93 -0
  242. package/vendor-core/docs-content/chapters/nav-index.json +370 -0
  243. package/vendor-core/docs-content/detection/cache.md +6 -0
  244. package/vendor-core/docs-content/detection/cfg.md +15 -0
  245. package/vendor-core/docs-content/detection/chrn.md +7 -0
  246. package/vendor-core/docs-content/detection/cold.md +4 -0
  247. package/vendor-core/docs-content/detection/cstor.md +4 -0
  248. package/vendor-core/docs-content/detection/fail.md +5 -0
  249. package/vendor-core/docs-content/detection/gc.md +4 -0
  250. package/vendor-core/docs-content/detection/host.md +5 -0
  251. package/vendor-core/docs-content/detection/incmp.md +6 -0
  252. package/vendor-core/docs-content/detection/jobs.md +4 -0
  253. package/vendor-core/docs-content/detection/local.md +7 -0
  254. package/vendor-core/docs-content/detection/mem.md +10 -0
  255. package/vendor-core/docs-content/detection/part.md +5 -0
  256. package/vendor-core/docs-content/detection/plan.md +14 -0
  257. package/vendor-core/docs-content/detection/retry.md +4 -0
  258. package/vendor-core/docs-content/detection/sfail.md +5 -0
  259. package/vendor-core/docs-content/detection/shape.md +5 -0
  260. package/vendor-core/docs-content/detection/shfl.md +4 -0
  261. package/vendor-core/docs-content/detection/skew.md +6 -0
  262. package/vendor-core/docs-content/detection/slow.md +6 -0
  263. package/vendor-core/docs-content/detection/spec.md +7 -0
  264. package/vendor-core/docs-content/detection/spill.md +7 -0
  265. package/vendor-core/docs-content/detection/strag.md +5 -0
  266. package/vendor-core/docs-content/detection/tiny.md +4 -0
  267. package/vendor-core/docs-content/detection/util.md +4 -0
  268. package/vendor-core/docs-content/diagrams/aqe-loop.dark.svg +1 -0
  269. package/vendor-core/docs-content/diagrams/aqe-loop.svg +1 -0
  270. package/vendor-core/docs-content/diagrams/broadcast-vs-shuffle.dark.svg +1 -0
  271. package/vendor-core/docs-content/diagrams/broadcast-vs-shuffle.svg +1 -0
  272. package/vendor-core/docs-content/diagrams/cache-lifecycle.dark.svg +1 -0
  273. package/vendor-core/docs-content/diagrams/cache-lifecycle.svg +1 -0
  274. package/vendor-core/docs-content/diagrams/cold-start-timeline.dark.svg +1 -0
  275. package/vendor-core/docs-content/diagrams/cold-start-timeline.svg +1 -0
  276. package/vendor-core/docs-content/diagrams/columnar-layout.dark.svg +1 -0
  277. package/vendor-core/docs-content/diagrams/columnar-layout.svg +1 -0
  278. package/vendor-core/docs-content/diagrams/container-memory.dark.svg +1 -0
  279. package/vendor-core/docs-content/diagrams/container-memory.svg +1 -0
  280. package/vendor-core/docs-content/diagrams/dag-stages.dark.svg +1 -0
  281. package/vendor-core/docs-content/diagrams/dag-stages.svg +1 -0
  282. package/vendor-core/docs-content/diagrams/driver-executor.dark.svg +1 -0
  283. package/vendor-core/docs-content/diagrams/driver-executor.svg +1 -0
  284. package/vendor-core/docs-content/diagrams/duplicate-plan-subtree.dark.svg +1 -0
  285. package/vendor-core/docs-content/diagrams/duplicate-plan-subtree.svg +1 -0
  286. package/vendor-core/docs-content/diagrams/join-strategy.dark.svg +1 -0
  287. package/vendor-core/docs-content/diagrams/join-strategy.svg +1 -0
  288. package/vendor-core/docs-content/diagrams/memory-borrowing.dark.svg +1 -0
  289. package/vendor-core/docs-content/diagrams/memory-borrowing.svg +1 -0
  290. package/vendor-core/docs-content/diagrams/memory-regions.dark.svg +1 -0
  291. package/vendor-core/docs-content/diagrams/memory-regions.svg +1 -0
  292. package/vendor-core/docs-content/diagrams/repartition-vs-coalesce.dark.svg +1 -0
  293. package/vendor-core/docs-content/diagrams/repartition-vs-coalesce.svg +1 -0
  294. package/vendor-core/docs-content/diagrams/retry-escalation-ladder.dark.svg +1 -0
  295. package/vendor-core/docs-content/diagrams/retry-escalation-ladder.svg +1 -0
  296. package/vendor-core/docs-content/diagrams/shuffle-map-reduce.dark.svg +1 -0
  297. package/vendor-core/docs-content/diagrams/shuffle-map-reduce.svg +1 -0
  298. package/vendor-core/docs-content/diagrams/spill-classification.dark.svg +1 -0
  299. package/vendor-core/docs-content/diagrams/spill-classification.svg +1 -0
  300. package/vendor-core/docs-content/diagrams/udf-execution-models.dark.svg +1 -0
  301. package/vendor-core/docs-content/diagrams/udf-execution-models.svg +1 -0
  302. package/vendor-core/docs-content/tuning/broadcast-sizing.md +78 -0
  303. package/vendor-core/docs-content/tuning/cold-start.md +81 -0
  304. package/vendor-core/docs-content/tuning/duplicate-plan-subtree.md +45 -0
  305. package/vendor-core/docs-content/tuning/failures.md +124 -0
  306. package/vendor-core/docs-content/tuning/gc.md +110 -0
  307. package/vendor-core/docs-content/tuning/job-failure-rate.md +101 -0
  308. package/vendor-core/docs-content/tuning/memory-utilization.md +58 -0
  309. package/vendor-core/docs-content/tuning/retry-waste.md +90 -0
  310. package/vendor-core/docs-content/tuning/shuffle.md +154 -0
  311. package/vendor-core/docs-content/tuning/skew.md +123 -0
  312. package/vendor-core/docs-content/tuning/slow-host.md +117 -0
  313. package/vendor-core/docs-content/tuning/small-files.md +99 -0
  314. package/vendor-core/docs-content/tuning/spill.md +114 -0
  315. package/vendor-core/docs-content/tuning/straggler.md +103 -0
  316. package/vendor-core/docs-content/tuning/tiny-tasks.md +94 -0
  317. package/vendor-core/docs-content/tuning/utilization.md +90 -0
  318. package/vendor-core/docs-site-config.js +10 -17
  319. package/vendor-core/efficiency-model.js +7 -13
  320. package/vendor-core/etl-phases.js +3 -5
  321. package/vendor-core/event-handlers.js +232 -134
  322. package/vendor-core/event-schemas.js +48 -114
  323. package/vendor-core/evidence-availability.js +5 -10
  324. package/vendor-core/evidence-report.js +72 -122
  325. package/vendor-core/export-data.js +48 -0
  326. package/vendor-core/finding-action-label.js +4 -10
  327. package/vendor-core/finding-filter-predicate.js +3 -7
  328. package/vendor-core/finding-generic-recommendation.js +112 -0
  329. package/vendor-core/finding-names.js +51 -0
  330. package/vendor-core/format-utils.js +112 -38
  331. package/vendor-core/impact-band.js +18 -24
  332. package/vendor-core/impact-estimator.js +38 -74
  333. package/vendor-core/ingest.js +7 -13
  334. package/vendor-core/job-groups.js +3 -6
  335. package/vendor-core/list-runs.js +278 -0
  336. package/vendor-core/load-vendored.js +6 -12
  337. package/vendor-core/log-header-peek.js +81 -0
  338. package/vendor-core/lz4-block.js +4 -6
  339. package/vendor-core/mcp-server-factory.js +38 -8
  340. package/vendor-core/mcp-tools.js +105 -76
  341. package/vendor-core/model-assembler.js +8 -16
  342. package/vendor-core/occupancy.js +5 -9
  343. package/vendor-core/parser-worker.js +18 -27
  344. package/vendor-core/plan-dot.js +2 -5
  345. package/vendor-core/plan-duration-attribution.js +78 -29
  346. package/vendor-core/plan-graph-model.js +126 -69
  347. package/vendor-core/plan-node-detail.js +31 -17
  348. package/vendor-core/plan-summary.js +19 -8
  349. package/vendor-core/recommendation-rollup.js +35 -39
  350. package/vendor-core/redact.js +72 -16
  351. package/vendor-core/rolling-log-reassembly.js +4 -6
  352. package/vendor-core/run-comparison.js +65 -70
  353. package/vendor-core/scaling-sim.js +5 -7
  354. package/vendor-core/session-snapshot.js +1 -1
  355. package/vendor-core/shs-fetch.js +4 -6
  356. package/vendor-core/shs-load.js +9 -13
  357. package/vendor-core/shs-request.js +1 -1
  358. package/vendor-core/stage-quantiles.js +14 -0
  359. package/vendor-core/types.js +78 -18
  360. package/vendor-core/wasted-core-hours.js +7 -12
@@ -0,0 +1,14 @@
1
+ import{_ as e,o as s,c as t,a5 as i}from"./chunks/framework.DSg0KOwT.js";const f=JSON.parse('{"title":"Task Skew","description":"","frontmatter":{"title":"Task Skew"},"headers":[],"relativePath":"tuning-reference/bottleneck-skew.md","filePath":"tuning-reference/bottleneck-skew.md"}'),n={name:"tuning-reference/bottleneck-skew.md"};function o(r,a,l,h,p,d){return s(),t("div",null,[...a[0]||(a[0]=[i(`<h1 id="bottleneck-skew" tabindex="-1">Task Skew <a class="header-anchor" href="#bottleneck-skew" aria-label="Permalink to &quot;Task Skew {#bottleneck-skew}&quot;">​</a></h1><p><span class="tag">SKEW</span></p><h2 id="what-it-is" tabindex="-1">What it is <a class="header-anchor" href="#what-it-is" aria-label="Permalink to &quot;What it is&quot;">​</a></h2><p>Task skew happens when a shuffle produces one or a few oversized partitions instead of a roughly even split. Spark&#39;s own skew-join optimizer defines a partition as skewed once it is both larger than a multiple of the median partition size and larger than an absolute byte threshold<sup class="footnote-ref"><a href="#fn1" id="fnref1">[1]</a></sup>: in the shipped implementation, more than 5× the median size and more than 256 MB by default<sup class="footnote-ref"><a href="#fn2" id="fnref2">[2]</a></sup>. Whatever task draws that oversized partition ends up processing far more data than everyone else in the same stage.</p><h2 id="how-it-s-detected" tabindex="-1">How it&#39;s detected <a class="header-anchor" href="#how-it-s-detected" aria-label="Permalink to &quot;How it&#39;s detected&quot;">​</a></h2><table tabindex="0"><thead><tr><th>Signal</th><th>Warning</th><th>Critical</th></tr></thead><tbody><tr><td>P95 / median task duration</td><td>&gt; 3×</td><td>&gt; 5×</td></tr><tr><td>Max / median task duration (task count &lt; 20)</td><td>&gt; 3×</td><td>&gt; 5×</td></tr></tbody></table><h2 id="why-it-matters" tabindex="-1">Why it matters <a class="header-anchor" href="#why-it-matters" aria-label="Permalink to &quot;Why it matters&quot;">​</a></h2><p>An oversized partition becomes a <a href="./bottleneck-straggler.html">straggler task</a>: it&#39;s processed inside a single task, so the stage can&#39;t finish until that task does, no matter how long it takes. Spark&#39;s own skew handling treats splitting the partition as a deliberate trade-off: reading the other join side&#39;s matching partition once per split costs extra I/O, but the design behind the feature argues that cost is worth paying once the skew is severe enough to be producing a straggler in the first place<sup class="footnote-ref"><a href="#fn1" id="fnref1:1">[1:1]</a></sup>.</p><h2 id="how-to-fix-it" tabindex="-1">How to fix it <a class="header-anchor" href="#how-to-fix-it" aria-label="Permalink to &quot;How to fix it&quot;">​</a></h2><ul><li>Enable Adaptive Query Execution&#39;s skew-join handling (<code>spark.sql.adaptive.skewJoin.enabled</code>, default <code>true</code> once <code>spark.sql.adaptive.enabled</code> is also on<sup class="footnote-ref"><a href="#fn2" id="fnref2:1">[2:1]</a></sup>). It divides any partition that crosses the skew thresholds into several smaller sub-partitions, joins each one against the matching data on the other side, and unions the results back together<sup class="footnote-ref"><a href="#fn1" id="fnref1:2">[1:2]</a></sup>.</li><li>Before AQE existed, the only options were manual, and each carries real limitations, which is exactly the gap AQE&#39;s skew handling was built to close<sup class="footnote-ref"><a href="#fn1" id="fnref1:3">[1:3]</a></sup>: salt the join key, over-size <code>spark.sql.shuffle.partitions</code>, or push the <a href="./bottleneck-broadcast-sizing.html">broadcast-join threshold</a> up so the join goes broadcast instead of sort-merge.</li><li>To salt by hand: add a random suffix to the join key on both sides, exploding the smaller side into one row per salt value so every salted variant of the key still finds a match, then join on the combined (key, salt) pair<sup class="footnote-ref"><a href="#fn3" id="fnref3">[3]</a></sup>.</li><li>On Databricks, a <code>SKEW</code> hint can name the skewed relation and column (and, optionally, the exact skewed values) directly, letting the planner build a skew-aware plan without hand-rolled salting<sup class="footnote-ref"><a href="#fn4" id="fnref4">[4]</a></sup>.</li></ul><blockquote><p><strong>PySpark:</strong> salting is plain DataFrame code, no special API, just <code>withColumn(&quot;salt&quot;, (fn.rand() * n).cast(&quot;int&quot;))</code> on the larger side and an <code>explode</code> over the salt range on the smaller side before joining on the combined key.</p></blockquote><p>The AQE toggle, plus the manual salting fallback as runnable code:</p><div class="language-python vp-adaptive-theme"><button title="Copy Code" class="copy"></button><span class="lang">python</span><pre class="shiki shiki-themes github-light github-dark vp-code" tabindex="0"><code><span class="line"><span style="--shiki-light:#D73A49;--shiki-dark:#F97583;">from</span><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;"> pyspark.sql </span><span style="--shiki-light:#D73A49;--shiki-dark:#F97583;">import</span><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;"> functions </span><span style="--shiki-light:#D73A49;--shiki-dark:#F97583;">as</span><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;"> fn</span></span>
2
+ <span class="line"></span>
3
+ <span class="line"><span style="--shiki-light:#6A737D;--shiki-dark:#6A737D;"># AQE skew-join handling, active once AQE itself is enabled</span></span>
4
+ <span class="line"><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;">spark.conf.set(</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;">&quot;spark.sql.adaptive.enabled&quot;</span><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;">, </span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;">&quot;true&quot;</span><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;">)</span></span>
5
+ <span class="line"><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;">spark.conf.set(</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;">&quot;spark.sql.adaptive.skewJoin.enabled&quot;</span><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;">, </span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;">&quot;true&quot;</span><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;">)</span></span>
6
+ <span class="line"></span>
7
+ <span class="line"><span style="--shiki-light:#6A737D;--shiki-dark:#6A737D;"># Manual salting fallback (pre-AQE): spread the skewed key across N salted variants.</span></span>
8
+ <span class="line"><span style="--shiki-light:#6A737D;--shiki-dark:#6A737D;"># N is an example, size it to how badly the key is skewed.</span></span>
9
+ <span class="line"><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;">N </span><span style="--shiki-light:#D73A49;--shiki-dark:#F97583;">=</span><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> 16</span></span>
10
+ <span class="line"><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;">salted_big </span><span style="--shiki-light:#D73A49;--shiki-dark:#F97583;">=</span><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;"> big.withColumn(</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;">&quot;salt&quot;</span><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;">, (fn.rand() </span><span style="--shiki-light:#D73A49;--shiki-dark:#F97583;">*</span><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;"> N).cast(</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;">&quot;int&quot;</span><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;">))</span></span>
11
+ <span class="line"><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;">salted_small </span><span style="--shiki-light:#D73A49;--shiki-dark:#F97583;">=</span><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;"> small.withColumn(</span></span>
12
+ <span class="line"><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> &quot;salt&quot;</span><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;">, fn.explode(fn.array([fn.lit(i) </span><span style="--shiki-light:#D73A49;--shiki-dark:#F97583;">for</span><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;"> i </span><span style="--shiki-light:#D73A49;--shiki-dark:#F97583;">in</span><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> range</span><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;">(N)]))</span></span>
13
+ <span class="line"><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;">)</span></span>
14
+ <span class="line"><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;">joined </span><span style="--shiki-light:#D73A49;--shiki-dark:#F97583;">=</span><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;"> salted_big.join(salted_small, [</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;">&quot;key&quot;</span><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;">, </span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;">&quot;salt&quot;</span><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;">]).drop(</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;">&quot;salt&quot;</span><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;">)</span></span></code></pre></div><h2 id="confidence" tabindex="-1">Confidence <a class="header-anchor" href="#confidence" aria-label="Permalink to &quot;Confidence&quot;">​</a></h2><p>Validated. These thresholds mirror Spark&#39;s own skew-join optimizer, which defines a skewed partition by the same style of ratio-plus-absolute test applied here to task durations<sup class="footnote-ref"><a href="#fn1" id="fnref1:4">[1:4]</a></sup><sup class="footnote-ref"><a href="#fn2" id="fnref2:2">[2:2]</a></sup>.</p><p>The <code>Stage shape</code> subsection below is a separate, experimental branch: its heuristics are informational stage-shape observations, not a validated finding, so it carries an <code>EXPERIMENTAL</code> badge and should not be read as a diagnosis on its own.</p><h2 id="limitations-false-positive-risk" tabindex="-1">Limitations / false-positive risk <a class="header-anchor" href="#limitations-false-positive-risk" aria-label="Permalink to &quot;Limitations / false-positive risk&quot;">​</a></h2><p>A task-duration ratio is a proxy for skew, not proof of it. A single long task can just as easily be a garbage-collection pause or a genuinely slow host, and either one inflates the max/median ratio without any partition being oversized. Small stages make the number jumpy: with only a handful of tasks, one slow outlier moves the median enough to trip the threshold on noise alone. Treat the signal as a prompt to look at the stage, not a verdict.</p><h2 id="bottleneck-stage-shape" tabindex="-1">Stage shape <a class="header-anchor" href="#bottleneck-stage-shape" aria-label="Permalink to &quot;Stage shape {#bottleneck-stage-shape}&quot;">​</a></h2><p><span class="tag">SHAPE</span> <span class="tag">EXPERIMENTAL</span></p><p>These are experimental, information-only heuristics about a stage&#39;s overall shape, not a graded finding. They read the relationship between a stage&#39;s task count and the cluster&#39;s resources, and none of them alone means something is wrong.</p><p>A stage&#39;s task count equals the number of partitions in its output, and each task runs as a single thread against one partition, so that count is what caps how much of the cluster the stage can actually use. Three shapes are worth noticing:</p><ul><li><strong>Low parallelism</strong> (task count far below total executor cores). When a stage has fewer tasks than there are core slots to run them in, cores sit idle and the stage cannot put the cluster&#39;s CPU to work; the same shape also concentrates more memory pressure into each task&#39;s aggregation<sup class="footnote-ref"><a href="#fn5" id="fnref5">[5]</a></sup>. Spark&#39;s guidance is to keep parallelism high enough that the cluster does not stay under-used, roughly 2 to 3 tasks per CPU core in general<sup class="footnote-ref"><a href="#fn6" id="fnref6">[6]</a></sup>.</li><li><strong>Data explosion</strong> (task count far above cores, partitions tiny). Push the count too high and partitions shrink until per-task overhead floods the stage, so the scheduling cost starts to dominate the useful work<sup class="footnote-ref"><a href="#fn7" id="fnref7">[7]</a></sup>.</li><li><strong>Task and stage skew</strong> (uneven task count or duration). A lopsided split shows up as a few tasks carrying the stage while the rest finish early, which is the same imbalance the graded <code>skew</code> finding above tracks by duration ratio.</li></ul><h2 id="related" tabindex="-1">Related <a class="header-anchor" href="#related" aria-label="Permalink to &quot;Related&quot;">​</a></h2><ul><li><strong>Why it happens:</strong> <a href="./partitioning.html">Partitioning</a>, <a href="./joins.html">Join Optimization</a></li><li><strong>How Spark fixes it automatically:</strong> <a href="./aqe.html">Adaptive Query Execution</a></li></ul><hr class="footnotes-sep"><section class="footnotes"><ol class="footnotes-list"><li id="fn1" class="footnote-item"><p><a href="https://issues.apache.org/jira/browse/SPARK-29544" target="_blank" rel="noreferrer">SPARK-29544: Optimize Skewed Join at Runtime</a> <a href="#fnref1" class="footnote-backref">↩︎</a> <a href="#fnref1:1" class="footnote-backref">↩︎</a> <a href="#fnref1:2" class="footnote-backref">↩︎</a> <a href="#fnref1:3" class="footnote-backref">↩︎</a> <a href="#fnref1:4" class="footnote-backref">↩︎</a></p></li><li id="fn2" class="footnote-item"><p><a href="https://spark.apache.org/docs/latest/configuration.html" target="_blank" rel="noreferrer">Configuration (Spark)</a> <a href="#fnref2" class="footnote-backref">↩︎</a> <a href="#fnref2:1" class="footnote-backref">↩︎</a> <a href="#fnref2:2" class="footnote-backref">↩︎</a></p></li><li id="fn3" class="footnote-item"><p><a href="https://luminousmen.com/post/spark-tips-partition-tuning" target="_blank" rel="noreferrer">Spark Tips: Partition Tuning</a> <a href="#fnref3" class="footnote-backref">↩︎</a></p></li><li id="fn4" class="footnote-item"><p><a href="https://docs.databricks.com/aws/en/archive/legacy/skew-join" target="_blank" rel="noreferrer">Skew Join Hint</a> <a href="#fnref4" class="footnote-backref">↩︎</a></p></li><li id="fn5" class="footnote-item"><p><a href="https://blog.cloudera.com/how-to-tune-your-apache-spark-jobs-part-2/" target="_blank" rel="noreferrer">How to Tune Your Apache Spark Jobs (Part 2)</a> <a href="#fnref5" class="footnote-backref">↩︎</a></p></li><li id="fn6" class="footnote-item"><p><a href="https://spark.apache.org/docs/latest/tuning.html" target="_blank" rel="noreferrer">Tuning - Spark</a> <a href="#fnref6" class="footnote-backref">↩︎</a></p></li><li id="fn7" class="footnote-item"><p><a href="https://luminousmen.com/post/spark-partitions" target="_blank" rel="noreferrer">Spark Partitions</a> <a href="#fnref7" class="footnote-backref">↩︎</a></p></li></ol></section>`,27)])])}const c=e(n,[["render",o]]);export{f as __pageData,c as default};
@@ -0,0 +1 @@
1
+ import{_ as e,o as s,c as t,a5 as i}from"./chunks/framework.DSg0KOwT.js";const f=JSON.parse('{"title":"Task Skew","description":"","frontmatter":{"title":"Task Skew"},"headers":[],"relativePath":"tuning-reference/bottleneck-skew.md","filePath":"tuning-reference/bottleneck-skew.md"}'),n={name:"tuning-reference/bottleneck-skew.md"};function o(r,a,l,h,p,d){return s(),t("div",null,[...a[0]||(a[0]=[i("",27)])])}const c=e(n,[["render",o]]);export{f as __pageData,c as default};
@@ -0,0 +1,7 @@
1
+ import{_ as t,o as a,c as s,a5 as o}from"./chunks/framework.DSg0KOwT.js";const p=JSON.parse('{"title":"Slow Host","description":"","frontmatter":{"title":"Slow Host"},"headers":[],"relativePath":"tuning-reference/bottleneck-slow-host.md","filePath":"tuning-reference/bottleneck-slow-host.md"}'),i={name:"tuning-reference/bottleneck-slow-host.md"};function n(r,e,l,h,c,d){return a(),s("div",null,[...e[0]||(e[0]=[o(`<h1 id="bottleneck-slow-host" tabindex="-1">Slow Host <a class="header-anchor" href="#bottleneck-slow-host" aria-label="Permalink to &quot;Slow Host {#bottleneck-slow-host}&quot;">​</a></h1><p><span class="tag">HOST</span></p><h2 id="what-it-is" tabindex="-1">What it is <a class="header-anchor" href="#what-it-is" aria-label="Permalink to &quot;What it is&quot;">​</a></h2><p>A slow host bottleneck shows up when one machine in the cluster consistently turns in slower task times than its peers, independent of any single task&#39;s own data size. Unlike a stray <a href="./bottleneck-straggler.html">straggler task</a>, the effect is host-wide: every task Spark schedules there runs behind, which drags out the stage even when the workload itself is partitioned evenly.</p><h2 id="how-it-s-detected" tabindex="-1">How it&#39;s detected <a class="header-anchor" href="#how-it-s-detected" aria-label="Permalink to &quot;How it&#39;s detected&quot;">​</a></h2><p>Spark&#39;s event log carries the per-task detail behind this: turning on <code>spark.eventLog.enabled</code> logs the events that encode what the UI displays, persisted to storage<sup class="footnote-ref"><a href="#fn1" id="fnref1">[1]</a></sup>, and that same log backs the UI&#39;s Stages tab, which drills down into individual tasks and shows per-task metrics such as duration, GC time, and shuffle bytes read<sup class="footnote-ref"><a href="#fn2" id="fnref2">[2]</a></sup>.</p><p>A host reads as slow against the following synthetic threshold: host mean task duration ≥ 2× overall median AND the host holds ≥ 20% task share; the signal only applies with <code>taskCount ≥ 15</code> and <code>hosts.length ≥ 3</code>.</p><h2 id="why-it-matters" tabindex="-1">Why it matters <a class="header-anchor" href="#why-it-matters" aria-label="Permalink to &quot;Why it matters&quot;">​</a></h2><p>A host running persistently slow tasks behaves like a bottleneck baked into the cluster rather than into the workload: every task Spark places there inherits the delay, and since a stage&#39;s completion time is bounded by its slowest tasks, the rest of the cluster idles while the affected host catches up. Left alone, the same host keeps dragging down every later stage and job that lands work on it.</p><h2 id="how-to-fix-it" tabindex="-1">How to fix it <a class="header-anchor" href="#how-to-fix-it" aria-label="Permalink to &quot;How to fix it&quot;">​</a></h2><ul><li>Turn on speculative execution (<code>spark.speculation</code>, off by default) so Spark relaunches a copy of a task that&#39;s lagging far behind its peers instead of waiting on the slow host to finish it. A stage only becomes eligible once <code>spark.speculation.quantile</code> (default <code>0.9</code>) of its tasks have completed, and a task then qualifies once it runs more than <code>spark.speculation.multiplier</code> (default <code>3</code>) times the median duration, subject to a <code>spark.speculation.minTaskRuntime</code> floor (default <code>100ms</code>) so short tasks aren&#39;t speculated purely for looking slow, or, since Spark 3.4, an efficiency check (<code>spark.speculation.efficiency.enabled</code>, default <code>true</code>) that also requires the task&#39;s data-processing rate to lag the stage average<sup class="footnote-ref"><a href="#fn3" id="fnref3">[3]</a></sup>.</li><li>If the slow host also happens to hold data locality for the affected tasks, lowering <code>spark.locality.wait</code> (default <code>3s</code>), or the level-specific <code>spark.locality.wait.node</code>, <code>.rack</code>, and <code>.process</code> overrides, shortens how long Spark waits for a data-local slot before falling back to a less-local executor<sup class="footnote-ref"><a href="#fn3" id="fnref3:1">[3:1]</a></sup>.</li></ul><p>Turn on speculation and, if the slow host holds locality, shorten the wait; defaults shown:</p><div class="language-properties vp-adaptive-theme"><button title="Copy Code" class="copy"></button><span class="lang">properties</span><pre class="shiki shiki-themes github-light github-dark vp-code" tabindex="0"><code><span class="line"><span style="--shiki-light:#6A737D;--shiki-dark:#6A737D;"># Relaunch tasks stuck on a slow host (speculation is OFF by default)</span></span>
2
+ <span class="line"><span style="--shiki-light:#D73A49;--shiki-dark:#F97583;">spark.speculation</span><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;">=true</span></span>
3
+ <span class="line"><span style="--shiki-light:#D73A49;--shiki-dark:#F97583;">spark.speculation.quantile</span><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;">=0.9 </span><span style="--shiki-light:#6A737D;--shiki-dark:#6A737D;"># default; fraction of tasks done before speculation starts</span></span>
4
+ <span class="line"><span style="--shiki-light:#D73A49;--shiki-dark:#F97583;">spark.speculation.multiplier</span><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;">=3 </span><span style="--shiki-light:#6A737D;--shiki-dark:#6A737D;"># default; a task must run &gt; 3x the median to be relaunched</span></span>
5
+ <span class="line"></span>
6
+ <span class="line"><span style="--shiki-light:#6A737D;--shiki-dark:#6A737D;"># If the slow host holds data locality, wait less before falling back to another executor</span></span>
7
+ <span class="line"><span style="--shiki-light:#D73A49;--shiki-dark:#F97583;">spark.locality.wait</span><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;">=3s </span><span style="--shiki-light:#6A737D;--shiki-dark:#6A737D;"># default; lower to fall back sooner</span></span></code></pre></div><h2 id="confidence" tabindex="-1">Confidence <a class="header-anchor" href="#confidence" aria-label="Permalink to &quot;Confidence&quot;">​</a></h2><p>The core <code>durationShare</code> dimension is validated: it measures host-wide task-duration inflation against the cluster median, the signal that separates a genuinely slow machine from normal task-time variance. Two secondary dimensions run best-effort and stay experimental. The <code>multiDim</code> dimension <span class="tag">EXPERIMENTAL</span> folds several per-host signals into one score, but that combined heuristic has not been validated. The <code>storageMemory</code> dimension <span class="tag">EXPERIMENTAL</span> likewise reads host-level memory pressure as a contributing factor without validation, so treat both as hints rather than verdicts.</p><h2 id="limitations-false-positive-risk" tabindex="-1">Limitations / false-positive risk <a class="header-anchor" href="#limitations-false-positive-risk" aria-label="Permalink to &quot;Limitations / false-positive risk&quot;">​</a></h2><p>A host that looks slow is not always a bad node. It can simply hold data locality for the tasks scheduled there, or carry one heavy stage that happens to be pinned to it, either of which inflates its mean task time without any hardware fault. The multi-dimensional scoring that reinforces the core signal is best-effort, so a match warrants a look at what that host was actually running before you conclude the machine itself is the problem.</p><h2 id="bottleneck-stage-slowness" tabindex="-1">Stage slowness <a class="header-anchor" href="#bottleneck-stage-slowness" aria-label="Permalink to &quot;Stage slowness {#bottleneck-stage-slowness}&quot;">​</a></h2><p><span class="tag">SLOW</span> <span class="tag">EXPERIMENTAL</span></p><p>This is an experimental fallback heuristic that shares the slow-host anchor and is suppressed whenever <code>slowHost</code> fires. It answers a different question: when no single machine is dragging, why does one stage still lag the rest of the job? The cause usually traces back to how the work was divided rather than where it ran.</p><p><strong>Too few partitions caps parallelism.</strong> Wide stages (joins, groupBy, aggregations) take their partition count from <code>spark.sql.shuffle.partitions</code>, which sits at a default of 200 whether the shuffle moves 20 MB or 500 GB unless you change it<sup class="footnote-ref"><a href="#fn4" id="fnref4">[4]</a></sup>. When that count is small relative to the cluster, only a handful of tasks carry the stage while cores sit idle, so it stretches out even as better-sized neighbors finish quickly. The first move is to raise its parallelism, aiming for at least two or three tasks per CPU core on a data-heavy stage, tuned through <code>spark.default.parallelism</code> and <code>spark.sql.shuffle.partitions</code><sup class="footnote-ref"><a href="#fn5" id="fnref5">[5]</a></sup>.</p><p><strong>Large data volume per task makes each task run long.</strong> That 200-partition default does not scale with input size, so a stage handling hundreds of gigabytes hands each task a large slice: fewer tasks run at once, per-executor load climbs, and the stage often trips memory errors. Roughly 100-200 MB per task tends to work well, and tasks grinding through around 3 GB each while spilling are a sign you need more partitions<sup class="footnote-ref"><a href="#fn4" id="fnref4:1">[4:1]</a></sup>.</p><p><strong>Heavy shuffle and spill inflate the stage&#39;s cost.</strong> Shuffles move data across executors so rows with the same key land together, and during those map and <a href="./shuffle.html">shuffle</a> operations Spark writes to and reads from local-disk shuffle files, which is heavy I/O that can become a bottleneck under the default configuration<sup class="footnote-ref"><a href="#fn2" id="fnref2:1">[2:1]</a></sup>. Volume per task feeds straight into it: once a partition outgrows the memory available in an executor, Spark <a href="./bottleneck-spill.html">spills</a> part of the data to disk, and spills are about the slowest thing a job can do because of the extra disk I/O and garbage collection, so the stage finishes but runs far less efficiently than the rest<sup class="footnote-ref"><a href="#fn4" id="fnref4:2">[4:2]</a></sup>.</p><h2 id="related" tabindex="-1">Related <a class="header-anchor" href="#related" aria-label="Permalink to &quot;Related&quot;">​</a></h2><ul><li><strong>Speculation &amp; locality tuning:</strong> <a href="./cluster-config.html">Cluster Tuning</a></li></ul><hr class="footnotes-sep"><section class="footnotes"><ol class="footnotes-list"><li id="fn1" class="footnote-item"><p><a href="https://spark.apache.org/docs/latest/monitoring.html#spark-history-server" target="_blank" rel="noreferrer">Monitoring and Instrumentation</a> <a href="#fnref1" class="footnote-backref">↩︎</a></p></li><li id="fn2" class="footnote-item"><p><em>Learning Spark, 2nd Edition</em>, Damji, Wenig, Das, Lee, ch. 7 <a href="#fnref2" class="footnote-backref">↩︎</a> <a href="#fnref2:1" class="footnote-backref">↩︎</a></p></li><li id="fn3" class="footnote-item"><p><a href="https://spark.apache.org/docs/latest/configuration.html" target="_blank" rel="noreferrer">Configuration (Spark)</a> <a href="#fnref3" class="footnote-backref">↩︎</a> <a href="#fnref3:1" class="footnote-backref">↩︎</a></p></li><li id="fn4" class="footnote-item"><p><a href="https://luminousmen.com/post/spark-partitions" target="_blank" rel="noreferrer">Spark Partitions</a> <a href="#fnref4" class="footnote-backref">↩︎</a> <a href="#fnref4:1" class="footnote-backref">↩︎</a> <a href="#fnref4:2" class="footnote-backref">↩︎</a></p></li><li id="fn5" class="footnote-item"><p><em>Spark: The Definitive Guide</em>, Chambers &amp; Zaharia (O&#39;Reilly, 2018) <a href="#fnref5" class="footnote-backref">↩︎</a></p></li></ol></section>`,27)])])}const u=t(i,[["render",n]]);export{p as __pageData,u as default};
@@ -0,0 +1 @@
1
+ import{_ as t,o as a,c as s,a5 as o}from"./chunks/framework.DSg0KOwT.js";const p=JSON.parse('{"title":"Slow Host","description":"","frontmatter":{"title":"Slow Host"},"headers":[],"relativePath":"tuning-reference/bottleneck-slow-host.md","filePath":"tuning-reference/bottleneck-slow-host.md"}'),i={name:"tuning-reference/bottleneck-slow-host.md"};function n(r,e,l,h,c,d){return a(),s("div",null,[...e[0]||(e[0]=[o("",27)])])}const u=t(i,[["render",n]]);export{p as __pageData,u as default};
@@ -0,0 +1,5 @@
1
+ import{_ as t,o as a,c as s,a5 as i}from"./chunks/framework.DSg0KOwT.js";const p=JSON.parse('{"title":"Small Files","description":"","frontmatter":{"title":"Small Files"},"headers":[],"relativePath":"tuning-reference/bottleneck-small-files.md","filePath":"tuning-reference/bottleneck-small-files.md"}'),o={name:"tuning-reference/bottleneck-small-files.md"};function n(r,e,f,l,h,c){return a(),s("div",null,[...e[0]||(e[0]=[i(`<h1 id="bottleneck-small-files" tabindex="-1">Small Files <a class="header-anchor" href="#bottleneck-small-files" aria-label="Permalink to &quot;Small Files {#bottleneck-small-files}&quot;">​</a></h1><p><span class="tag">SMALLFILES</span></p><h2 id="what-it-is" tabindex="-1">What it is <a class="header-anchor" href="#what-it-is" aria-label="Permalink to &quot;What it is&quot;">​</a></h2><p>The small-files problem is the cost of managing a large count of tiny files. Writing many small files runs up significant metadata overhead, and Spark handles that pattern badly; distributed filesystems such as HDFS handle it badly too, which is why it earned the name &quot;small file problem&quot;<sup class="footnote-ref"><a href="#fn1" id="fnref1">[1]</a></sup>. The same trap shows up on the read side. Push <code>spark.sql.files.maxPartitionBytes</code> (default 128 MB, aligned to the <a href="./data-formats.html">Parquet</a> block size) too low and Spark carves the input into many small partition files, piling on disk I/O and the filesystem overhead of opening, closing, and listing directories, all of which are slow on a distributed store<sup class="footnote-ref"><a href="#fn2" id="fnref2">[2]</a></sup><sup class="footnote-ref"><a href="#fn3" id="fnref3">[3]</a></sup>.</p><p>Spark writes one file per output partition, so a job left with 200 partitions writes 200 files and one left with 3000 writes 3000, even when many of those files hold only a handful of rows. That count then punishes every downstream job forced to read thousands of tiny files<sup class="footnote-ref"><a href="#fn4" id="fnref4">[4]</a></sup>. The opposite extreme is not free either: files that are too large make it inefficient to read a whole block when you only need a few rows<sup class="footnote-ref"><a href="#fn5" id="fnref5">[5]</a></sup>.</p><h2 id="how-it-s-detected" tabindex="-1">How it&#39;s detected <a class="header-anchor" href="#how-it-s-detected" aria-label="Permalink to &quot;How it&#39;s detected&quot;">​</a></h2><p>Reading the SQL plan surfaces write paths that will emit far more output files than the data warrants, keying off the one-file-per-output-partition rule<sup class="footnote-ref"><a href="#fn4" id="fnref4:1">[4:1]</a></sup>. The signal is a high output-partition count against a modest data volume, which foreshadows a directory full of tiny files.</p><table tabindex="0"><thead><tr><th>Signal</th><th>What it points to</th></tr></thead><tbody><tr><td>Output partition count high vs. data volume</td><td>Many tiny files on write</td></tr><tr><td>Read partitioning below <code>spark.sql.files.maxPartitionBytes</code> (128 MB)</td><td>Over-split input, excess filesystem I/O<sup class="footnote-ref"><a href="#fn2" id="fnref2:1">[2:1]</a></sup><sup class="footnote-ref"><a href="#fn3" id="fnref3:1">[3:1]</a></sup></td></tr></tbody></table><h2 id="why-it-matters" tabindex="-1">Why it matters <a class="header-anchor" href="#why-it-matters" aria-label="Permalink to &quot;Why it matters&quot;">​</a></h2><p>Every extra file is another open, close, and directory-list operation, and on a distributed filesystem those are not cheap<sup class="footnote-ref"><a href="#fn2" id="fnref2:2">[2:2]</a></sup>. The count compounds downstream: a stage that scatters 3000 tiny files hands the next job 3000 files to read back, so the metadata tax is paid twice over<sup class="footnote-ref"><a href="#fn4" id="fnref4:2">[4:2]</a></sup>. None of it is a single dramatic failure. The waste is spread thin across the file count, which is exactly why it goes unnoticed until listing and I/O start to dominate.</p><h2 id="how-to-fix-it" tabindex="-1">How to fix it <a class="header-anchor" href="#how-to-fix-it" aria-label="Permalink to &quot;How to fix it&quot;">​</a></h2><p>Repartition right before the write so the output lands in fewer, better-sized files.</p><ul><li><code>coalesce()</code> is a shuffle-free merge: it fuses existing partitions with no data movement, so <code>df.coalesce(100).write.parquet(...)</code> collapses 3000 tiny files into 100 reasonable ones<sup class="footnote-ref"><a href="#fn4" id="fnref4:3">[4:3]</a></sup>. It does not rebalance, though, so uneven inputs stay uneven once lumped together, and pushing it to <code>coalesce(1)</code> kills parallelism by forcing one executor to do all the work<sup class="footnote-ref"><a href="#fn4" id="fnref4:4">[4:4]</a></sup>.</li><li><code>repartition()</code> is a full reshuffle that buys even distribution at the cost of that <a href="./shuffle.html">shuffle</a>. Reach for it when the distribution itself needs fixing, not just the file count<sup class="footnote-ref"><a href="#fn4" id="fnref4:5">[4:5]</a></sup>.</li></ul><blockquote><p><strong>PySpark:</strong> both are one-line calls on a DataFrame: <code>df.coalesce(100)</code> for a shuffle-free merge, or <code>df.repartition(100)</code> when the data also needs rebalancing.</p></blockquote><div class="language-python vp-adaptive-theme"><button title="Copy Code" class="copy"></button><span class="lang">python</span><pre class="shiki shiki-themes github-light github-dark vp-code" tabindex="0"><code><span class="line"><span style="--shiki-light:#6A737D;--shiki-dark:#6A737D;"># Collapse many tiny output files without a shuffle (distribution already even)</span></span>
2
+ <span class="line"><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;">df.coalesce(</span><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;">100</span><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;">).write.parquet(path)</span></span>
3
+ <span class="line"></span>
4
+ <span class="line"><span style="--shiki-light:#6A737D;--shiki-dark:#6A737D;"># Full reshuffle to a target count when the distribution itself needs fixing</span></span>
5
+ <span class="line"><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;">df.repartition(</span><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;">100</span><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;">).write.parquet(path)</span></span></code></pre></div><p>For runtime control, <a href="./aqe.html">Adaptive Query Execution</a> can merge tiny shuffle partitions on its own. With <code>spark.sql.adaptive.enabled</code> and <code>spark.sql.adaptive.coalescePartitions.enabled</code> (default true) set, AQE coalesces contiguous shuffle partitions toward <code>spark.sql.adaptive.advisoryPartitionSizeInBytes</code> (default 64 MB)<sup class="footnote-ref"><a href="#fn4" id="fnref4:6">[4:6]</a></sup><sup class="footnote-ref"><a href="#fn5" id="fnref5:1">[5:1]</a></sup>. Just mind the scope: AQE only engages after the first shuffle, so it fixes shuffle-output partition counts but will not repair input-side partitioning or a bad file layout on disk<sup class="footnote-ref"><a href="#fn6" id="fnref6">[6]</a></sup>.</p><h2 id="confidence" tabindex="-1">Confidence <a class="header-anchor" href="#confidence" aria-label="Permalink to &quot;Confidence&quot;">​</a></h2><p>The core inference, that one output file per partition times a high partition count yields many tiny files, is grounded directly in Spark&#39;s write behavior<sup class="footnote-ref"><a href="#fn4" id="fnref4:7">[4:7]</a></sup>, and the mitigation levers (<code>coalesce()</code>, <code>repartition()</code>, and AQE coalescing) are documented Spark features<sup class="footnote-ref"><a href="#fn4" id="fnref4:8">[4:8]</a></sup><sup class="footnote-ref"><a href="#fn5" id="fnref5:2">[5:2]</a></sup>.</p><h2 id="limitations-false-positive-risk" tabindex="-1">Limitations / false-positive risk <a class="header-anchor" href="#limitations-false-positive-risk" aria-label="Permalink to &quot;Limitations / false-positive risk&quot;">​</a></h2><p>A high file count can be perfectly legitimate: partitioned output deliberately fans data across many files by partition column, so a large count there is by design, not a defect. And AQE coalescing is not a cure-all here, since it only affects post-shuffle partitions and leaves the input file layout untouched<sup class="footnote-ref"><a href="#fn6" id="fnref6:1">[6:1]</a></sup>. The signal reflects the shape, not intent, so confirm the write is not an intended partitioned layout before acting.</p><h2 id="related" tabindex="-1">Related <a class="header-anchor" href="#related" aria-label="Permalink to &quot;Related&quot;">​</a></h2><ul><li><strong>Partition sizing:</strong> <a href="./partitioning.html">Partitioning</a></li><li><strong>Too many small tasks:</strong> <a href="./bottleneck-tiny-tasks.html">Tiny Tasks</a></li></ul><hr class="footnotes-sep"><section class="footnotes"><ol class="footnotes-list"><li id="fn1" class="footnote-item"><p><em>Spark: The Definitive Guide</em>, Chambers &amp; Zaharia, ch. 19 <a href="#fnref1" class="footnote-backref">↩︎</a></p></li><li id="fn2" class="footnote-item"><p><em>Learning Spark, 2nd Edition</em>, Damji, Wenig, Das, Lee, ch. 7 <a href="#fnref2" class="footnote-backref">↩︎</a> <a href="#fnref2:1" class="footnote-backref">↩︎</a> <a href="#fnref2:2" class="footnote-backref">↩︎</a></p></li><li id="fn3" class="footnote-item"><p><a href="https://raw.githubusercontent.com/apache/spark/v3.5.0/sql/catalyst/src/main/scala/org/apache/spark/sql/internal/SQLConf.scala" target="_blank" rel="noreferrer">SQLConf.scala</a> <a href="#fnref3" class="footnote-backref">↩︎</a> <a href="#fnref3:1" class="footnote-backref">↩︎</a></p></li><li id="fn4" class="footnote-item"><p><a href="https://luminousmen.com/post/spark-partitions" target="_blank" rel="noreferrer">Spark Partitions</a> <a href="#fnref4" class="footnote-backref">↩︎</a> <a href="#fnref4:1" class="footnote-backref">↩︎</a> <a href="#fnref4:2" class="footnote-backref">↩︎</a> <a href="#fnref4:3" class="footnote-backref">↩︎</a> <a href="#fnref4:4" class="footnote-backref">↩︎</a> <a href="#fnref4:5" class="footnote-backref">↩︎</a> <a href="#fnref4:6" class="footnote-backref">↩︎</a> <a href="#fnref4:7" class="footnote-backref">↩︎</a> <a href="#fnref4:8" class="footnote-backref">↩︎</a></p></li><li id="fn5" class="footnote-item"><p><a href="https://spark.apache.org/docs/latest/sql-performance-tuning.html" target="_blank" rel="noreferrer">Performance Tuning (Spark SQL, DataFrames and Datasets Guide)</a> <a href="#fnref5" class="footnote-backref">↩︎</a> <a href="#fnref5:1" class="footnote-backref">↩︎</a> <a href="#fnref5:2" class="footnote-backref">↩︎</a></p></li><li id="fn6" class="footnote-item"><p><a href="https://luminousmen.com/post/the-apache-spark-optimization-checklist" target="_blank" rel="noreferrer">The Apache Spark Optimization Checklist</a> <a href="#fnref6" class="footnote-backref">↩︎</a> <a href="#fnref6:1" class="footnote-backref">↩︎</a></p></li></ol></section>`,24)])])}const u=t(o,[["render",n]]);export{p as __pageData,u as default};
@@ -0,0 +1 @@
1
+ import{_ as t,o as a,c as s,a5 as i}from"./chunks/framework.DSg0KOwT.js";const p=JSON.parse('{"title":"Small Files","description":"","frontmatter":{"title":"Small Files"},"headers":[],"relativePath":"tuning-reference/bottleneck-small-files.md","filePath":"tuning-reference/bottleneck-small-files.md"}'),o={name:"tuning-reference/bottleneck-small-files.md"};function n(r,e,f,l,h,c){return a(),s("div",null,[...e[0]||(e[0]=[i("",24)])])}const u=t(o,[["render",n]]);export{p as __pageData,u as default};
@@ -0,0 +1,6 @@
1
+ import{_ as a,o as t,c as o,a5 as s}from"./chunks/framework.DSg0KOwT.js";const i="../assets/spill-classification.BU2euYDO.svg",r="../assets/spill-classification.dark.D7i1M40d.svg",u=JSON.parse('{"title":"Memory / Disk Spill","description":"","frontmatter":{"title":"Memory / Disk Spill"},"headers":[],"relativePath":"tuning-reference/bottleneck-spill.md","filePath":"tuning-reference/bottleneck-spill.md"}'),n={name:"tuning-reference/bottleneck-spill.md"};function l(c,e,f,h,p,d){return t(),o("div",null,[...e[0]||(e[0]=[s('<h1 id="bottleneck-spill" tabindex="-1">Memory / Disk Spill <a class="header-anchor" href="#bottleneck-spill" aria-label="Permalink to &quot;Memory / Disk Spill {#bottleneck-spill}&quot;">​</a></h1><p><span class="tag">SPILL</span></p><h2 id="what-it-is" tabindex="-1">What it is <a class="header-anchor" href="#what-it-is" aria-label="Permalink to &quot;What it is&quot;">​</a></h2><p>Spill happens when a task&#39;s share of execution memory runs out during a sort or hash aggregation. Each task gets its own <code>TaskMemoryManager</code>, which arbitrates the shared execution pool across every task running concurrently on an executor; with <code>n</code> tasks running concurrently, each is allowed to allocate somewhere between <code>1/(2n)</code> and <code>1/n</code> of total execution memory: a soft limit, not a hard stop.<sup class="footnote-ref"><a href="#fn1" id="fnref1">[1]</a></sup> When a task&#39;s sort or hash-aggregation operator keeps requesting execution-memory pages it can&#39;t get, Spark doesn&#39;t fail immediately: it blocks the requesting task, triggers a spill of that task&#39;s in-memory data structure to disk, or, in the worst case, throws an <code>OutOfMemoryError</code>.<sup class="footnote-ref"><a href="#fn1" id="fnref1:1">[1:1]</a></sup></p><p>The spill path runs through Spark&#39;s map/shuffle I/O machinery: shuffle partitions created by wide transformations like <code>groupBy()</code> or <code>join()</code> spill to the executors&#39; local disks, at the location set by <code>spark.local.directory</code>.<sup class="footnote-ref"><a href="#fn2" id="fnref2">[2]</a></sup> SQL physical operators apply the same idea with their own guardrails: <code>SortMergeJoinExec</code>&#39;s in-memory buffer and the cartesian-product operator&#39;s buffer both spill once they cross a configured row-count threshold, which by default is <code>spark.shuffle.spill.numElementsForceSpillThreshold</code>.<sup class="footnote-ref"><a href="#fn3" id="fnref3">[3]</a></sup> Whatever the trigger, the resulting spilled bytes are exposed on the task metrics as <code>memoryBytesSpilled</code>, visible in the Spark UI and event log.<sup class="footnote-ref"><a href="#fn4" id="fnref4">[4]</a></sup></p><h2 id="how-it-s-detected" tabindex="-1">How it&#39;s detected <a class="header-anchor" href="#how-it-s-detected" aria-label="Permalink to &quot;How it&#39;s detected&quot;">​</a></h2><p>Detection: any <code>memoryBytesSpilled &gt; 0</code> → Warning. The classification is the actionable signal.</p><p>Spill classification:</p><ul><li><code>skew</code>: ≥ 80% of tasks have zero spill; fix: address <a href="./bottleneck-skew.html">task skew</a>, not memory.</li><li><code>volume</code>: &lt; 20% of tasks have zero spill; fix: more partitions or more memory.</li><li><code>unclassified</code>: neither condition; treat as volume.</li></ul><img class="light-only" src="'+i+'" alt="How a nonzero memoryBytesSpilled is classified as skew, volume, or unclassified from the share of tasks with zero spill, and the fix each classification points to."><img class="dark-only" src="'+r+`" alt="How a nonzero memoryBytesSpilled is classified as skew, volume, or unclassified from the share of tasks with zero spill, and the fix each classification points to."><h2 id="why-it-matters" tabindex="-1">Why it matters <a class="header-anchor" href="#why-it-matters" aria-label="Permalink to &quot;Why it matters&quot;">​</a></h2><p>Spill is the fallback Spark reaches for once a task can no longer get the execution memory it&#39;s asking for: rather than failing outright, it blocks the task, writes its buffered data to disk, or (if that&#39;s not enough) throws an <code>OutOfMemoryError</code>.<sup class="footnote-ref"><a href="#fn1" id="fnref1:2">[1:2]</a></sup> A task that&#39;s spilling has already given up pure in-memory speed to keep running at all, which is why the skew-vs-volume split above is the actionable part of the signal: the two failure shapes call for opposite fixes.</p><h2 id="how-to-fix-it" tabindex="-1">How to fix it <a class="header-anchor" href="#how-to-fix-it" aria-label="Permalink to &quot;How to fix it&quot;">​</a></h2><ul><li><strong>Skew</strong>: When only a few tasks spill, the problem is the shape of the data, not the size of the memory pool. <code>repartition()</code> is the tool for this: reach for it specifically to fix a lopsided partition distribution or to increase partition count<sup class="footnote-ref"><a href="#fn5" id="fnref5">[5]</a></sup>, in contrast to <code>coalesce()</code>, which never rebalances skewed data because it just stacks existing partitions together: partitions that were uneven going in are still uneven coming out.<sup class="footnote-ref"><a href="#fn6" id="fnref6">[6]</a></sup> For skewed joins, <a href="./aqe.html">Adaptive Query Execution</a> can detect and split<sup class="footnote-ref"><a href="#fn7" id="fnref7">[7]</a></sup> oversized partitions automatically: a partition counts as skewed once it&#39;s larger than <code>spark.sql.adaptive.skewJoin.skewedPartitionFactor</code> (default <code>5.0</code>) times the median partition size, and also larger than <code>spark.sql.adaptive.skewJoin.skewedPartitionThresholdInBytes</code> (default <code>256 MB</code>).<sup class="footnote-ref"><a href="#fn8" id="fnref8">[8]</a></sup><sup class="footnote-ref"><a href="#fn9" id="fnref9">[9]</a></sup> Before AQE existed, the manual equivalent was salting: adding a prefix to the skewed keys to make the same key look different, then adjusting the data distribution accordingly.<sup class="footnote-ref"><a href="#fn7" id="fnref7:1">[7:1]</a></sup> Databricks&#39; <code>/*+ SKEW(...) */</code> hint lets you name the skewed relation and specific key values directly, so the planner targets just those keys instead of relying on automatic detection.<sup class="footnote-ref"><a href="#fn10" id="fnref10">[10]</a></sup></li><li><strong>Volume</strong>: When most tasks spill, the fix is more room to work with: more partitions, more memory, or both. <code>repartition(n)</code> guarantees exactly <code>n</code> output partitions via a hash shuffle<sup class="footnote-ref"><a href="#fn11" id="fnref11">[11]</a></sup>, and since Spark&#39;s per-task startup overhead is low (unlike MapReduce&#39;s), the general bias is to err toward more partitions rather than fewer.<sup class="footnote-ref"><a href="#fn12" id="fnref12">[12]</a></sup><code>spark.sql.files.maxPartitionBytes</code> (default <code>128 MB</code>) is the equivalent lever on the input side, governing how much file data gets packed into each input partition before a shuffle even happens.<sup class="footnote-ref"><a href="#fn6" id="fnref6:1">[6:1]</a></sup> On the memory side, execution memory is a fraction of the JVM heap. <code>spark.memory.fraction</code> (default <code>0.6</code>) sizes the shared execution/storage region as <code>(JVM heap − 300 MiB) × spark.memory.fraction</code><sup class="footnote-ref"><a href="#fn13" id="fnref13">[13]</a></sup>. Because each task&#39;s slice of that region is capped by the <code>TaskMemoryManager</code> at between <code>1/(2n)</code> and <code>1/n</code> of the total,<sup class="footnote-ref"><a href="#fn1" id="fnref1:3">[1:3]</a></sup> giving executors more memory, or running fewer concurrent tasks on each one, directly raises the ceiling before spill kicks in.</li></ul><p>For the <strong>volume</strong> case, give tasks more room: defaults shown, comments say which way to move:</p><div class="language-properties vp-adaptive-theme"><button title="Copy Code" class="copy"></button><span class="lang">properties</span><pre class="shiki shiki-themes github-light github-dark vp-code" tabindex="0"><code><span class="line"><span style="--shiki-light:#6A737D;--shiki-dark:#6A737D;"># Create more, smaller input partitions before the shuffle (default 128m)</span></span>
2
+ <span class="line"><span style="--shiki-light:#D73A49;--shiki-dark:#F97583;">spark.sql.files.maxPartitionBytes</span><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;">=128m</span></span>
3
+ <span class="line"></span>
4
+ <span class="line"><span style="--shiki-light:#6A737D;--shiki-dark:#6A737D;"># Execution/storage share of (JVM heap - 300MiB) (default 0.6);</span></span>
5
+ <span class="line"><span style="--shiki-light:#6A737D;--shiki-dark:#6A737D;"># raising it grows execution memory but starves the untracked user-memory region</span></span>
6
+ <span class="line"><span style="--shiki-light:#D73A49;--shiki-dark:#F97583;">spark.memory.fraction</span><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;">=0.6</span></span></code></pre></div><p>For the <strong>skew</strong> case (only a few tasks spill), the fix is <code>df.repartition(n)</code>, not more memory: see <a href="./partitioning.html">Partitioning</a>.</p><h2 id="confidence" tabindex="-1">Confidence <a class="header-anchor" href="#confidence" aria-label="Permalink to &quot;Confidence&quot;">​</a></h2><p>The core detection (any <code>memoryBytesSpilled &gt; 0</code>) and the skew-vs-volume classification are validated: both the skew branch (most tasks spill nothing, so the shape of the data is the problem) and the volume branch (nearly every task spills, so the pool is too small) map to a well-understood fix, and neither needs further validation. <span class="tag">EXPERIMENTAL</span> When a spill matches neither shape, it falls back to a low-confidence unclassified finding that still requires validation, because there is no confirmed cause to act on; the page defaults it to the volume remedy, but that is a guess, not a diagnosis.</p><h2 id="limitations-false-positive-risk" tabindex="-1">Limitations / false-positive risk <a class="header-anchor" href="#limitations-false-positive-risk" aria-label="Permalink to &quot;Limitations / false-positive risk&quot;">​</a></h2><p>Some spill is normal. A large aggregation or join can spill by design once its working set outgrows execution memory, and a small spill on a healthy stage rarely repays the effort of chasing it. The raw <code>memoryBytesSpilled &gt; 0</code> trigger fires on both the actionable and the routine cases, so treat a small spill as informational until the skew-vs-volume split says otherwise.</p><h2 id="related" tabindex="-1">Related <a class="header-anchor" href="#related" aria-label="Permalink to &quot;Related&quot;">​</a></h2><ul><li><strong>Why it happens:</strong> <a href="./memory-model.html">Memory Management</a>, <a href="./partitioning.html">Partitioning</a></li></ul><hr class="footnotes-sep"><section class="footnotes"><ol class="footnotes-list"><li id="fn1" class="footnote-item"><p><a href="https://luminousmen.com/post/dive-into-spark-memory" target="_blank" rel="noreferrer">Diving into Spark Memory Management</a> <a href="#fnref1" class="footnote-backref">↩︎</a> <a href="#fnref1:1" class="footnote-backref">↩︎</a> <a href="#fnref1:2" class="footnote-backref">↩︎</a> <a href="#fnref1:3" class="footnote-backref">↩︎</a></p></li><li id="fn2" class="footnote-item"><p><em>Learning Spark, 2nd Edition</em>, Damji, Wenig, Das, Lee, ch. 7 <a href="#fnref2" class="footnote-backref">↩︎</a></p></li><li id="fn3" class="footnote-item"><p><a href="https://raw.githubusercontent.com/apache/spark/v3.5.0/sql/catalyst/src/main/scala/org/apache/spark/sql/internal/SQLConf.scala" target="_blank" rel="noreferrer">SQLConf.scala</a> <a href="#fnref3" class="footnote-backref">↩︎</a></p></li><li id="fn4" class="footnote-item"><p><a href="https://spark.apache.org/docs/latest/monitoring.html#spark-history-server" target="_blank" rel="noreferrer">Monitoring and Instrumentation</a> <a href="#fnref4" class="footnote-backref">↩︎</a></p></li><li id="fn5" class="footnote-item"><p><em>Spark: The Definitive Guide</em>, Chambers &amp; Zaharia, ch. 19 <a href="#fnref5" class="footnote-backref">↩︎</a></p></li><li id="fn6" class="footnote-item"><p><a href="https://luminousmen.com/post/spark-partitions" target="_blank" rel="noreferrer">Spark Partitions</a> <a href="#fnref6" class="footnote-backref">↩︎</a> <a href="#fnref6:1" class="footnote-backref">↩︎</a></p></li><li id="fn7" class="footnote-item"><p><a href="https://issues.apache.org/jira/browse/SPARK-29544" target="_blank" rel="noreferrer">SPARK-29544: Optimize Skewed Join in SQL Adaptive Execution</a> <a href="#fnref7" class="footnote-backref">↩︎</a> <a href="#fnref7:1" class="footnote-backref">↩︎</a></p></li><li id="fn8" class="footnote-item"><p><a href="https://spark.apache.org/docs/latest/configuration.html" target="_blank" rel="noreferrer">Configuration (Spark)</a> <a href="#fnref8" class="footnote-backref">↩︎</a></p></li><li id="fn9" class="footnote-item"><p><a href="https://spark.apache.org/docs/latest/sql-performance-tuning.html" target="_blank" rel="noreferrer">Performance Tuning (Spark SQL, DataFrames and Datasets Guide)</a> <a href="#fnref9" class="footnote-backref">↩︎</a></p></li><li id="fn10" class="footnote-item"><p><a href="https://docs.databricks.com/aws/en/archive/legacy/skew-join" target="_blank" rel="noreferrer">Skew Join Optimization</a> <a href="#fnref10" class="footnote-backref">↩︎</a></p></li><li id="fn11" class="footnote-item"><p><a href="https://raw.githubusercontent.com/apache/spark/v3.5.0/core/src/main/scala/org/apache/spark/rdd/RDD.scala" target="_blank" rel="noreferrer">RDD.scala</a> <a href="#fnref11" class="footnote-backref">↩︎</a></p></li><li id="fn12" class="footnote-item"><p><a href="https://blog.cloudera.com/how-to-tune-your-apache-spark-jobs-part-2/" target="_blank" rel="noreferrer">How to Tune Your Apache Spark Jobs (Part 2): Cloudera Engineering Blog</a> <a href="#fnref12" class="footnote-backref">↩︎</a></p></li><li id="fn13" class="footnote-item"><p><a href="https://spark.apache.org/docs/latest/tuning.html" target="_blank" rel="noreferrer">Spark Tuning Guide</a> <a href="#fnref13" class="footnote-backref">↩︎</a></p></li></ol></section>`,26)])])}const k=a(n,[["render",l]]);export{u as __pageData,k as default};
@@ -0,0 +1 @@
1
+ import{_ as a,o as t,c as o,a5 as s}from"./chunks/framework.DSg0KOwT.js";const i="../assets/spill-classification.BU2euYDO.svg",r="../assets/spill-classification.dark.D7i1M40d.svg",u=JSON.parse('{"title":"Memory / Disk Spill","description":"","frontmatter":{"title":"Memory / Disk Spill"},"headers":[],"relativePath":"tuning-reference/bottleneck-spill.md","filePath":"tuning-reference/bottleneck-spill.md"}'),n={name:"tuning-reference/bottleneck-spill.md"};function l(c,e,f,h,p,d){return t(),o("div",null,[...e[0]||(e[0]=[s("",26)])])}const k=a(n,[["render",l]]);export{u as __pageData,k as default};
@@ -0,0 +1,7 @@
1
+ import{_ as a,o as t,c as s,a5 as o}from"./chunks/framework.DSg0KOwT.js";const p=JSON.parse('{"title":"Stragglers","description":"","frontmatter":{"title":"Stragglers"},"headers":[],"relativePath":"tuning-reference/bottleneck-straggler.md","filePath":"tuning-reference/bottleneck-straggler.md"}'),i={name:"tuning-reference/bottleneck-straggler.md"};function n(r,e,l,h,c,f){return t(),s("div",null,[...e[0]||(e[0]=[o(`<h1 id="bottleneck-straggler" tabindex="-1">Stragglers <a class="header-anchor" href="#bottleneck-straggler" aria-label="Permalink to &quot;Stragglers {#bottleneck-straggler}&quot;">​</a></h1><p><span class="tag">STRAG</span></p><h2 id="what-it-is" tabindex="-1">What it is <a class="header-anchor" href="#what-it-is" aria-label="Permalink to &quot;What it is&quot;">​</a></h2><p>A straggler is one task (or a small handful of tasks) that runs far longer than the rest of the tasks in its stage, even when the stage is otherwise healthy. Because a stage only completes once its last task finishes, a single straggler holds up the whole stage, and every other executor sits idle waiting for it to catch up.</p><h2 id="how-it-s-detected" tabindex="-1">How it&#39;s detected <a class="header-anchor" href="#how-it-s-detected" aria-label="Permalink to &quot;How it&#39;s detected&quot;">​</a></h2><p>A task counts as a straggler when speculative tasks fired for it, or when more than 5% of the tasks in a stage run at least 4× the stage&#39;s median task duration; this rule only applies once a stage has at least 10 tasks.</p><p>&quot;Duration&quot; here is <code>executorRunTime</code>: elapsed wall-clock time, not CPU time, and it already includes any time the task spent blocked fetching shuffle data<sup class="footnote-ref"><a href="#fn1" id="fnref1">[1]</a></sup>. That matters when tracking down why a task is slow: a long <code>executorRunTime</code> doesn&#39;t necessarily mean more compute happened. A <a href="./bottleneck-gc.html">garbage-collection pause</a> is counted inside it rather than added on top (<code>jvmGCTime</code> is a subset of <code>executorRunTime</code>, not additive<sup class="footnote-ref"><a href="#fn1" id="fnref1:1">[1:1]</a></sup>), and a task waiting on a remote shuffle block it needs next shows up in <code>shuffleReadMetrics.fetchWaitTime</code>, which only counts genuine blocking time, not blocks being prefetched in the background<sup class="footnote-ref"><a href="#fn1" id="fnref1:2">[1:2]</a></sup>.</p><p>The &quot;speculative tasks fired&quot; half of the rule is Spark&#39;s own detector: once <code>spark.speculation.quantile</code> (default <code>0.9</code>) of a stage&#39;s tasks finish, Spark compares each remaining task&#39;s duration against <code>spark.speculation.multiplier</code> (default <code>3</code>) times the median of the tasks that already finished, subject to a <code>spark.speculation.minTaskRuntime</code> floor (default <code>100ms</code>) so short tasks aren&#39;t flagged just for being slower than a tiny median<sup class="footnote-ref"><a href="#fn2" id="fnref2">[2]</a></sup>. Speculation itself is off by default (<code>spark.speculation</code> defaults to <code>false</code>), so this half of the rule only fires on stages where it&#39;s been turned on<sup class="footnote-ref"><a href="#fn2" id="fnref2:1">[2:1]</a></sup>.</p><p>Watch for overlap with <a href="./bottleneck-skew.html">task skew</a>: an unevenly distributed key sends one partition far more data than its peers, and that partition&#39;s task will trip this same duration threshold even though the underlying cause is data volume, not a slow host or a GC pause. Check the skew signals before assuming the latter.</p><h2 id="why-it-matters" tabindex="-1">Why it matters <a class="header-anchor" href="#why-it-matters" aria-label="Permalink to &quot;Why it matters&quot;">​</a></h2><p>A straggler wastes cluster capacity the same way a skewed stage does: every executor other than the one running the slow task finishes early and idles, while total job time still tracks the single slowest task. Because a straggler&#39;s duration can be inflated by a GC pause<sup class="footnote-ref"><a href="#fn1" id="fnref1:3">[1:3]</a></sup> or a slow <a href="./shuffle.html">shuffle</a> fetch<sup class="footnote-ref"><a href="#fn1" id="fnref1:4">[1:4]</a></sup> rather than genuinely more work, it&#39;s worth checking those angles (and whether the real cause is skew rather than anything task-local) before assuming a hardware explanation.</p><h2 id="how-to-fix-it" tabindex="-1">How to fix it <a class="header-anchor" href="#how-to-fix-it" aria-label="Permalink to &quot;How to fix it&quot;">​</a></h2><ul><li>Enable speculative execution: <code>spark.speculation</code> is <code>false</code> by default, so nothing reruns a slow task automatically until it&#39;s turned on<sup class="footnote-ref"><a href="#fn2" id="fnref2:2">[2:2]</a></sup>.</li><li>Tune the trigger: <code>spark.speculation.quantile</code> (default <code>0.9</code>) sets how much of the stage must finish before speculation kicks in, and <code>spark.speculation.multiplier</code> (default <code>3</code>) sets how many times slower than the median a task must be. For stages with very few tasks, <code>spark.speculation.task.duration.threshold</code> (available since 3.0.0) gives an absolute-duration trigger instead of relying on the median<sup class="footnote-ref"><a href="#fn2" id="fnref2:3">[2:3]</a></sup>.</li><li>Since Spark 3.4, <code>spark.speculation.efficiency.enabled</code> (default <code>true</code>) adds a filter so a task is only speculated if its data-process rate is below the stage average (times <code>spark.speculation.efficiency.processRateMultiplier</code>, default <code>0.75</code>) or its duration exceeds <code>spark.speculation.efficiency.longRunTaskFactor</code> (default <code>2</code>) times the same time threshold. This avoids wasting a duplicate task slot on work that&#39;s simply doing more, not running slower<sup class="footnote-ref"><a href="#fn2" id="fnref2:4">[2:4]</a></sup>.</li><li>If the cause is really a skewed key, treat it as <a href="./bottleneck-skew.html">task skew</a> instead: AQE&#39;s skew-join optimization automatically splits a partition once it&#39;s larger than <code>spark.sql.adaptive.skewJoin.skewedPartitionFactor</code> (default <code>5.0</code>) times the median partition size and above <code>spark.sql.adaptive.skewJoin.skewedPartitionThresholdInBytes</code> (default <code>256 MB</code>)<sup class="footnote-ref"><a href="#fn2" id="fnref2:5">[2:5]</a></sup><sup class="footnote-ref"><a href="#fn3" id="fnref3">[3]</a></sup>. Manual salting (appending a random prefix to the skewed key so it spreads across more partitions) is the pre-AQE fallback, though the design doc behind AQE&#39;s skew handling calls salting and other manual approaches limited compared to the automatic option<sup class="footnote-ref"><a href="#fn4" id="fnref4">[4]</a></sup>.</li></ul><blockquote><p><strong>PySpark:</strong> speculation settings can be set on the session directly, no <code>spark-submit</code> flag needed: <code>spark.conf.set(&quot;spark.speculation&quot;, &quot;true&quot;)</code>, then the quantile/multiplier equivalents the same way.</p></blockquote><p>Speculation config: defaults shown, plus the absolute-duration trigger for small stages:</p><div class="language-properties vp-adaptive-theme"><button title="Copy Code" class="copy"></button><span class="lang">properties</span><pre class="shiki shiki-themes github-light github-dark vp-code" tabindex="0"><code><span class="line"><span style="--shiki-light:#6A737D;--shiki-dark:#6A737D;"># Speculatively relaunch straggler tasks (OFF by default)</span></span>
2
+ <span class="line"><span style="--shiki-light:#D73A49;--shiki-dark:#F97583;">spark.speculation</span><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;">=true</span></span>
3
+ <span class="line"><span style="--shiki-light:#D73A49;--shiki-dark:#F97583;">spark.speculation.quantile</span><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;">=0.9 </span><span style="--shiki-light:#6A737D;--shiki-dark:#6A737D;"># default; portion of tasks finished before speculation begins</span></span>
4
+ <span class="line"><span style="--shiki-light:#D73A49;--shiki-dark:#F97583;">spark.speculation.multiplier</span><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;">=3 </span><span style="--shiki-light:#6A737D;--shiki-dark:#6A737D;"># default; multiple of median duration that marks a task slow</span></span>
5
+ <span class="line"></span>
6
+ <span class="line"><span style="--shiki-light:#6A737D;--shiki-dark:#6A737D;"># Absolute-duration trigger for stages with very few tasks (since Spark 3.0)</span></span>
7
+ <span class="line"><span style="--shiki-light:#D73A49;--shiki-dark:#F97583;">spark.speculation.task.duration.threshold</span><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;">=10s </span><span style="--shiki-light:#6A737D;--shiki-dark:#6A737D;"># 10s is an example</span></span></code></pre></div><h2 id="confidence" tabindex="-1">Confidence <a class="header-anchor" href="#confidence" aria-label="Permalink to &quot;Confidence&quot;">​</a></h2><p>Validated.</p><h2 id="limitations-false-positive-risk" tabindex="-1">Limitations / false-positive risk <a class="header-anchor" href="#limitations-false-positive-risk" aria-label="Permalink to &quot;Limitations / false-positive risk&quot;">​</a></h2><p>The 4x-median rule flags a slow task, but slow is not the same as broken. The same threshold trips on a skewed key that simply has more data to process, or on a task that spent its time in a GC pause rather than doing extra work, so a flagged task is not automatically a slow host. Small stages make this worse: with only a handful of tasks the median is unstable, and one moderately slow task can look like a straggler against a median computed from too few peers.</p><h2 id="related" tabindex="-1">Related <a class="header-anchor" href="#related" aria-label="Permalink to &quot;Related&quot;">​</a></h2><ul><li><strong>When the real cause is a skewed key:</strong> <a href="./partitioning.html">Partitioning</a>, <a href="./aqe.html">Adaptive Query Execution</a></li></ul><hr class="footnotes-sep"><section class="footnotes"><ol class="footnotes-list"><li id="fn1" class="footnote-item"><p><a href="https://spark.apache.org/docs/latest/monitoring.html#spark-history-server" target="_blank" rel="noreferrer">Monitoring and Instrumentation</a> <a href="#fnref1" class="footnote-backref">↩︎</a> <a href="#fnref1:1" class="footnote-backref">↩︎</a> <a href="#fnref1:2" class="footnote-backref">↩︎</a> <a href="#fnref1:3" class="footnote-backref">↩︎</a> <a href="#fnref1:4" class="footnote-backref">↩︎</a></p></li><li id="fn2" class="footnote-item"><p><a href="https://spark.apache.org/docs/latest/configuration.html" target="_blank" rel="noreferrer">Configuration (Spark)</a> <a href="#fnref2" class="footnote-backref">↩︎</a> <a href="#fnref2:1" class="footnote-backref">↩︎</a> <a href="#fnref2:2" class="footnote-backref">↩︎</a> <a href="#fnref2:3" class="footnote-backref">↩︎</a> <a href="#fnref2:4" class="footnote-backref">↩︎</a> <a href="#fnref2:5" class="footnote-backref">↩︎</a></p></li><li id="fn3" class="footnote-item"><p><a href="https://spark.apache.org/docs/latest/sql-performance-tuning.html#optimizing-skew-join" target="_blank" rel="noreferrer">Optimizing Skew Join (Spark SQL, DataFrames and Datasets Guide)</a> <a href="#fnref3" class="footnote-backref">↩︎</a></p></li><li id="fn4" class="footnote-item"><p><a href="https://issues.apache.org/jira/browse/SPARK-29544" target="_blank" rel="noreferrer">SPARK-29544: Optimize Skewed Join at Runtime with New Adaptive Execution</a> <a href="#fnref4" class="footnote-backref">↩︎</a></p></li></ol></section>`,24)])])}const u=a(i,[["render",n]]);export{p as __pageData,u as default};
@@ -0,0 +1 @@
1
+ import{_ as a,o as t,c as s,a5 as o}from"./chunks/framework.DSg0KOwT.js";const p=JSON.parse('{"title":"Stragglers","description":"","frontmatter":{"title":"Stragglers"},"headers":[],"relativePath":"tuning-reference/bottleneck-straggler.md","filePath":"tuning-reference/bottleneck-straggler.md"}'),i={name:"tuning-reference/bottleneck-straggler.md"};function n(r,e,l,h,c,f){return t(),s("div",null,[...e[0]||(e[0]=[o("",24)])])}const u=a(i,[["render",n]]);export{p as __pageData,u as default};
@@ -0,0 +1,8 @@
1
+ import{_ as t,o as a,c as s,a5 as i}from"./chunks/framework.DSg0KOwT.js";const p=JSON.parse('{"title":"Tiny Tasks","description":"","frontmatter":{"title":"Tiny Tasks"},"headers":[],"relativePath":"tuning-reference/bottleneck-tiny-tasks.md","filePath":"tuning-reference/bottleneck-tiny-tasks.md"}'),o={name:"tuning-reference/bottleneck-tiny-tasks.md"};function n(r,e,l,h,f,d){return a(),s("div",null,[...e[0]||(e[0]=[i(`<h1 id="bottleneck-tiny-tasks" tabindex="-1">Tiny Tasks <a class="header-anchor" href="#bottleneck-tiny-tasks" aria-label="Permalink to &quot;Tiny Tasks {#bottleneck-tiny-tasks}&quot;">​</a></h1><p><span class="tag">TINY</span></p><h2 id="what-it-is" tabindex="-1">What it is <a class="header-anchor" href="#what-it-is" aria-label="Permalink to &quot;What it is&quot;">​</a></h2><p>Every task pays a fixed placement and serialization cost before it does any real work. Spark schedules around data locality rather than moving data to code: &quot;Spark builds its scheduling around this general principle&quot; that shipping serialized code is cheaper than shipping data, preferring <code>PROCESS_LOCAL</code> locality down to <code>ANY</code> and waiting a configurable timeout for a busy CPU to free up before shipping data to a farther executor<sup class="footnote-ref"><a href="#fn1" id="fnref1">[1]</a></sup>. On top of that placement cost, every task also pays a serialization cost for its closure and, if not using Kryo, for its data: Java&#39;s default <code>ObjectOutputStream</code>-based serialization &quot;is flexible but often quite slow, and leads to large serialized formats,&quot; while Kryo is &quot;significantly faster and more compact than Java serialization (often as much as 10x)&quot;<sup class="footnote-ref"><a href="#fn1" id="fnref1:1">[1:1]</a></sup>. None of that overhead is large per task: Spark&#39;s own guidance leans toward more, smaller tasks rather than fewer, larger ones, unlike MapReduce: &quot;it&#39;s almost always better to err on the side of a larger number of tasks (and thus partitions)... The difference stems from the fact that MapReduce has a high startup overhead for tasks, while Spark does not&quot;<sup class="footnote-ref"><a href="#fn2" id="fnref2">[2]</a></sup>. But once partitions get small enough, that per-task overhead stops being negligible and starts to dominate.</p><h2 id="how-it-s-detected" tabindex="-1">How it&#39;s detected <a class="header-anchor" href="#how-it-s-detected" aria-label="Permalink to &quot;How it&#39;s detected&quot;">​</a></h2><p>The failure mode is partitions so small that scheduling and serialization overhead outweighs the actual work. With Spark&#39;s default of 200 shuffle partitions applied to a dataset of only a few megabytes, &quot;those 200 partitions will each get like ten rows. Tasks become microscopic. Most of your CPUs will just be sitting there doing nothing, wasting cluster hours&quot;<sup class="footnote-ref"><a href="#fn3" id="fnref3">[3]</a></sup>. The same small-partition penalty shows up on the <a href="./bottleneck-shuffle.html">shuffle-read side</a>: production measurements found &quot;the average shuffle block size is only 10s of KBs, which leads to delayed shuffle data fetch,&quot; and the shuffle reduce stages with the largest fetch delays (over 30 seconds per task) were consistently the ones with small block sizes<sup class="footnote-ref"><a href="#fn4" id="fnref4">[4]</a></sup>. A high task count paired with very short median task duration, and/or very small shuffle block sizes on the read side, are the signals to look for.</p><h2 id="why-it-matters" tabindex="-1">Why it matters <a class="header-anchor" href="#why-it-matters" aria-label="Permalink to &quot;Why it matters&quot;">​</a></h2><p>Tiny tasks waste cluster capacity even though no single task is a straggler the way <a href="./bottleneck-skew.html">skewed</a> or <a href="./bottleneck-straggler.html">straggler</a> tasks are: the cost here is spread evenly across thousands of tasks, each individually cheap but collectively adding up in scheduling and serialization overhead, while cores that could be doing other useful work sit mostly idle between task launches<sup class="footnote-ref"><a href="#fn3" id="fnref3:1">[3:1]</a></sup>.</p><h2 id="how-to-fix-it" tabindex="-1">How to fix it <a class="header-anchor" href="#how-to-fix-it" aria-label="Permalink to &quot;How to fix it&quot;">​</a></h2><ul><li><code>coalesce()</code> merges existing partitions without a shuffle (&quot;no data movement, no shuffle&quot;) and is typically used right before a write, e.g. <code>df.coalesce(100).write.parquet(...)</code>, to collapse many tiny output files into a manageable number cheaply<sup class="footnote-ref"><a href="#fn3" id="fnref3:2">[3:2]</a></sup>. Because it only stacks partitions together rather than redistributing data, it doesn&#39;t fix an uneven underlying distribution (uneven input partitions stay uneven, just grouped) and pushing it too far (<code>coalesce(1)</code>) kills parallelism by funneling all work onto one executor while the rest sit idle<sup class="footnote-ref"><a href="#fn3" id="fnref3:3">[3:3]</a></sup>.</li><li><code>repartition()</code> performs a full reshuffle, giving control and rebalancing that Spark won&#39;t do on its own: with <code>df.repartition(200)</code>, &quot;you&#39;re paying for predictability&quot;<sup class="footnote-ref"><a href="#fn3" id="fnref3:4">[3:4]</a></sup>. Use it when the partitioning itself needs to be fixed or rebalanced, not merely reduced in count, accepting the shuffle cost to get there.</li><li>There&#39;s a middle option, <code>coalesce(N, shuffle=True)</code>, which &quot;acts more like a repartition, but leaning toward reduction... not free (you pay for the shuffle cost) but you get better distribution and fewer partitions&quot;<sup class="footnote-ref"><a href="#fn3" id="fnref3:5">[3:5]</a></sup>.</li><li>Rule of thumb: reach for <code>coalesce()</code> when only the partition count needs to shrink and the existing distribution is already reasonably even (e.g. collapsing output files); reach for <code>repartition()</code> when the distribution itself is the problem.</li></ul><blockquote><p><strong>PySpark:</strong> both are one-line calls on a DataFrame: <code>df.coalesce(100)</code> for a shuffle-free merge, <code>df.repartition(200)</code> for a full reshuffle, or <code>df.coalesce(100, shuffle=True)</code> for the shuffled middle option.</p></blockquote><p>The three sizing calls, side by side:</p><div class="language-python vp-adaptive-theme"><button title="Copy Code" class="copy"></button><span class="lang">python</span><pre class="shiki shiki-themes github-light github-dark vp-code" tabindex="0"><code><span class="line"><span style="--shiki-light:#6A737D;--shiki-dark:#6A737D;"># Collapse many tiny output files without a shuffle (use when distribution is already even)</span></span>
2
+ <span class="line"><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;">df.coalesce(</span><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;">100</span><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;">).write.parquet(path)</span></span>
3
+ <span class="line"></span>
4
+ <span class="line"><span style="--shiki-light:#6A737D;--shiki-dark:#6A737D;"># Full reshuffle to a target count, use when the distribution itself needs rebalancing</span></span>
5
+ <span class="line"><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;">df </span><span style="--shiki-light:#D73A49;--shiki-dark:#F97583;">=</span><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;"> df.repartition(</span><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;">200</span><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;">)</span></span>
6
+ <span class="line"></span>
7
+ <span class="line"><span style="--shiki-light:#6A737D;--shiki-dark:#6A737D;"># Middle ground: reduce partition count but still rebalance (pays a shuffle)</span></span>
8
+ <span class="line"><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;">df </span><span style="--shiki-light:#D73A49;--shiki-dark:#F97583;">=</span><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;"> df.coalesce(</span><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;">100</span><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;">, </span><span style="--shiki-light:#E36209;--shiki-dark:#FFAB70;">shuffle</span><span style="--shiki-light:#D73A49;--shiki-dark:#F97583;">=</span><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;">True</span><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;">)</span></span></code></pre></div><h2 id="limitations-false-positive-risk" tabindex="-1">Limitations / false-positive risk <a class="header-anchor" href="#limitations-false-positive-risk" aria-label="Permalink to &quot;Limitations / false-positive risk&quot;">​</a></h2><p>Very short tasks are not always a problem. A tiny final stage can be perfectly fine, and one that just writes a small result set doesn&#39;t need to be resized. The scheduling-overhead penalty only bites when tiny tasks dominate the stage, so treat a handful of short tasks as noise rather than a finding.</p><h2 id="related" tabindex="-1">Related <a class="header-anchor" href="#related" aria-label="Permalink to &quot;Related&quot;">​</a></h2><ul><li><strong>Partition sizing:</strong> <a href="./partitioning.html">Partitioning</a></li></ul><hr class="footnotes-sep"><section class="footnotes"><ol class="footnotes-list"><li id="fn1" class="footnote-item"><p><a href="https://spark.apache.org/docs/latest/tuning.html" target="_blank" rel="noreferrer">Tuning (Spark)</a> <a href="#fnref1" class="footnote-backref">↩︎</a> <a href="#fnref1:1" class="footnote-backref">↩︎</a></p></li><li id="fn2" class="footnote-item"><p><a href="https://blog.cloudera.com/how-to-tune-your-apache-spark-jobs-part-2/" target="_blank" rel="noreferrer">How to Tune Your Apache Spark Jobs (Part 2): Cloudera</a> <a href="#fnref2" class="footnote-backref">↩︎</a></p></li><li id="fn3" class="footnote-item"><p><a href="https://luminousmen.com/post/spark-partitions" target="_blank" rel="noreferrer">Spark Partitions</a> <a href="#fnref3" class="footnote-backref">↩︎</a> <a href="#fnref3:1" class="footnote-backref">↩︎</a> <a href="#fnref3:2" class="footnote-backref">↩︎</a> <a href="#fnref3:3" class="footnote-backref">↩︎</a> <a href="#fnref3:4" class="footnote-backref">↩︎</a> <a href="#fnref3:5" class="footnote-backref">↩︎</a></p></li><li id="fn4" class="footnote-item"><p><a href="https://issues.apache.org/jira/browse/SPARK-30602" target="_blank" rel="noreferrer">SPARK-30602</a> <a href="#fnref4" class="footnote-backref">↩︎</a></p></li></ol></section>`,19)])])}const u=t(o,[["render",n]]);export{p as __pageData,u as default};
@@ -0,0 +1 @@
1
+ import{_ as t,o as a,c as s,a5 as i}from"./chunks/framework.DSg0KOwT.js";const p=JSON.parse('{"title":"Tiny Tasks","description":"","frontmatter":{"title":"Tiny Tasks"},"headers":[],"relativePath":"tuning-reference/bottleneck-tiny-tasks.md","filePath":"tuning-reference/bottleneck-tiny-tasks.md"}'),o={name:"tuning-reference/bottleneck-tiny-tasks.md"};function n(r,e,l,h,f,d){return a(),s("div",null,[...e[0]||(e[0]=[i("",19)])])}const u=t(o,[["render",n]]);export{p as __pageData,u as default};
@@ -0,0 +1,5 @@
1
+ import{_ as a,a as t}from"./chunks/duplicate-plan-subtree.dark.Cdp70QhV.js";import{_ as o,o as s,c as r,a5 as i}from"./chunks/framework.DSg0KOwT.js";const k=JSON.parse('{"title":"Executor Utilization","description":"","frontmatter":{"title":"Executor Utilization"},"headers":[],"relativePath":"tuning-reference/bottleneck-utilization.md","filePath":"tuning-reference/bottleneck-utilization.md"}'),n={name:"tuning-reference/bottleneck-utilization.md"};function l(c,e,f,h,d,p){return s(),r("div",null,[...e[0]||(e[0]=[i(`<h1 id="bottleneck-utilization" tabindex="-1">Executor Utilization <a class="header-anchor" href="#bottleneck-utilization" aria-label="Permalink to &quot;Executor Utilization {#bottleneck-utilization}&quot;">​</a></h1><p><span class="tag">UTIL</span></p><h2 id="what-it-is" tabindex="-1">What it is <a class="header-anchor" href="#what-it-is" aria-label="Permalink to &quot;What it is&quot;">​</a></h2><p>Executor utilization measures how much of the cluster&#39;s allocated executor capacity a job actually keeps busy. When the average number of active executors trails the peak number allocated, the cluster is holding compute (cores and memory) that isn&#39;t running any tasks.</p><h2 id="how-it-s-detected" tabindex="-1">How it&#39;s detected <a class="header-anchor" href="#how-it-s-detected" aria-label="Permalink to &quot;How it&#39;s detected&quot;">​</a></h2><p>The signal is the ratio of average active executors to the peak active executor count observed over the job&#39;s lifetime:</p><table tabindex="0"><thead><tr><th>avg active executors / peak</th><th>Level</th></tr></thead><tbody><tr><td>&lt; 60%</td><td>Info</td></tr><tr><td>&lt; 40%</td><td>Warning</td></tr><tr><td>&lt; 20%</td><td>Critical</td></tr></tbody></table><h2 id="why-it-matters" tabindex="-1">Why it matters <a class="header-anchor" href="#why-it-matters" aria-label="Permalink to &quot;Why it matters&quot;">​</a></h2><p>Idle executors are allocation you&#39;re paying for without getting work done. One common cause is too little task parallelism to occupy the allocated cores: the guidance is to keep at least as many partitions as there are cores across the executors, so no core sits idle<sup class="footnote-ref"><a href="#fn1" id="fnref1">[1]</a></sup>. Parallelism can also be lost by accident rather than by under-partitioning upfront: because <code>coalesce</code> is a narrow transformation, reducing partition count with it forces the <em>entire</em> upstream stage down to the reduced parallelism, not just the coalesce step, trading a shuffle for lost concurrency<sup class="footnote-ref"><a href="#fn2" id="fnref2">[2]</a></sup>. Separately, under dynamic allocation, a workload with many small tasks can end up over-provisioned: by default it requests enough executors to maximize parallelism for the task count, and with small tasks that can mean some executors &quot;might not even do any work,&quot; wasting resources on allocation overhead<sup class="footnote-ref"><a href="#fn3" id="fnref3">[3]</a></sup>.</p><h2 id="how-to-fix-it" tabindex="-1">How to fix it <a class="header-anchor" href="#how-to-fix-it" aria-label="Permalink to &quot;How to fix it&quot;">​</a></h2><ul><li>Reach for <code>repartition</code> (not <code>coalesce</code>) when you need to raise partition count or rebalance data: <code>coalesce</code> only avoids a shuffle when shrinking partition count, and doing so also drags the entire upstream stage down to the reduced parallelism<sup class="footnote-ref"><a href="#fn4" id="fnref4">[4]</a></sup><sup class="footnote-ref"><a href="#fn2" id="fnref2:1">[2:1]</a></sup>.</li><li>Size partitions so the task count is at least the core count available across executors, to avoid leaving cores idle<sup class="footnote-ref"><a href="#fn1" id="fnref1:1">[1:1]</a></sup>.</li><li>Under dynamic allocation, lower <code>spark.dynamicAllocation.executorAllocationRatio</code> below its default of <code>1.0</code> to scale back the number of executors requested for workloads made up of many small tasks<sup class="footnote-ref"><a href="#fn3" id="fnref3:1">[3:1]</a></sup>.</li><li>Dynamic allocation requires shuffle tracking, the external shuffle service, or shuffle-block decommissioning to be enabled as a precondition<sup class="footnote-ref"><a href="#fn3" id="fnref3:2">[3:2]</a></sup>; without one of them, an executor removed mid-shuffle takes its unfetched shuffle files with it, forcing a recompute<sup class="footnote-ref"><a href="#fn5" id="fnref5">[5]</a></sup>, which undercuts using dynamic allocation to shed idle executors in the first place. <code>spark.dynamicAllocation.shuffleTracking.enabled</code> defaults to <code>true</code> since Spark 3.0, satisfying that precondition without needing a separate external shuffle service<sup class="footnote-ref"><a href="#fn3" id="fnref3:3">[3:3]</a></sup>.</li></ul><p>Restore parallelism after a heavy filter, and rein in over-provisioning for many-small-task jobs:</p><div class="language-python vp-adaptive-theme"><button title="Copy Code" class="copy"></button><span class="lang">python</span><pre class="shiki shiki-themes github-light github-dark vp-code" tabindex="0"><code><span class="line"><span style="--shiki-light:#6A737D;--shiki-dark:#6A737D;"># repartition (not coalesce) rebalances and can raise partition count; aim &gt;= total executor cores</span></span>
2
+ <span class="line"><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;">df </span><span style="--shiki-light:#D73A49;--shiki-dark:#F97583;">=</span><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;"> df.filter(heavy_predicate).repartition(spark.sparkContext.defaultParallelism)</span></span>
3
+ <span class="line"></span>
4
+ <span class="line"><span style="--shiki-light:#6A737D;--shiki-dark:#6A737D;"># Scale back executors requested for many-small-task workloads (default ratio 1.0)</span></span>
5
+ <span class="line"><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;">spark.conf.set(</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;">&quot;spark.dynamicAllocation.executorAllocationRatio&quot;</span><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;">, </span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;">&quot;0.5&quot;</span><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;">) </span><span style="--shiki-light:#6A737D;--shiki-dark:#6A737D;"># 0.5 is an example</span></span></code></pre></div><h2 id="limitations-false-positive-risk" tabindex="-1">Limitations / false-positive risk <a class="header-anchor" href="#limitations-false-positive-risk" aria-label="Permalink to &quot;Limitations / false-positive risk&quot;">​</a></h2><p>A low average-to-peak ratio isn&#39;t always waste. A bursty or I/O-bound job legitimately holds executors while tasks wait on external systems rather than burning cores, and the ratio is sensitive to short stages, where a brief spike in allocation skews the average without meaning the cluster was genuinely idle.</p><h2 id="caching-opportunity" tabindex="-1">Caching opportunity <a class="header-anchor" href="#caching-opportunity" aria-label="Permalink to &quot;Caching opportunity&quot;">​</a></h2><p><span class="tag">CACHE</span></p><p>When the same DataFrame, RDD, or input is scanned more than once, low utilization can trace back to repeated recomputation rather than idle cores. Spark keeps nothing between actions: transformations only build a DAG, and once an action finishes its intermediate results are discarded<sup class="footnote-ref"><a href="#fn6" id="fnref6">[6]</a></sup>. Call a second action on the same logic and Spark re-runs the whole DAG from the source, which can mean re-reading a terabyte from S3, re-reading Kafka, or repeating expensive decompression<sup class="footnote-ref"><a href="#fn6" id="fnref6:1">[6:1]</a></sup>. Fork that logic into two pipeline branches and you sign up to recompute everything twice<sup class="footnote-ref"><a href="#fn6" id="fnref6:2">[6:2]</a></sup>.</p><img class="light-only" src="`+a+'" alt="How two branches that repeat the same scan and operators each recompute it, until a shared cached or reused node lets both read one materialized result."><img class="dark-only" src="'+t+'" alt="How two branches that repeat the same scan and operators each recompute it, until a shared cached or reused node lets both read one materialized result."><p><a href="./caching.html"><code>cache()</code> and <code>persist()</code></a> are how you stop that. Persisting materializes the RDD (usually in memory on the executors) so it can be reused within the job, while Spark keeps its lineage to recompute any lost partition<sup class="footnote-ref"><a href="#fn7" id="fnref7">[7]</a></sup>. For a dataset you read repeatedly, this is one of the most useful optimizations available: it parks a DataFrame, table, or RDD in temporary storage across the executors and makes later reads fast<sup class="footnote-ref"><a href="#fn8" id="fnref8">[8]</a></sup>. Materialization is lazy and per-block, so only an action that touches every partition (<code>count()</code> or a full write) caches all of it; a subset scan like <code>take(10)</code> leaves a partial cache<sup class="footnote-ref"><a href="#fn6" id="fnref6:3">[6:3]</a></sup>.</p><p><code>cache()</code> is shorthand; <code>persist()</code> takes a <code>StorageLevel</code> and lets you pick the tradeoff:</p><table tabindex="0"><thead><tr><th>Level</th><th>Tradeoff</th></tr></thead><tbody><tr><td><code>MEMORY_ONLY</code></td><td>Fastest reads (deserialized JVM objects), but partitions that don&#39;t fit are recomputed on the fly each time<sup class="footnote-ref"><a href="#fn8" id="fnref8:1">[8:1]</a></sup>.</td></tr><tr><td><code>MEMORY_AND_DISK</code></td><td>Spills the overflow to disk instead of recomputing; Spark&#39;s default caching strategy and fine for most pipelines<sup class="footnote-ref"><a href="#fn9" id="fnref9">[9]</a></sup>. For DataFrames, <code>.cache()</code> maps to this<sup class="footnote-ref"><a href="#fn6" id="fnref6:4">[6:4]</a></sup>.</td></tr><tr><td>Serialized (<code>_SER</code>)</td><td>Byte arrays instead of raw objects: smaller footprint, slower reads. Reach for it when memory is tight<sup class="footnote-ref"><a href="#fn9" id="fnref9:1">[9:1]</a></sup>.</td></tr><tr><td>Replicated (<code>_2</code>)</td><td>A second copy for fast fault recovery instead of waiting on recomputation, at twice the space<sup class="footnote-ref"><a href="#fn10" id="fnref10">[10]</a></sup><sup class="footnote-ref"><a href="#fn9" id="fnref9:2">[9:2]</a></sup>.</td></tr></tbody></table><p>Whether to spill to disk hinges on recomputation cost: reading a block back from disk only beats recomputing it when the work that produced the data is expensive or filters out a large fraction, so the RDD guide says don&#39;t enable disk otherwise<sup class="footnote-ref"><a href="#fn10" id="fnref10:1">[10:1]</a></sup>.</p><p>Caching is not free, and several cases don&#39;t warrant it:</p><ul><li><strong>Single use.</strong> Caching adds serialization, deserialization, and storage cost, so caching data you read once only slows you down<sup class="footnote-ref"><a href="#fn8" id="fnref8:2">[8:2]</a></sup>.</li><li><strong>Data larger than storage memory,</strong> or a cheap transformation that isn&#39;t reused often regardless of size<sup class="footnote-ref"><a href="#fn11" id="fnref11">[11]</a></sup>.</li><li><strong>Memory pressure.</strong> Cache memory is memory taken from processing, and under the default spill strategy cached data can land on slower disk, so caching can cost more than just re-reading the source<sup class="footnote-ref"><a href="#fn9" id="fnref9:3">[9:3]</a></sup>.</li><li><strong>Lost optimizer freedom.</strong> Once a dataset is cached, Catalyst works on the in-memory copy and can no longer push filters down to the source<sup class="footnote-ref"><a href="#fn9" id="fnref9:4">[9:4]</a></sup>.</li></ul><p>A persisted RDD you&#39;ve stopped using still occupies memory until the app ends or eviction forces it out, so call <code>unpersist()</code> to reclaim it deliberately, which matters most on shared clusters<sup class="footnote-ref"><a href="#fn9" id="fnref9:5">[9:5]</a></sup>.</p><h2 id="related" tabindex="-1">Related <a class="header-anchor" href="#related" aria-label="Permalink to &quot;Related&quot;">​</a></h2><ul><li><strong>Task parallelism:</strong> <a href="./partitioning.html">Partitioning</a></li><li><strong>Executor sizing:</strong> <a href="./cluster-config.html">Cluster Tuning</a></li></ul><hr class="footnotes-sep"><section class="footnotes"><ol class="footnotes-list"><li id="fn1" class="footnote-item"><p><em>Learning Spark, 2nd Edition</em>, Damji, Wenig, Das &amp; Lee, ch. 7 <a href="#fnref1" class="footnote-backref">↩︎</a> <a href="#fnref1:1" class="footnote-backref">↩︎</a></p></li><li id="fn2" class="footnote-item"><p><em>High Performance Spark, 2nd Edition</em>, Karau, Polak &amp; Warren, ch. 8 <a href="#fnref2" class="footnote-backref">↩︎</a> <a href="#fnref2:1" class="footnote-backref">↩︎</a></p></li><li id="fn3" class="footnote-item"><p><a href="https://spark.apache.org/docs/latest/configuration.html" target="_blank" rel="noreferrer">Configuration — Spark</a> <a href="#fnref3" class="footnote-backref">↩︎</a> <a href="#fnref3:1" class="footnote-backref">↩︎</a> <a href="#fnref3:2" class="footnote-backref">↩︎</a> <a href="#fnref3:3" class="footnote-backref">↩︎</a></p></li><li id="fn4" class="footnote-item"><p><em>Spark: The Definitive Guide</em>, Chambers &amp; Zaharia, ch. 19 <a href="#fnref4" class="footnote-backref">↩︎</a></p></li><li id="fn5" class="footnote-item"><p><a href="https://spark.apache.org/docs/latest/job-scheduling.html" target="_blank" rel="noreferrer">Job Scheduling — Dynamic Resource Allocation</a> <a href="#fnref5" class="footnote-backref">↩︎</a></p></li><li id="fn6" class="footnote-item"><p><a href="https://luminousmen.com/post/explaining-the-mechanics-of-spark-caching" target="_blank" rel="noreferrer">Explaining the Mechanics of Spark Caching</a> <a href="#fnref6" class="footnote-backref">↩︎</a> <a href="#fnref6:1" class="footnote-backref">↩︎</a> <a href="#fnref6:2" class="footnote-backref">↩︎</a> <a href="#fnref6:3" class="footnote-backref">↩︎</a> <a href="#fnref6:4" class="footnote-backref">↩︎</a></p></li><li id="fn7" class="footnote-item"><p><em>High Performance Spark, 2nd Edition</em>, Karau, Polak &amp; Warren, ch. 2 <a href="#fnref7" class="footnote-backref">↩︎</a></p></li><li id="fn8" class="footnote-item"><p><em>Spark: The Definitive Guide</em>, Chambers &amp; Zaharia, ch. 19 <a href="#fnref8" class="footnote-backref">↩︎</a> <a href="#fnref8:1" class="footnote-backref">↩︎</a> <a href="#fnref8:2" class="footnote-backref">↩︎</a></p></li><li id="fn9" class="footnote-item"><p><a href="https://luminousmen.com/post/spark-tips-caching" target="_blank" rel="noreferrer">Spark Tips: Caching</a> <a href="#fnref9" class="footnote-backref">↩︎</a> <a href="#fnref9:1" class="footnote-backref">↩︎</a> <a href="#fnref9:2" class="footnote-backref">↩︎</a> <a href="#fnref9:3" class="footnote-backref">↩︎</a> <a href="#fnref9:4" class="footnote-backref">↩︎</a> <a href="#fnref9:5" class="footnote-backref">↩︎</a></p></li><li id="fn10" class="footnote-item"><p><a href="https://spark.apache.org/docs/latest/rdd-programming-guide.html" target="_blank" rel="noreferrer">RDD Programming Guide — Spark</a> <a href="#fnref10" class="footnote-backref">↩︎</a> <a href="#fnref10:1" class="footnote-backref">↩︎</a></p></li><li id="fn11" class="footnote-item"><p><em>Learning Spark, 2nd Edition</em>, Damji, Wenig, Das &amp; Lee, ch. 7 <a href="#fnref11" class="footnote-backref">↩︎</a></p></li></ol></section>',31)])])}const g=o(n,[["render",l]]);export{k as __pageData,g as default};
@@ -0,0 +1 @@
1
+ import{_ as a,a as t}from"./chunks/duplicate-plan-subtree.dark.Cdp70QhV.js";import{_ as o,o as s,c as r,a5 as i}from"./chunks/framework.DSg0KOwT.js";const k=JSON.parse('{"title":"Executor Utilization","description":"","frontmatter":{"title":"Executor Utilization"},"headers":[],"relativePath":"tuning-reference/bottleneck-utilization.md","filePath":"tuning-reference/bottleneck-utilization.md"}'),n={name:"tuning-reference/bottleneck-utilization.md"};function l(c,e,f,h,d,p){return s(),r("div",null,[...e[0]||(e[0]=[i("",31)])])}const g=o(n,[["render",l]]);export{k as __pageData,g as default};
@@ -0,0 +1 @@
1
+ import{_ as a,o,c as t,a5 as r}from"./chunks/framework.DSg0KOwT.js";const s="../assets/cache-lifecycle.rEOVYQNU.svg",n="../assets/cache-lifecycle.dark.B-hS7AgU.svg",m=JSON.parse('{"title":"Caching & Persistence","description":"","frontmatter":{"title":"Caching & Persistence"},"headers":[],"relativePath":"tuning-reference/caching.md","filePath":"tuning-reference/caching.md"}'),f={name:"tuning-reference/caching.md"};function c(i,e,l,h,d,p){return o(),t("div",null,[...e[0]||(e[0]=[r('<h1 id="caching" tabindex="-1">Caching &amp; Persistence <a class="header-anchor" href="#caching" aria-label="Permalink to &quot;Caching &amp; Persistence {#caching}&quot;">​</a></h1><h2 id="how-persisting-and-checkpointing-differ" tabindex="-1">How persisting and checkpointing differ <a class="header-anchor" href="#how-persisting-and-checkpointing-differ" aria-label="Permalink to &quot;How persisting and checkpointing differ&quot;">​</a></h2><p>Spark&#39;s <code>persist()</code> is the general mechanism for keeping a DataFrame&#39;s already-computed partitions around instead of recomputing them from source on every action; <code>cache()</code> is shorthand for <code>persist(StorageLevel.MEMORY_AND_DISK)</code>: it calls persist with that one fixed level internally<sup class="footnote-ref"><a href="#fn1" id="fnref1">[1]</a></sup>, whereas persist() lets you choose any storage level (in-memory vs on-disk, serialized vs deserialized, replicated or not)<sup class="footnote-ref"><a href="#fn2" id="fnref2">[2]</a></sup>. When the level passed to persist() is exactly <code>MEMORY_AND_DISK</code>, the two calls are identical<sup class="footnote-ref"><a href="#fn1" id="fnref1:1">[1:1]</a></sup>.</p><p>Each storage level is defined by five attributes: <code>useDisk</code>, <code>useMemory</code>, <code>useOffHeap</code>, <code>deserialized</code>, and <code>replication</code><sup class="footnote-ref"><a href="#fn3" id="fnref3">[3]</a></sup>. Under <code>MEMORY_AND_DISK</code>, Spark stores data directly as objects in memory and serializes only the portion that doesn&#39;t fit, writing that overflow to disk<sup class="footnote-ref"><a href="#fn2" id="fnref2:1">[2:1]</a></sup>. <code>MEMORY_AND_DISK_SER</code> behaves the same way except the data kept in memory is also serialized (data written to disk is always serialized under either level)<sup class="footnote-ref"><a href="#fn2" id="fnref2:2">[2:2]</a></sup>. Serialized byte streams use less memory than deserialized JVM objects, which carry structural overhead, but reading them back costs more CPU than reading deserialized objects directly<sup class="footnote-ref"><a href="#fn3" id="fnref3:1">[3:1]</a></sup><sup class="footnote-ref"><a href="#fn4" id="fnref4">[4]</a></sup>.</p><p>Checkpointing is a related but distinct mechanism. Instead of persisting, it writes an RDD&#39;s partitions to an external, reliable store (HDFS, S3) and drops the lineage, the dependency chain Spark would otherwise use to recompute it<sup class="footnote-ref"><a href="#fn3" id="fnref3:2">[3:2]</a></sup><sup class="footnote-ref"><a href="#fn5" id="fnref5">[5]</a></sup>. That truncation doesn&#39;t happen the instant <code>.checkpoint()</code> is called; the call only marks the RDD, and the actual truncation runs later, in <code>doCheckpoint()</code>, after a job using the RDD completes. At that point the RDD is already materialized, and its dependencies and old parents are cleared<sup class="footnote-ref"><a href="#fn6" id="fnref6">[6]</a></sup>. A related call, <code>localCheckpoint()</code>, also truncates lineage using Spark&#39;s caching layer, but trades fault tolerance for speed: its data lives in ephemeral local executor storage rather than a reliable filesystem, so losing an executor mid-computation can make that data permanently unrecoverable<sup class="footnote-ref"><a href="#fn6" id="fnref6:1">[6:1]</a></sup>.</p><h2 id="when-caching-pays-off" tabindex="-1">When caching pays off <a class="header-anchor" href="#when-caching-pays-off" aria-label="Permalink to &quot;When caching pays off&quot;">​</a></h2><p>Because of how that materialization works, caching earns its keep only in specific situations. It&#39;s worth reaching for when the same DataFrame is scanned or recomputed more than once downstream: repeated queries against the same base data, iterative ML training loops, or interactive exploration where a dataset feeds multiple branches<sup class="footnote-ref"><a href="#fn2" id="fnref2:3">[2:3]</a></sup><sup class="footnote-ref"><a href="#fn7" id="fnref7">[7]</a></sup>. In one book benchmark, caching a 10M-row DataFrame and materializing it with <code>count()</code> cut a subsequent <code>count()</code> from 5.11s to 0.44s, roughly a 12x speedup<sup class="footnote-ref"><a href="#fn2" id="fnref2:4">[2:4]</a></sup>.</p><p>A few signals point the other way, toward caching having no effect or actively hurting:</p><ul><li><strong>Single-use data.</strong> If a dataset is processed only once downstream, caching just adds serialization and storage-bookkeeping cost for no reuse benefit<sup class="footnote-ref"><a href="#fn7" id="fnref7:1">[7:1]</a></sup>.</li><li><strong>Data too big to fit.</strong> Caching is all-or-nothing per partition: a DataFrame can be &quot;fractionally cached&quot; across its partitions, but individual partitions can&#39;t be split. With room for only 4.5 of 8 partitions, exactly 4 get cached; the uncached remainder is recomputed on every access, which can end up slower than not caching at all<sup class="footnote-ref"><a href="#fn2" id="fnref2:5">[2:5]</a></sup>.</li><li><strong>Partial materialization.</strong> <code>cache()</code>/<code>persist()</code> are lazy: nothing is cached until an action forces a full pass. <code>count()</code> forces a genuine full pass and materializes every partition, but an action like <code>take(1)</code> computes and caches only the one partition Catalyst needs, so a later full scan still recomputes most of the data. The same applies to <code>take(10)</code>, <code>limit(100)</code>, or any other partial or filtered scan: only the blocks Spark was forced to touch get cached, and the rest stays lazily uncomputed<sup class="footnote-ref"><a href="#fn2" id="fnref2:6">[2:6]</a></sup><sup class="footnote-ref"><a href="#fn1" id="fnref1:2">[1:2]</a></sup>.</li><li><strong>Plan mismatch.</strong> Caching wraps the <em>analyzed</em> (pre-optimization) logical plan in an <code>InMemoryRelation</code>. A logically-equivalent query written differently produces a different analyzed plan, so Spark can silently miss the cache and recompute from scratch even though the optimized plan would have been identical<sup class="footnote-ref"><a href="#fn1" id="fnref1:3">[1:3]</a></sup>.</li><li><strong>Optimizer limitations.</strong> Caching freezes Catalyst&#39;s optimization opportunities at the cached point. It can, for example, block <a href="./data-formats.html">predicate pushdown</a> once data is served from the in-memory cache instead of the source<sup class="footnote-ref"><a href="#fn5" id="fnref5:1">[5:1]</a></sup>.</li></ul><h2 id="why-cached-blocks-don-t-stick-around" tabindex="-1">Why cached blocks don&#39;t stick around <a class="header-anchor" href="#why-cached-blocks-don-t-stick-around" aria-label="Permalink to &quot;Why cached blocks don&#39;t stick around&quot;">​</a></h2><p>Materializing a cache is only half the story, though. A <code>cache()</code> followed by <code>count()</code> guarantees a materialization pass over every partition at that moment: caching happens locally and incrementally, with each executor&#39;s BlockManager storing only the blocks it computes, as it computes them, and no centralized &quot;cache the whole dataset&quot; step<sup class="footnote-ref"><a href="#fn1" id="fnref1:4">[1:4]</a></sup>. It does not guarantee those partitions stay cached afterward. Two things can undo it:</p><ul><li><strong>Executor loss.</strong> If a block was cached on an executor that later goes away, the cached data goes with it, unless a replicated storage level such as <code>MEMORY_AND_DISK_2</code> was used<sup class="footnote-ref"><a href="#fn1" id="fnref1:5">[1:5]</a></sup>. This matters specifically for <a href="./cluster-config.html">dynamic allocation</a>: Spark&#39;s documentation is explicit that caching together with dynamic allocation &quot;is NOT safe,&quot; because reclaiming idle executors takes their cached blocks with them<sup class="footnote-ref"><a href="#fn6" id="fnref6:2">[6:2]</a></sup>.</li><li><strong>Memory pressure eviction.</strong> Cached blocks can be evicted by other operations&#39; memory demands after materialization, independent of any executor loss<sup class="footnote-ref"><a href="#fn1" id="fnref1:6">[1:6]</a></sup>.</li></ul><p>Eviction itself is governed by LRU: Spark removes the least-recently-used cached blocks when storage memory is under pressure<sup class="footnote-ref"><a href="#fn3" id="fnref3:3">[3:3]</a></sup><sup class="footnote-ref"><a href="#fn1" id="fnref1:7">[1:7]</a></sup><sup class="footnote-ref"><a href="#fn8" id="fnref8">[8]</a></sup>. The trigger is memory contention between Spark&#39;s <a href="./memory-model.html">unified Execution and Storage regions</a>: under the <code>UnifiedMemoryManager</code>, Execution has priority, so if a task needs execution memory for a shuffle, join, aggregation, or sort and Storage is occupying that space, Spark evicts cached blocks to free it, and this can happen at any time, not only the next time the cache is accessed<sup class="footnote-ref"><a href="#fn9" id="fnref9">[9]</a></sup><sup class="footnote-ref"><a href="#fn1" id="fnref1:8">[1:8]</a></sup>. What happens to an evicted block depends on its storage level, not on a separate spill decision: a disk-backed level (<code>MEMORY_AND_DISK</code>, <code>MEMORY_AND_DISK_SER</code>) spills the block to disk and reads it back on next use, at the cost of disk I/O; a memory-only level (<code>MEMORY_ONLY</code>) simply drops the block and recomputes it from source when needed again<sup class="footnote-ref"><a href="#fn2" id="fnref2:7">[2:7]</a></sup><sup class="footnote-ref"><a href="#fn10" id="fnref10">[10]</a></sup><sup class="footnote-ref"><a href="#fn3" id="fnref3:4">[3:4]</a></sup><sup class="footnote-ref"><a href="#fn1" id="fnref1:9">[1:9]</a></sup>.</p><img class="light-only" src="'+s+'" alt="A cached block moves to evicted under memory pressure or lost when an executor goes away, then either spills to disk under MEMORY_AND_DISK or recomputes from lineage under MEMORY_ONLY."><img class="dark-only" src="'+n+'" alt="A cached block moves to evicted under memory pressure or lost when an executor goes away, then either spills to disk under MEMORY_AND_DISK or recomputes from lineage under MEMORY_ONLY."><p>There&#39;s no hard limit on how many DataFrames can be cached simultaneously; the constraint is aggregate storage memory, not a count of objects<sup class="footnote-ref"><a href="#fn2" id="fnref2:8">[2:8]</a></sup>. That budget is fraction-based configuration translated into a live byte allowance, not a fixed absolute constant: <code>spark.memory.fraction</code> (default 0.6) sets the fraction of (heap minus a 300MB reserved region) used for the combined Execution+Storage region, and <code>spark.memory.storageFraction</code> (default 0.5) sets the portion of that region immune to eviction. Lowering either makes <a href="./bottleneck-spill.html">spills</a> and evictions more frequent<sup class="footnote-ref"><a href="#fn4" id="fnref4:1">[4:1]</a></sup><sup class="footnote-ref"><a href="#fn9" id="fnref9:1">[9:1]</a></sup><sup class="footnote-ref"><a href="#fn1" id="fnref1:10">[1:10]</a></sup>.</p><h2 id="habits-that-keep-caching-effective" tabindex="-1">Habits that keep caching effective <a class="header-anchor" href="#habits-that-keep-caching-effective" aria-label="Permalink to &quot;Habits that keep caching effective&quot;">​</a></h2><p>A few habits keep caching effective despite how easily any of that goes wrong:</p><ul><li><strong>Materialize deliberately.</strong> Because caching is lazy, follow <code>cache()</code>/<code>persist()</code> with an action that forces a full pass (<code>count()</code> is the standard choice) rather than assuming the call alone did the work<sup class="footnote-ref"><a href="#fn2" id="fnref2:9">[2:9]</a></sup><sup class="footnote-ref"><a href="#fn7" id="fnref7:2">[7:2]</a></sup>.</li><li><strong>Match the storage level to the constraint you&#39;re solving.</strong> <code>cache()</code> / <code>MEMORY_AND_DISK</code> is a reasonable default: Spark keeps deserialized objects in memory and only serializes the overflow to disk. Reach for <code>MEMORY_AND_DISK_SER</code> when memory pressure is the binding constraint and you can afford the extra CPU cost of deserializing on read<sup class="footnote-ref"><a href="#fn2" id="fnref2:10">[2:10]</a></sup><sup class="footnote-ref"><a href="#fn4" id="fnref4:2">[4:2]</a></sup><sup class="footnote-ref"><a href="#fn3" id="fnref3:5">[3:5]</a></sup>.</li><li><strong>Protect cached data from dynamic allocation.</strong> If executors holding cached blocks might be reclaimed as idle, raise <code>spark.dynamicAllocation.cachedExecutorIdleTimeout</code> so they aren&#39;t pulled out from under the cache<sup class="footnote-ref"><a href="#fn6" id="fnref6:3">[6:3]</a></sup>, or use a replicated storage level (e.g., <code>MEMORY_AND_DISK_2</code>) so losing one executor doesn&#39;t lose the data<sup class="footnote-ref"><a href="#fn1" id="fnref1:11">[1:11]</a></sup>.</li><li><strong>Reach for checkpointing, not persisting, for genuine lineage truncation with fault tolerance</strong> across a long transformation chain, since it writes to a reliable external filesystem rather than relying on executor-local memory or disk<sup class="footnote-ref"><a href="#fn3" id="fnref3:6">[3:6]</a></sup><sup class="footnote-ref"><a href="#fn5" id="fnref5:2">[5:2]</a></sup>. Avoid <code>localCheckpoint()</code> under dynamic allocation for the same reason caching is unsafe there: its data lives in ephemeral local executor storage that can vanish along with a reclaimed executor<sup class="footnote-ref"><a href="#fn6" id="fnref6:4">[6:4]</a></sup>. Also set a checkpoint directory via <code>SparkContext.setCheckpointDir</code> before calling <code>.checkpoint()</code>; without one, the call throws immediately rather than silently doing nothing<sup class="footnote-ref"><a href="#fn6" id="fnref6:5">[6:5]</a></sup>.</li><li><strong>Don&#39;t cache data that won&#39;t benefit:</strong> single-use datasets, datasets that don&#39;t fit in available memory, or datasets you only ever touch through partial actions (<code>take</code>, <code>limit</code>). In each of these cases the caching overhead isn&#39;t paid back<sup class="footnote-ref"><a href="#fn7" id="fnref7:3">[7:3]</a></sup><sup class="footnote-ref"><a href="#fn2" id="fnref2:11">[2:11]</a></sup>.</li></ul><h2 id="sources" tabindex="-1">Sources <a class="header-anchor" href="#sources" aria-label="Permalink to &quot;Sources&quot;">​</a></h2><hr class="footnotes-sep"><section class="footnotes"><ol class="footnotes-list"><li id="fn1" class="footnote-item"><p><a href="https://luminousmen.com/post/explaining-the-mechanics-of-spark-caching" target="_blank" rel="noreferrer">Explaining the Mechanics of Spark Caching</a> <a href="#fnref1" class="footnote-backref">↩︎</a> <a href="#fnref1:1" class="footnote-backref">↩︎</a> <a href="#fnref1:2" class="footnote-backref">↩︎</a> <a href="#fnref1:3" class="footnote-backref">↩︎</a> <a href="#fnref1:4" class="footnote-backref">↩︎</a> <a href="#fnref1:5" class="footnote-backref">↩︎</a> <a href="#fnref1:6" class="footnote-backref">↩︎</a> <a href="#fnref1:7" class="footnote-backref">↩︎</a> <a href="#fnref1:8" class="footnote-backref">↩︎</a> <a href="#fnref1:9" class="footnote-backref">↩︎</a> <a href="#fnref1:10" class="footnote-backref">↩︎</a> <a href="#fnref1:11" class="footnote-backref">↩︎</a></p></li><li id="fn2" class="footnote-item"><p><em>Learning Spark, 2nd Edition</em>, Damji, Wenig, Das &amp; Lee, ch. 7 — Optimizing and Tuning Spark Applications <a href="#fnref2" class="footnote-backref">↩︎</a> <a href="#fnref2:1" class="footnote-backref">↩︎</a> <a href="#fnref2:2" class="footnote-backref">↩︎</a> <a href="#fnref2:3" class="footnote-backref">↩︎</a> <a href="#fnref2:4" class="footnote-backref">↩︎</a> <a href="#fnref2:5" class="footnote-backref">↩︎</a> <a href="#fnref2:6" class="footnote-backref">↩︎</a> <a href="#fnref2:7" class="footnote-backref">↩︎</a> <a href="#fnref2:8" class="footnote-backref">↩︎</a> <a href="#fnref2:9" class="footnote-backref">↩︎</a> <a href="#fnref2:10" class="footnote-backref">↩︎</a> <a href="#fnref2:11" class="footnote-backref">↩︎</a></p></li><li id="fn3" class="footnote-item"><p><em>High Performance Spark, 2nd Edition</em>, Karau, Polak &amp; Warren, ch. 7 — Effective Transformations <a href="#fnref3" class="footnote-backref">↩︎</a> <a href="#fnref3:1" class="footnote-backref">↩︎</a> <a href="#fnref3:2" class="footnote-backref">↩︎</a> <a href="#fnref3:3" class="footnote-backref">↩︎</a> <a href="#fnref3:4" class="footnote-backref">↩︎</a> <a href="#fnref3:5" class="footnote-backref">↩︎</a> <a href="#fnref3:6" class="footnote-backref">↩︎</a></p></li><li id="fn4" class="footnote-item"><p><a href="https://spark.apache.org/docs/latest/configuration.html" target="_blank" rel="noreferrer">Configuration — Spark</a> <a href="#fnref4" class="footnote-backref">↩︎</a> <a href="#fnref4:1" class="footnote-backref">↩︎</a> <a href="#fnref4:2" class="footnote-backref">↩︎</a></p></li><li id="fn5" class="footnote-item"><p><a href="https://luminousmen.com/post/spark-tips-caching" target="_blank" rel="noreferrer">Spark Tips: Caching</a> <a href="#fnref5" class="footnote-backref">↩︎</a> <a href="#fnref5:1" class="footnote-backref">↩︎</a> <a href="#fnref5:2" class="footnote-backref">↩︎</a></p></li><li id="fn6" class="footnote-item"><p><a href="https://raw.githubusercontent.com/apache/spark/v3.5.0/core/src/main/scala/org/apache/spark/rdd/RDD.scala" target="_blank" rel="noreferrer">RDD.scala</a> <a href="#fnref6" class="footnote-backref">↩︎</a> <a href="#fnref6:1" class="footnote-backref">↩︎</a> <a href="#fnref6:2" class="footnote-backref">↩︎</a> <a href="#fnref6:3" class="footnote-backref">↩︎</a> <a href="#fnref6:4" class="footnote-backref">↩︎</a> <a href="#fnref6:5" class="footnote-backref">↩︎</a></p></li><li id="fn7" class="footnote-item"><p><em>Spark: The Definitive Guide</em>, Chambers &amp; Zaharia, ch. 19 — Performance Tuning <a href="#fnref7" class="footnote-backref">↩︎</a> <a href="#fnref7:1" class="footnote-backref">↩︎</a> <a href="#fnref7:2" class="footnote-backref">↩︎</a> <a href="#fnref7:3" class="footnote-backref">↩︎</a></p></li><li id="fn8" class="footnote-item"><p><a href="https://raw.githubusercontent.com/spoddutur/spark-notes/master/task_memory_management_in_spark.md" target="_blank" rel="noreferrer">Task Memory Management in Spark</a> <a href="#fnref8" class="footnote-backref">↩︎</a></p></li><li id="fn9" class="footnote-item"><p><a href="https://luminousmen.com/post/dive-into-spark-memory" target="_blank" rel="noreferrer">Dive into Spark Memory</a> <a href="#fnref9" class="footnote-backref">↩︎</a> <a href="#fnref9:1" class="footnote-backref">↩︎</a></p></li><li id="fn10" class="footnote-item"><p><a href="https://spark.apache.org/docs/latest/rdd-programming-guide.html" target="_blank" rel="noreferrer">RDD Programming Guide</a> <a href="#fnref10" class="footnote-backref">↩︎</a></p></li></ol></section>',22)])])}const g=a(f,[["render",c]]);export{m as __pageData,g as default};
@@ -0,0 +1 @@
1
+ import{_ as a,o,c as t,a5 as r}from"./chunks/framework.DSg0KOwT.js";const s="../assets/cache-lifecycle.rEOVYQNU.svg",n="../assets/cache-lifecycle.dark.B-hS7AgU.svg",m=JSON.parse('{"title":"Caching & Persistence","description":"","frontmatter":{"title":"Caching & Persistence"},"headers":[],"relativePath":"tuning-reference/caching.md","filePath":"tuning-reference/caching.md"}'),f={name:"tuning-reference/caching.md"};function c(i,e,l,h,d,p){return o(),t("div",null,[...e[0]||(e[0]=[r("",22)])])}const g=a(f,[["render",c]]);export{m as __pageData,g as default};
@@ -0,0 +1 @@
1
+ import{_ as o,o as a,c as t,a5 as r}from"./chunks/framework.DSg0KOwT.js";const s="../assets/container-memory.DIO0AnIm.svg",n="../assets/container-memory.dark.CP-5zuCl.svg",m=JSON.parse('{"title":"Cluster Tuning","description":"","frontmatter":{"title":"Cluster Tuning"},"headers":[],"relativePath":"tuning-reference/cluster-config.md","filePath":"tuning-reference/cluster-config.md"}'),c={name:"tuning-reference/cluster-config.md"};function f(i,e,l,u,d,h){return a(),t("div",null,[...e[0]||(e[0]=[r('<h1 id="cluster-config" tabindex="-1">Cluster Tuning <a class="header-anchor" href="#cluster-config" aria-label="Permalink to &quot;Cluster Tuning {#cluster-config}&quot;">​</a></h1><h2 id="sizing-executors-and-containers" tabindex="-1">Sizing executors and containers <a class="header-anchor" href="#sizing-executors-and-containers" aria-label="Permalink to &quot;Sizing executors and containers&quot;">​</a></h2><p>Cluster tuning is the set of decisions that map an application&#39;s cores, memory, and executor count onto the underlying cluster (YARN or Kubernetes) plus a handful of related knobs (data locality, dynamic allocation, per-task CPU reservation) that all interact with that sizing decision.</p><p>The starting point is executor sizing. Working through a concrete example (a 10-node cluster with 16 cores and 64GB RAM per node), the derivation runs: assign 5 cores per executor; reserve about 1 core per node for Hadoop/YARN/OS daemons, leaving 15 usable cores per node (150 total); dividing by 5 cores per executor gives 30 executors, minus 1 reserved for the YARN ApplicationMaster (29); with 3 executors per node, memory per executor is 64GB / 3 ≈ 21GB, and after subtracting about 7% for YARN memory overhead that comes out to roughly 18GB usable, a recommended shape of 29 executors × 5 cores × 18GB<sup class="footnote-ref"><a href="#fn1" id="fnref1">[1]</a></sup>. Cloudera&#39;s own worked example, on a 6-node cluster with the same 16-core / 64GB-per-node hardware, lands on the same shape: <code>--num-executors 17 --executor-cores 5 --executor-memory 19G</code>, rather than one &quot;fat&quot; executor per node (<code>--num-executors 6 --executor-cores 15 --executor-memory 63G</code>)<sup class="footnote-ref"><a href="#fn2" id="fnref2">[2]</a></sup>. Both sources treat the resulting task count (executor-cores × num-executors) as the single most important tuning lever, since Spark cannot compensate for too little parallelism on its own<sup class="footnote-ref"><a href="#fn2" id="fnref2:1">[2:1]</a></sup>. <code>spark.executor.cores</code> is the config that controls concurrent tasks per executor; it defaults to 1 in YARN mode, or to all available cores on the worker in standalone mode<sup class="footnote-ref"><a href="#fn3" id="fnref3">[3]</a></sup>, and <code>spark.executor.memory</code> sets the JVM heap size per executor<sup class="footnote-ref"><a href="#fn2" id="fnref2:2">[2:2]</a></sup>. Independent of the sizing formula above, the Spark tuning guide recommends targeting 2–3 tasks per CPU core across the cluster<sup class="footnote-ref"><a href="#fn4" id="fnref4">[4]</a></sup>.</p><p>On top of executor memory sits container memory. Spark computes the total container/pod allocation as <code>executorMemoryMiB + memoryOverheadMiB + memoryOffHeapMiB + pysparkMemToUseMiB</code>. On YARN this is what gets allocated for the container, on Kubernetes it becomes the pod memory limit<sup class="footnote-ref"><a href="#fn5" id="fnref5">[5]</a></sup>. <code>spark.executor.memoryOverhead</code> itself defaults to <code>max(0.1 * executorMemory, 384MB)</code><sup class="footnote-ref"><a href="#fn5" id="fnref5:1">[5:1]</a></sup><sup class="footnote-ref"><a href="#fn3" id="fnref3:1">[3:1]</a></sup>.</p><img class="light-only" src="'+s+'" alt="Container memory sums executor heap, memory overhead, off-heap size, and pyspark memory into the requested container size; the resource manager grants a container of that size and OOMKills the executor when its actual runtime footprint spills past that limit."><img class="dark-only" src="'+n+'" alt="Container memory sums executor heap, memory overhead, off-heap size, and pyspark memory into the requested container size; the resource manager grants a container of that size and OOMKills the executor when its actual runtime footprint spills past that limit."><blockquote><p><strong>PySpark:</strong> <code>spark.executor.pyspark.memory</code> is only added as its own container-memory term when it&#39;s set explicitly; otherwise PySpark&#39;s memory use is folded into the general overhead budget rather than tracked separately.<sup class="footnote-ref"><a href="#fn5" id="fnref5:2">[5:2]</a></sup></p></blockquote><p>On Kubernetes specifically, <code>spark.kubernetes.executor.request.cores</code> and <code>spark.kubernetes.executor.limit.cores</code> take priority over <code>spark.executor.cores</code> for the pod&#39;s CPU request/limit sent to the Kubernetes scheduler<sup class="footnote-ref"><a href="#fn6" id="fnref6">[6]</a></sup>, while <code>spark.executor.cores</code> remains the config Spark itself uses to size the number of concurrent task slots<sup class="footnote-ref"><a href="#fn3" id="fnref3:2">[3:2]</a></sup>.</p><p>Two more knobs round out the picture. <code>spark.locality.wait</code> (default <code>3s</code>) controls how long a task waits for a data-local placement before Spark gives up and schedules it less locally, stepping through the same wait across process-local → node-local → rack-local → any; each level can be overridden independently via <code>spark.locality.wait.process</code>, <code>.node</code>, and <code>.rack</code>, all of which default to the base wait when left unset<sup class="footnote-ref"><a href="#fn3" id="fnref3:3">[3:3]</a></sup>. Dynamic allocation lets Spark add and remove executors as work changes, but it requires one of several supporting mechanisms: an external shuffle service, shuffle tracking (<code>spark.dynamicAllocation.shuffleTracking.enabled</code>, default <code>true</code> since Spark 3.0), shuffle-block decommission, or the experimental sort-IO plugin<sup class="footnote-ref"><a href="#fn3" id="fnref3:4">[3:4]</a></sup>. Removing an executor can otherwise destroy shuffle state it&#39;s holding. Finally, <code>spark.task.cpus</code> (default <code>1</code>) sets how many cores each task reserves; the number of concurrent task slots per executor is derived jointly from it and <code>spark.executor.cores</code>, roughly <code>spark.executor.cores / spark.task.cpus</code><sup class="footnote-ref"><a href="#fn3" id="fnref3:5">[3:5]</a></sup>.</p><h2 id="spotting-a-bad-sizing-decision" tabindex="-1">Spotting a bad sizing decision <a class="header-anchor" href="#spotting-a-bad-sizing-decision" aria-label="Permalink to &quot;Spotting a bad sizing decision&quot;">​</a></h2><p>Getting that sizing wrong shows up in a handful of specific symptoms. The clearest sign of an under-sized executor layout is HDFS throughput dropping under load: &quot;HDFS client has trouble with tons of concurrent threads. It was observed that HDFS achieves full write throughput with ~5 tasks per executor&quot;<sup class="footnote-ref"><a href="#fn1" id="fnref1:1">[1:1]</a></sup>. The &quot;fat executor&quot; case (one executor per node, using all 16 cores) shows the failure mode directly: &quot;with all 16 cores per executor... HDFS throughput will hurt and it&#39;ll result in excessive garbage [collection]&quot;<sup class="footnote-ref"><a href="#fn1" id="fnref1:2">[1:2]</a></sup>.</p><p>On the memory side, an executor or pod that dies with no Spark-level error is a strong signal that off-heap memory was never added into the container budget. Enabling <code>spark.memory.offHeap.enabled=true</code> with <code>spark.memory.offHeap.size=1g</code> on top of an 8G executor with default overhead (819MB) means real usage is 8192+819+1024 = 10,035MB, while the container was only granted 8192+819 = 9,011MB. The executor gets killed by YARN, or OOMKilled by the Kubernetes kubelet, with no warning from Spark itself<sup class="footnote-ref"><a href="#fn5" id="fnref5:3">[5:3]</a></sup>.</p><p>With dynamic allocation on, unexplained shuffle recomputation is a detectable symptom of executors being reclaimed mid-shuffle: &quot;In the event of stragglers... dynamic allocation may remove an executor before the shuffle completes, in which case the shuffle files written by that executor must be recomputed unnecessarily&quot;<sup class="footnote-ref"><a href="#fn6" id="fnref6:1">[6:1]</a></sup>. Jobs that abort with a serialized-result-size error are hitting the <code>spark.driver.maxResultSize</code> guardrail rather than a driver heap exhaustion: &quot;Jobs will be aborted if the total size [of serialized action results] is above this limit&quot;<sup class="footnote-ref"><a href="#fn3" id="fnref3:6">[3:6]</a></sup>. And on the resource-waste side, executors that are provisioned but never assigned work are a sign that dynamic allocation is targeting full parallelism against a workload made of many small tasks<sup class="footnote-ref"><a href="#fn3" id="fnref3:7">[3:7]</a></sup>.</p><h2 id="what-a-bad-sizing-decision-costs" tabindex="-1">What a bad sizing decision costs <a class="header-anchor" href="#what-a-bad-sizing-decision-costs" aria-label="Permalink to &quot;What a bad sizing decision costs&quot;">​</a></h2><p>Each symptom above carries a specific price tag. Task count (<code>executor-cores × num-executors</code>) is treated by both cited sources as the single most important tuning lever, since Spark cannot compensate for too little parallelism on its own<sup class="footnote-ref"><a href="#fn2" id="fnref2:3">[2:3]</a></sup>. Going the other way, cramming too many cores into one executor degrades HDFS throughput and drives up garbage collection<sup class="footnote-ref"><a href="#fn1" id="fnref1:3">[1:3]</a></sup>.</p><p>Container-memory misconfiguration matters because the failure is silent from Spark&#39;s point of view: off-heap memory that isn&#39;t accounted for in <code>spark.executor.memoryOverhead</code> causes the executor to actually use more memory than the container/pod was granted, so it gets killed externally (by YARN or the kubelet) with no Spark-level diagnostic to point at the real cause<sup class="footnote-ref"><a href="#fn5" id="fnref5:4">[5:4]</a></sup>.</p><p>Dynamic allocation&#39;s interaction with shuffle state matters because, before dynamic allocation existed, an executor exiting alongside its application meant all its state could be safely discarded; with dynamic allocation, the application keeps running after an executor is explicitly removed, so any later need for that executor&#39;s state forces a recompute<sup class="footnote-ref"><a href="#fn6" id="fnref6:2">[6:2]</a></sup>. That is exactly why Spark needs &quot;a mechanism to decommission an executor gracefully by preserving its state before removing it&quot;<sup class="footnote-ref"><a href="#fn6" id="fnref6:3">[6:3]</a></sup>. Without shuffle tracking, an external shuffle service, or shuffle-block decommission enabled, dynamic allocation either can&#39;t be turned on at all, or, if it&#39;s active regardless, an executor holding unpreserved shuffle output that gets removed forces exactly that unnecessary recompute<sup class="footnote-ref"><a href="#fn3" id="fnref3:8">[3:8]</a></sup><sup class="footnote-ref"><a href="#fn6" id="fnref6:4">[6:4]</a></sup>.</p><p>On the driver side, <code>spark.driver.maxResultSize</code> and <code>spark.driver.memory</code> are separate budgets that still interact: whether a high <code>maxResultSize</code> actually causes an out-of-memory error &quot;depends on spark.driver.memory and memory overhead of objects in JVM&quot;<sup class="footnote-ref"><a href="#fn3" id="fnref3:9">[3:9]</a></sup>. Raising one without considering the other doesn&#39;t fully protect the driver.</p><p>Finally, over-provisioning executors against a small-task workload wastes cluster resources: &quot;with small tasks this setting can waste a lot of resources due to executor allocation overhead, as some executor might not even do any work&quot;<sup class="footnote-ref"><a href="#fn3" id="fnref3:10">[3:10]</a></sup>.</p><h2 id="sizing-the-cluster-correctly" tabindex="-1">Sizing the cluster correctly <a class="header-anchor" href="#sizing-the-cluster-correctly" aria-label="Permalink to &quot;Sizing the cluster correctly&quot;">​</a></h2><p>Avoiding those costs starts from a formula, not one fat executor per node. Apply the balanced-executor formula instead; the worked examples above give shapes like 29 × 5 × 18GB for a 10-node/16-core/64GB cluster, or 17 × 5 × 19G for a 6-node cluster with the same per-node hardware<sup class="footnote-ref"><a href="#fn1" id="fnref1:4">[1:4]</a></sup><sup class="footnote-ref"><a href="#fn2" id="fnref2:4">[2:4]</a></sup>. Keep <code>--executor-cores</code> at or below roughly 5 so HDFS client concurrency stays in the range where it sustains full write throughput<sup class="footnote-ref"><a href="#fn1" id="fnref1:5">[1:5]</a></sup>.</p><p>When enabling off-heap memory, account for <code>spark.memory.offHeap.size</code> in the container/pod memory budget explicitly; <code>spark.executor.memoryOverhead</code>&#39;s default of <code>max(0.1 * executorMemory, 384MB)</code> does not include it<sup class="footnote-ref"><a href="#fn5" id="fnref5:5">[5:5]</a></sup>. If you&#39;re still referencing the older <code>spark.yarn.executor.memoryOverhead</code> name, note it was removed in Spark 3.0<sup class="footnote-ref"><a href="#fn5" id="fnref5:6">[5:6]</a></sup>.</p><p>On Kubernetes, remember that <code>spark.kubernetes.executor.request.cores</code> and <code>.limit.cores</code> govern the pod&#39;s CPU request/limit and take priority over <code>spark.executor.cores</code> for that purpose<sup class="footnote-ref"><a href="#fn6" id="fnref6:5">[6:5]</a></sup>, while <code>spark.executor.cores</code> separately governs Spark&#39;s own task-slot count<sup class="footnote-ref"><a href="#fn3" id="fnref3:11">[3:11]</a></sup>. The two need to be reasoned about together rather than assumed to be redundant.</p><p>Tune <code>spark.locality.wait</code> (and its per-level overrides) upward when tasks are long-running and locality is poor, since the default is tuned for typical workloads<sup class="footnote-ref"><a href="#fn3" id="fnref3:12">[3:12]</a></sup>; set <code>spark.locality.wait.node</code> to <code>0</code> to skip straight to rack locality when node locality isn&#39;t achievable<sup class="footnote-ref"><a href="#fn3" id="fnref3:13">[3:13]</a></sup>.</p><p>Turn on shuffle tracking (<code>spark.dynamicAllocation.shuffleTracking.enabled</code>, default <code>true</code> since Spark 3.0) or the external shuffle service so dynamic allocation can reclaim executors without losing shuffle output or forcing recomputation<sup class="footnote-ref"><a href="#fn3" id="fnref3:14">[3:14]</a></sup><sup class="footnote-ref"><a href="#fn6" id="fnref6:6">[6:6]</a></sup>. The external shuffle service additionally lets persisted RDD blocks survive executor removal when <code>spark.shuffle.service.fetch.rdd.enabled</code> is set, and executors holding cached blocks are, by default, never removed at all, tunable via <code>spark.dynamicAllocation.cachedExecutorIdleTimeout</code><sup class="footnote-ref"><a href="#fn6" id="fnref6:7">[6:7]</a></sup>.</p><p>Set <code>spark.driver.maxResultSize</code> as a guardrail on serialized action-result size, and size <code>spark.driver.memory</code> with it in mind rather than in isolation<sup class="footnote-ref"><a href="#fn3" id="fnref3:15">[3:15]</a></sup>. Use <code>spark.dynamicAllocation.executorAllocationRatio</code> (default <code>1.0</code>, added in 2.4.0) to scale down over-allocation when tasks are small; for example, a value of <code>0.5</code> halves the target executor count dynamic allocation would otherwise compute<sup class="footnote-ref"><a href="#fn3" id="fnref3:16">[3:16]</a></sup>. Raise <code>spark.task.cpus</code> above its default of <code>1</code> when a task needs more than one core; doing so proportionally reduces concurrent task slots per executor (<code>spark.executor.cores / spark.task.cpus</code>)<sup class="footnote-ref"><a href="#fn3" id="fnref3:17">[3:17]</a></sup>.</p><h2 id="sources" tabindex="-1">Sources <a class="header-anchor" href="#sources" aria-label="Permalink to &quot;Sources&quot;">​</a></h2><hr class="footnotes-sep"><section class="footnotes"><ol class="footnotes-list"><li id="fn1" class="footnote-item"><p><a href="https://raw.githubusercontent.com/spoddutur/spark-notes/master/distribution_of_executors_cores_and_memory_for_spark_application.md" target="_blank" rel="noreferrer">Distribution of Executors, Cores and Memory for a Spark Application</a> <a href="#fnref1" class="footnote-backref">↩︎</a> <a href="#fnref1:1" class="footnote-backref">↩︎</a> <a href="#fnref1:2" class="footnote-backref">↩︎</a> <a href="#fnref1:3" class="footnote-backref">↩︎</a> <a href="#fnref1:4" class="footnote-backref">↩︎</a> <a href="#fnref1:5" class="footnote-backref">↩︎</a></p></li><li id="fn2" class="footnote-item"><p><a href="https://blog.cloudera.com/how-to-tune-your-apache-spark-jobs-part-2/" target="_blank" rel="noreferrer">How to Tune Your Apache Spark Jobs (Part 2)</a> <a href="#fnref2" class="footnote-backref">↩︎</a> <a href="#fnref2:1" class="footnote-backref">↩︎</a> <a href="#fnref2:2" class="footnote-backref">↩︎</a> <a href="#fnref2:3" class="footnote-backref">↩︎</a> <a href="#fnref2:4" class="footnote-backref">↩︎</a></p></li><li id="fn3" class="footnote-item"><p><a href="https://spark.apache.org/docs/latest/configuration.html" target="_blank" rel="noreferrer">Configuration — Spark</a> <a href="#fnref3" class="footnote-backref">↩︎</a> <a href="#fnref3:1" class="footnote-backref">↩︎</a> <a href="#fnref3:2" class="footnote-backref">↩︎</a> <a href="#fnref3:3" class="footnote-backref">↩︎</a> <a href="#fnref3:4" class="footnote-backref">↩︎</a> <a href="#fnref3:5" class="footnote-backref">↩︎</a> <a href="#fnref3:6" class="footnote-backref">↩︎</a> <a href="#fnref3:7" class="footnote-backref">↩︎</a> <a href="#fnref3:8" class="footnote-backref">↩︎</a> <a href="#fnref3:9" class="footnote-backref">↩︎</a> <a href="#fnref3:10" class="footnote-backref">↩︎</a> <a href="#fnref3:11" class="footnote-backref">↩︎</a> <a href="#fnref3:12" class="footnote-backref">↩︎</a> <a href="#fnref3:13" class="footnote-backref">↩︎</a> <a href="#fnref3:14" class="footnote-backref">↩︎</a> <a href="#fnref3:15" class="footnote-backref">↩︎</a> <a href="#fnref3:16" class="footnote-backref">↩︎</a> <a href="#fnref3:17" class="footnote-backref">↩︎</a></p></li><li id="fn4" class="footnote-item"><p><a href="https://spark.apache.org/docs/latest/tuning.html" target="_blank" rel="noreferrer">Spark Tuning Guide</a> <a href="#fnref4" class="footnote-backref">↩︎</a></p></li><li id="fn5" class="footnote-item"><p><a href="https://luminousmen.com/post/dive-into-spark-memory" target="_blank" rel="noreferrer">Dive into Spark Memory</a> <a href="#fnref5" class="footnote-backref">↩︎</a> <a href="#fnref5:1" class="footnote-backref">↩︎</a> <a href="#fnref5:2" class="footnote-backref">↩︎</a> <a href="#fnref5:3" class="footnote-backref">↩︎</a> <a href="#fnref5:4" class="footnote-backref">↩︎</a> <a href="#fnref5:5" class="footnote-backref">↩︎</a> <a href="#fnref5:6" class="footnote-backref">↩︎</a></p></li><li id="fn6" class="footnote-item"><p><a href="https://spark.apache.org/docs/latest/job-scheduling.html" target="_blank" rel="noreferrer">Job Scheduling — Dynamic Resource Allocation</a> <a href="#fnref6" class="footnote-backref">↩︎</a> <a href="#fnref6:1" class="footnote-backref">↩︎</a> <a href="#fnref6:2" class="footnote-backref">↩︎</a> <a href="#fnref6:3" class="footnote-backref">↩︎</a> <a href="#fnref6:4" class="footnote-backref">↩︎</a> <a href="#fnref6:5" class="footnote-backref">↩︎</a> <a href="#fnref6:6" class="footnote-backref">↩︎</a> <a href="#fnref6:7" class="footnote-backref">↩︎</a></p></li></ol></section>',30)])])}const k=o(c,[["render",f]]);export{m as __pageData,k as default};
@@ -0,0 +1 @@
1
+ import{_ as o,o as a,c as t,a5 as r}from"./chunks/framework.DSg0KOwT.js";const s="../assets/container-memory.DIO0AnIm.svg",n="../assets/container-memory.dark.CP-5zuCl.svg",m=JSON.parse('{"title":"Cluster Tuning","description":"","frontmatter":{"title":"Cluster Tuning"},"headers":[],"relativePath":"tuning-reference/cluster-config.md","filePath":"tuning-reference/cluster-config.md"}'),c={name:"tuning-reference/cluster-config.md"};function f(i,e,l,u,d,h){return a(),t("div",null,[...e[0]||(e[0]=[r("",30)])])}const k=o(c,[["render",f]]);export{m as __pageData,k as default};