sparkforensics-cli 0.1.0 → 0.2.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (361) hide show
  1. package/README.md +6 -0
  2. package/bin/sparkforensics-analyze.mjs +113 -48
  3. package/export-template/docs/404.html +25 -0
  4. package/export-template/docs/assets/app.DQTZyGL1.js +1 -0
  5. package/export-template/docs/assets/aqe-loop.IwQSATHw.svg +1 -0
  6. package/export-template/docs/assets/aqe-loop.dark.DGbaxqJE.svg +1 -0
  7. package/export-template/docs/assets/broadcast-vs-shuffle.Db4WY1XK.svg +1 -0
  8. package/export-template/docs/assets/broadcast-vs-shuffle.dark.C7Bxs0mG.svg +1 -0
  9. package/export-template/docs/assets/cache-lifecycle.dark.B-hS7AgU.svg +1 -0
  10. package/export-template/docs/assets/cache-lifecycle.rEOVYQNU.svg +1 -0
  11. package/export-template/docs/assets/chunks/@localSearchIndexroot.DppnXnDE.js +1 -0
  12. package/export-template/docs/assets/chunks/VPLocalSearchBox.BkBIPFs6.js +9 -0
  13. package/export-template/docs/assets/chunks/duplicate-plan-subtree.dark.Cdp70QhV.js +1 -0
  14. package/export-template/docs/assets/chunks/framework.DSg0KOwT.js +20 -0
  15. package/export-template/docs/assets/chunks/retry-escalation-ladder.dark.DHipdJgZ.js +1 -0
  16. package/export-template/docs/assets/chunks/theme.DP0u1AUq.js +2 -0
  17. package/export-template/docs/assets/cold-start-timeline.DxC_Sc7w.svg +1 -0
  18. package/export-template/docs/assets/cold-start-timeline.dark.CZ17YcAG.svg +1 -0
  19. package/export-template/docs/assets/columnar-layout.PghGeOEA.svg +1 -0
  20. package/export-template/docs/assets/columnar-layout.dark.BVNlz0ff.svg +1 -0
  21. package/export-template/docs/assets/container-memory.DIO0AnIm.svg +1 -0
  22. package/export-template/docs/assets/container-memory.dark.CP-5zuCl.svg +1 -0
  23. package/export-template/docs/assets/contributor-guide_architecture_board-widgets.md.CWpj01WU.js +1 -0
  24. package/export-template/docs/assets/contributor-guide_architecture_board-widgets.md.CWpj01WU.lean.js +1 -0
  25. package/export-template/docs/assets/contributor-guide_architecture_detector-contract.md.CgzUsQ6W.js +1 -0
  26. package/export-template/docs/assets/contributor-guide_architecture_detector-contract.md.CgzUsQ6W.lean.js +1 -0
  27. package/export-template/docs/assets/contributor-guide_architecture_drill-down.md.BtPdlM7r.js +1 -0
  28. package/export-template/docs/assets/contributor-guide_architecture_drill-down.md.BtPdlM7r.lean.js +1 -0
  29. package/export-template/docs/assets/contributor-guide_architecture_impact-estimation.md.CooslVJt.js +1 -0
  30. package/export-template/docs/assets/contributor-guide_architecture_impact-estimation.md.CooslVJt.lean.js +1 -0
  31. package/export-template/docs/assets/contributor-guide_architecture_index.md.3TO9ic6w.js +1 -0
  32. package/export-template/docs/assets/contributor-guide_architecture_index.md.3TO9ic6w.lean.js +1 -0
  33. package/export-template/docs/assets/contributor-guide_architecture_overview.md.CehiRmGn.js +1 -0
  34. package/export-template/docs/assets/contributor-guide_architecture_overview.md.CehiRmGn.lean.js +1 -0
  35. package/export-template/docs/assets/contributor-guide_architecture_state-and-history.md.C-xxn0q7.js +1 -0
  36. package/export-template/docs/assets/contributor-guide_architecture_state-and-history.md.C-xxn0q7.lean.js +1 -0
  37. package/export-template/docs/assets/contributor-guide_architecture_widget-rendering.md.R27gQrgY.js +1 -0
  38. package/export-template/docs/assets/contributor-guide_architecture_widget-rendering.md.R27gQrgY.lean.js +1 -0
  39. package/export-template/docs/assets/contributor-guide_architecture_worker-protocol.md.IbnfNrV3.js +6 -0
  40. package/export-template/docs/assets/contributor-guide_architecture_worker-protocol.md.IbnfNrV3.lean.js +1 -0
  41. package/export-template/docs/assets/contributor-guide_contributing.md.CvRsdr6J.js +1 -0
  42. package/export-template/docs/assets/contributor-guide_contributing.md.CvRsdr6J.lean.js +1 -0
  43. package/export-template/docs/assets/contributor-guide_development-setup.md.DvAN_9mK.js +12 -0
  44. package/export-template/docs/assets/contributor-guide_development-setup.md.DvAN_9mK.lean.js +1 -0
  45. package/export-template/docs/assets/contributor-guide_testing.md.6rIKqSyY.js +1 -0
  46. package/export-template/docs/assets/contributor-guide_testing.md.6rIKqSyY.lean.js +1 -0
  47. package/export-template/docs/assets/dag-stages.DSz_S937.svg +1 -0
  48. package/export-template/docs/assets/dag-stages.dark.F72UzxH4.svg +1 -0
  49. package/export-template/docs/assets/driver-executor.D5pQ7YN1.svg +1 -0
  50. package/export-template/docs/assets/driver-executor.dark.BmX9cPvh.svg +1 -0
  51. package/export-template/docs/assets/duplicate-plan-subtree.B4cvN6fj.svg +1 -0
  52. package/export-template/docs/assets/duplicate-plan-subtree.dark.Dw8wS0Ag.svg +1 -0
  53. package/export-template/docs/assets/index.md.CHJVslga.js +1 -0
  54. package/export-template/docs/assets/index.md.CHJVslga.lean.js +1 -0
  55. package/export-template/docs/assets/inter-italic-cyrillic-ext.r48I6akx.woff2 +0 -0
  56. package/export-template/docs/assets/inter-italic-cyrillic.By2_1cv3.woff2 +0 -0
  57. package/export-template/docs/assets/inter-italic-greek-ext.1u6EdAuj.woff2 +0 -0
  58. package/export-template/docs/assets/inter-italic-greek.DJ8dCoTZ.woff2 +0 -0
  59. package/export-template/docs/assets/inter-italic-latin-ext.CN1xVJS-.woff2 +0 -0
  60. package/export-template/docs/assets/inter-italic-latin.C2AdPX0b.woff2 +0 -0
  61. package/export-template/docs/assets/inter-italic-vietnamese.BSbpV94h.woff2 +0 -0
  62. package/export-template/docs/assets/inter-roman-cyrillic-ext.BBPuwvHQ.woff2 +0 -0
  63. package/export-template/docs/assets/inter-roman-cyrillic.C5lxZ8CY.woff2 +0 -0
  64. package/export-template/docs/assets/inter-roman-greek-ext.CqjqNYQ-.woff2 +0 -0
  65. package/export-template/docs/assets/inter-roman-greek.BBVDIX6e.woff2 +0 -0
  66. package/export-template/docs/assets/inter-roman-latin-ext.4ZJIpNVo.woff2 +0 -0
  67. package/export-template/docs/assets/inter-roman-latin.Di8DUHzh.woff2 +0 -0
  68. package/export-template/docs/assets/inter-roman-vietnamese.BjW4sHH5.woff2 +0 -0
  69. package/export-template/docs/assets/join-strategy.C_FvrCEo.svg +1 -0
  70. package/export-template/docs/assets/join-strategy.dark.ChMLnNII.svg +1 -0
  71. package/export-template/docs/assets/memory-borrowing.BqQRJg0u.svg +1 -0
  72. package/export-template/docs/assets/memory-borrowing.dark.Yhh20O9C.svg +1 -0
  73. package/export-template/docs/assets/memory-regions.XHvO7jHG.svg +1 -0
  74. package/export-template/docs/assets/memory-regions.dark.D4TP9_08.svg +1 -0
  75. package/export-template/docs/assets/repartition-vs-coalesce.BovLRrpj.svg +1 -0
  76. package/export-template/docs/assets/repartition-vs-coalesce.dark.BhAczKZQ.svg +1 -0
  77. package/export-template/docs/assets/retry-escalation-ladder.DyTKJJmZ.svg +1 -0
  78. package/export-template/docs/assets/retry-escalation-ladder.dark.BdsabtU3.svg +1 -0
  79. package/export-template/docs/assets/shuffle-map-reduce.KuOEZVmg.svg +1 -0
  80. package/export-template/docs/assets/shuffle-map-reduce.dark.BgQZnFSb.svg +1 -0
  81. package/export-template/docs/assets/spill-classification.BU2euYDO.svg +1 -0
  82. package/export-template/docs/assets/spill-classification.dark.D7i1M40d.svg +1 -0
  83. package/export-template/docs/assets/style.DSixAiZE.css +1 -0
  84. package/export-template/docs/assets/tuning-reference_anti-patterns.md.Df1YMIHu.js +1 -0
  85. package/export-template/docs/assets/tuning-reference_anti-patterns.md.Df1YMIHu.lean.js +1 -0
  86. package/export-template/docs/assets/tuning-reference_aqe.md.BIsCtLzm.js +1 -0
  87. package/export-template/docs/assets/tuning-reference_aqe.md.BIsCtLzm.lean.js +1 -0
  88. package/export-template/docs/assets/tuning-reference_bottleneck-broadcast-sizing.md.CEstB3Ia.js +1 -0
  89. package/export-template/docs/assets/tuning-reference_bottleneck-broadcast-sizing.md.CEstB3Ia.lean.js +1 -0
  90. package/export-template/docs/assets/tuning-reference_bottleneck-cold-start.md.CEuy-72y.js +7 -0
  91. package/export-template/docs/assets/tuning-reference_bottleneck-cold-start.md.CEuy-72y.lean.js +1 -0
  92. package/export-template/docs/assets/tuning-reference_bottleneck-duplicate-plan-subtree.md.CIohQDfn.js +1 -0
  93. package/export-template/docs/assets/tuning-reference_bottleneck-duplicate-plan-subtree.md.CIohQDfn.lean.js +1 -0
  94. package/export-template/docs/assets/tuning-reference_bottleneck-failures.md.4z5BXGJ2.js +6 -0
  95. package/export-template/docs/assets/tuning-reference_bottleneck-failures.md.4z5BXGJ2.lean.js +1 -0
  96. package/export-template/docs/assets/tuning-reference_bottleneck-gc.md.DSxzZRK7.js +6 -0
  97. package/export-template/docs/assets/tuning-reference_bottleneck-gc.md.DSxzZRK7.lean.js +1 -0
  98. package/export-template/docs/assets/tuning-reference_bottleneck-job-failure-rate.md.BaJl__1W.js +8 -0
  99. package/export-template/docs/assets/tuning-reference_bottleneck-job-failure-rate.md.BaJl__1W.lean.js +1 -0
  100. package/export-template/docs/assets/tuning-reference_bottleneck-memory-utilization.md.DbP-SJZc.js +7 -0
  101. package/export-template/docs/assets/tuning-reference_bottleneck-memory-utilization.md.DbP-SJZc.lean.js +1 -0
  102. package/export-template/docs/assets/tuning-reference_bottleneck-retry-waste.md.D5JMjVOt.js +1 -0
  103. package/export-template/docs/assets/tuning-reference_bottleneck-retry-waste.md.D5JMjVOt.lean.js +1 -0
  104. package/export-template/docs/assets/tuning-reference_bottleneck-shuffle.md.CM-nTmIH.js +12 -0
  105. package/export-template/docs/assets/tuning-reference_bottleneck-shuffle.md.CM-nTmIH.lean.js +1 -0
  106. package/export-template/docs/assets/tuning-reference_bottleneck-skew.md.BdUwiDhn.js +14 -0
  107. package/export-template/docs/assets/tuning-reference_bottleneck-skew.md.BdUwiDhn.lean.js +1 -0
  108. package/export-template/docs/assets/tuning-reference_bottleneck-slow-host.md.BlIo6UDW.js +7 -0
  109. package/export-template/docs/assets/tuning-reference_bottleneck-slow-host.md.BlIo6UDW.lean.js +1 -0
  110. package/export-template/docs/assets/tuning-reference_bottleneck-small-files.md.B8kloyx8.js +5 -0
  111. package/export-template/docs/assets/tuning-reference_bottleneck-small-files.md.B8kloyx8.lean.js +1 -0
  112. package/export-template/docs/assets/tuning-reference_bottleneck-spill.md.PNH7mITt.js +6 -0
  113. package/export-template/docs/assets/tuning-reference_bottleneck-spill.md.PNH7mITt.lean.js +1 -0
  114. package/export-template/docs/assets/tuning-reference_bottleneck-straggler.md.DY36fHN5.js +7 -0
  115. package/export-template/docs/assets/tuning-reference_bottleneck-straggler.md.DY36fHN5.lean.js +1 -0
  116. package/export-template/docs/assets/tuning-reference_bottleneck-tiny-tasks.md.QTV7O8kU.js +8 -0
  117. package/export-template/docs/assets/tuning-reference_bottleneck-tiny-tasks.md.QTV7O8kU.lean.js +1 -0
  118. package/export-template/docs/assets/tuning-reference_bottleneck-utilization.md.DTiueZC3.js +5 -0
  119. package/export-template/docs/assets/tuning-reference_bottleneck-utilization.md.DTiueZC3.lean.js +1 -0
  120. package/export-template/docs/assets/tuning-reference_caching.md.B7aQ8asB.js +1 -0
  121. package/export-template/docs/assets/tuning-reference_caching.md.B7aQ8asB.lean.js +1 -0
  122. package/export-template/docs/assets/tuning-reference_cluster-config.md.ZVmDGsQ3.js +1 -0
  123. package/export-template/docs/assets/tuning-reference_cluster-config.md.ZVmDGsQ3.lean.js +1 -0
  124. package/export-template/docs/assets/tuning-reference_config.md.UvveiWG3.js +1 -0
  125. package/export-template/docs/assets/tuning-reference_config.md.UvveiWG3.lean.js +1 -0
  126. package/export-template/docs/assets/tuning-reference_data-formats.md.bjCAWH3N.js +1 -0
  127. package/export-template/docs/assets/tuning-reference_data-formats.md.bjCAWH3N.lean.js +1 -0
  128. package/export-template/docs/assets/tuning-reference_index.md.BQ_NooMV.js +1 -0
  129. package/export-template/docs/assets/tuning-reference_index.md.BQ_NooMV.lean.js +1 -0
  130. package/export-template/docs/assets/tuning-reference_intro.md.CobD-lGB.js +1 -0
  131. package/export-template/docs/assets/tuning-reference_intro.md.CobD-lGB.lean.js +1 -0
  132. package/export-template/docs/assets/tuning-reference_joins.md.BtKs_CuW.js +1 -0
  133. package/export-template/docs/assets/tuning-reference_joins.md.BtKs_CuW.lean.js +1 -0
  134. package/export-template/docs/assets/tuning-reference_memory-model.md.DhT-n4y3.js +1 -0
  135. package/export-template/docs/assets/tuning-reference_memory-model.md.DhT-n4y3.lean.js +1 -0
  136. package/export-template/docs/assets/tuning-reference_metrics.md.mLOh7Apj.js +1 -0
  137. package/export-template/docs/assets/tuning-reference_metrics.md.mLOh7Apj.lean.js +1 -0
  138. package/export-template/docs/assets/tuning-reference_partitioning.md.q0zKF_8X.js +1 -0
  139. package/export-template/docs/assets/tuning-reference_partitioning.md.q0zKF_8X.lean.js +1 -0
  140. package/export-template/docs/assets/tuning-reference_pyspark.md.DDCfvN9t.js +6 -0
  141. package/export-template/docs/assets/tuning-reference_pyspark.md.DDCfvN9t.lean.js +1 -0
  142. package/export-template/docs/assets/tuning-reference_shuffle.md.BZZ7R4Ix.js +1 -0
  143. package/export-template/docs/assets/tuning-reference_shuffle.md.BZZ7R4Ix.lean.js +1 -0
  144. package/export-template/docs/assets/tuning-reference_spark-architecture.md.Dwzm5avO.js +1 -0
  145. package/export-template/docs/assets/tuning-reference_spark-architecture.md.Dwzm5avO.lean.js +1 -0
  146. package/export-template/docs/assets/tuning-reference_table-formats.md.D6wj-2dX.js +1 -0
  147. package/export-template/docs/assets/tuning-reference_table-formats.md.D6wj-2dX.lean.js +1 -0
  148. package/export-template/docs/assets/udf-execution-models.BUFDICuG.svg +1 -0
  149. package/export-template/docs/assets/udf-execution-models.dark.YTNS6GDq.svg +1 -0
  150. package/export-template/docs/assets/user-guide_alternative-log-retrieval.md.B4tPGIal.js +1 -0
  151. package/export-template/docs/assets/user-guide_alternative-log-retrieval.md.B4tPGIal.lean.js +1 -0
  152. package/export-template/docs/assets/user-guide_getting-started.md.BJvwLEIM.js +3 -0
  153. package/export-template/docs/assets/user-guide_getting-started.md.BJvwLEIM.lean.js +1 -0
  154. package/export-template/docs/assets/user-guide_mcp-tools.md.Vi3RoflJ.js +125 -0
  155. package/export-template/docs/assets/user-guide_mcp-tools.md.Vi3RoflJ.lean.js +1 -0
  156. package/export-template/docs/assets/user-guide_run-comparison.md.CQc1aoU8.js +1 -0
  157. package/export-template/docs/assets/user-guide_run-comparison.md.CQc1aoU8.lean.js +1 -0
  158. package/export-template/docs/assets/user-guide_understanding-findings.md.DL1UDhvR.js +1 -0
  159. package/export-template/docs/assets/user-guide_understanding-findings.md.DL1UDhvR.lean.js +1 -0
  160. package/export-template/docs/contributor-guide/architecture/board-widgets.html +25 -0
  161. package/export-template/docs/contributor-guide/architecture/detector-contract.html +25 -0
  162. package/export-template/docs/contributor-guide/architecture/drill-down.html +25 -0
  163. package/export-template/docs/contributor-guide/architecture/impact-estimation.html +25 -0
  164. package/export-template/docs/contributor-guide/architecture/index.html +25 -0
  165. package/export-template/docs/contributor-guide/architecture/overview.html +25 -0
  166. package/export-template/docs/contributor-guide/architecture/state-and-history.html +25 -0
  167. package/export-template/docs/contributor-guide/architecture/widget-rendering.html +25 -0
  168. package/export-template/docs/contributor-guide/architecture/worker-protocol.html +30 -0
  169. package/export-template/docs/contributor-guide/contributing.html +25 -0
  170. package/export-template/docs/contributor-guide/development-setup.html +36 -0
  171. package/export-template/docs/contributor-guide/testing.html +25 -0
  172. package/export-template/docs/favicon.svg +4 -0
  173. package/export-template/docs/hashmap.json +1 -0
  174. package/export-template/docs/index.html +25 -0
  175. package/export-template/docs/package.json +1 -0
  176. package/export-template/docs/tuning-reference/anti-patterns.html +25 -0
  177. package/export-template/docs/tuning-reference/aqe.html +25 -0
  178. package/export-template/docs/tuning-reference/bottleneck-broadcast-sizing.html +25 -0
  179. package/export-template/docs/tuning-reference/bottleneck-cold-start.html +31 -0
  180. package/export-template/docs/tuning-reference/bottleneck-duplicate-plan-subtree.html +25 -0
  181. package/export-template/docs/tuning-reference/bottleneck-failures.html +30 -0
  182. package/export-template/docs/tuning-reference/bottleneck-gc.html +30 -0
  183. package/export-template/docs/tuning-reference/bottleneck-job-failure-rate.html +32 -0
  184. package/export-template/docs/tuning-reference/bottleneck-memory-utilization.html +31 -0
  185. package/export-template/docs/tuning-reference/bottleneck-retry-waste.html +25 -0
  186. package/export-template/docs/tuning-reference/bottleneck-shuffle.html +36 -0
  187. package/export-template/docs/tuning-reference/bottleneck-skew.html +38 -0
  188. package/export-template/docs/tuning-reference/bottleneck-slow-host.html +31 -0
  189. package/export-template/docs/tuning-reference/bottleneck-small-files.html +29 -0
  190. package/export-template/docs/tuning-reference/bottleneck-spill.html +30 -0
  191. package/export-template/docs/tuning-reference/bottleneck-straggler.html +31 -0
  192. package/export-template/docs/tuning-reference/bottleneck-tiny-tasks.html +32 -0
  193. package/export-template/docs/tuning-reference/bottleneck-utilization.html +29 -0
  194. package/export-template/docs/tuning-reference/caching.html +25 -0
  195. package/export-template/docs/tuning-reference/cluster-config.html +25 -0
  196. package/export-template/docs/tuning-reference/config.html +25 -0
  197. package/export-template/docs/tuning-reference/data-formats.html +25 -0
  198. package/export-template/docs/tuning-reference/index.html +25 -0
  199. package/export-template/docs/tuning-reference/intro.html +25 -0
  200. package/export-template/docs/tuning-reference/joins.html +25 -0
  201. package/export-template/docs/tuning-reference/memory-model.html +25 -0
  202. package/export-template/docs/tuning-reference/metrics.html +25 -0
  203. package/export-template/docs/tuning-reference/partitioning.html +25 -0
  204. package/export-template/docs/tuning-reference/pyspark.html +30 -0
  205. package/export-template/docs/tuning-reference/shuffle.html +25 -0
  206. package/export-template/docs/tuning-reference/spark-architecture.html +25 -0
  207. package/export-template/docs/tuning-reference/table-formats.html +25 -0
  208. package/export-template/docs/user-guide/alternative-log-retrieval.html +25 -0
  209. package/export-template/docs/user-guide/getting-started.html +27 -0
  210. package/export-template/docs/user-guide/mcp-tools.html +149 -0
  211. package/export-template/docs/user-guide/run-comparison.html +25 -0
  212. package/export-template/docs/user-guide/understanding-findings.html +25 -0
  213. package/export-template/docs/vp-icons.css +0 -0
  214. package/export-template/favicon.svg +4 -0
  215. package/export-template/index.html +111 -0
  216. package/export-template/parser-worker-DyjiQvfP.js +112 -0
  217. package/export-template/sample-runs/sample-run.ndjson.gz +0 -0
  218. package/package.json +20 -6
  219. package/vendor-core/analyzer.js +74 -74
  220. package/vendor-core/cli/budgets.js +13 -27
  221. package/vendor-core/cli/collect-run.js +43 -19
  222. package/vendor-core/core-count.js +25 -27
  223. package/vendor-core/core-locality-ratio.js +4 -11
  224. package/vendor-core/core-time-series.js +6 -12
  225. package/vendor-core/core-usage-locality.js +3 -4
  226. package/vendor-core/detectors.js +395 -389
  227. package/vendor-core/docs-config.js +69 -21
  228. package/vendor-core/docs-content/chapters/01-intro.md +32 -0
  229. package/vendor-core/docs-content/chapters/02-spark-architecture.md +76 -0
  230. package/vendor-core/docs-content/chapters/03-memory-model.md +73 -0
  231. package/vendor-core/docs-content/chapters/04-partitioning.md +65 -0
  232. package/vendor-core/docs-content/chapters/05-joins.md +62 -0
  233. package/vendor-core/docs-content/chapters/06-shuffle.md +59 -0
  234. package/vendor-core/docs-content/chapters/07-data-formats.md +81 -0
  235. package/vendor-core/docs-content/chapters/07b-table-formats.md +56 -0
  236. package/vendor-core/docs-content/chapters/08-caching.md +58 -0
  237. package/vendor-core/docs-content/chapters/09-pyspark.md +78 -0
  238. package/vendor-core/docs-content/chapters/10-aqe.md +167 -0
  239. package/vendor-core/docs-content/chapters/11-cluster-config.md +170 -0
  240. package/vendor-core/docs-content/chapters/12-anti-patterns.md +171 -0
  241. package/vendor-core/docs-content/chapters/14-metrics.md +87 -0
  242. package/vendor-core/docs-content/chapters/15-config.md +93 -0
  243. package/vendor-core/docs-content/chapters/nav-index.json +370 -0
  244. package/vendor-core/docs-content/detection/cache.md +7 -0
  245. package/vendor-core/docs-content/detection/cfg.md +15 -0
  246. package/vendor-core/docs-content/detection/chrn.md +9 -0
  247. package/vendor-core/docs-content/detection/cold.md +4 -0
  248. package/vendor-core/docs-content/detection/cstor.md +4 -0
  249. package/vendor-core/docs-content/detection/fail.md +5 -0
  250. package/vendor-core/docs-content/detection/gc.md +4 -0
  251. package/vendor-core/docs-content/detection/host.md +5 -0
  252. package/vendor-core/docs-content/detection/incmp.md +6 -0
  253. package/vendor-core/docs-content/detection/jobs.md +4 -0
  254. package/vendor-core/docs-content/detection/local.md +6 -0
  255. package/vendor-core/docs-content/detection/mem.md +10 -0
  256. package/vendor-core/docs-content/detection/part.md +5 -0
  257. package/vendor-core/docs-content/detection/plan.md +14 -0
  258. package/vendor-core/docs-content/detection/retry.md +4 -0
  259. package/vendor-core/docs-content/detection/sfail.md +5 -0
  260. package/vendor-core/docs-content/detection/shape.md +5 -0
  261. package/vendor-core/docs-content/detection/shfl.md +4 -0
  262. package/vendor-core/docs-content/detection/skew.md +6 -0
  263. package/vendor-core/docs-content/detection/slow.md +6 -0
  264. package/vendor-core/docs-content/detection/spec.md +8 -0
  265. package/vendor-core/docs-content/detection/spill.md +7 -0
  266. package/vendor-core/docs-content/detection/strag.md +5 -0
  267. package/vendor-core/docs-content/detection/tiny.md +4 -0
  268. package/vendor-core/docs-content/detection/util.md +4 -0
  269. package/vendor-core/docs-content/diagrams/aqe-loop.dark.svg +1 -0
  270. package/vendor-core/docs-content/diagrams/aqe-loop.svg +1 -0
  271. package/vendor-core/docs-content/diagrams/broadcast-vs-shuffle.dark.svg +1 -0
  272. package/vendor-core/docs-content/diagrams/broadcast-vs-shuffle.svg +1 -0
  273. package/vendor-core/docs-content/diagrams/cache-lifecycle.dark.svg +1 -0
  274. package/vendor-core/docs-content/diagrams/cache-lifecycle.svg +1 -0
  275. package/vendor-core/docs-content/diagrams/cold-start-timeline.dark.svg +1 -0
  276. package/vendor-core/docs-content/diagrams/cold-start-timeline.svg +1 -0
  277. package/vendor-core/docs-content/diagrams/columnar-layout.dark.svg +1 -0
  278. package/vendor-core/docs-content/diagrams/columnar-layout.svg +1 -0
  279. package/vendor-core/docs-content/diagrams/container-memory.dark.svg +1 -0
  280. package/vendor-core/docs-content/diagrams/container-memory.svg +1 -0
  281. package/vendor-core/docs-content/diagrams/dag-stages.dark.svg +1 -0
  282. package/vendor-core/docs-content/diagrams/dag-stages.svg +1 -0
  283. package/vendor-core/docs-content/diagrams/driver-executor.dark.svg +1 -0
  284. package/vendor-core/docs-content/diagrams/driver-executor.svg +1 -0
  285. package/vendor-core/docs-content/diagrams/duplicate-plan-subtree.dark.svg +1 -0
  286. package/vendor-core/docs-content/diagrams/duplicate-plan-subtree.svg +1 -0
  287. package/vendor-core/docs-content/diagrams/join-strategy.dark.svg +1 -0
  288. package/vendor-core/docs-content/diagrams/join-strategy.svg +1 -0
  289. package/vendor-core/docs-content/diagrams/memory-borrowing.dark.svg +1 -0
  290. package/vendor-core/docs-content/diagrams/memory-borrowing.svg +1 -0
  291. package/vendor-core/docs-content/diagrams/memory-regions.dark.svg +1 -0
  292. package/vendor-core/docs-content/diagrams/memory-regions.svg +1 -0
  293. package/vendor-core/docs-content/diagrams/repartition-vs-coalesce.dark.svg +1 -0
  294. package/vendor-core/docs-content/diagrams/repartition-vs-coalesce.svg +1 -0
  295. package/vendor-core/docs-content/diagrams/retry-escalation-ladder.dark.svg +1 -0
  296. package/vendor-core/docs-content/diagrams/retry-escalation-ladder.svg +1 -0
  297. package/vendor-core/docs-content/diagrams/shuffle-map-reduce.dark.svg +1 -0
  298. package/vendor-core/docs-content/diagrams/shuffle-map-reduce.svg +1 -0
  299. package/vendor-core/docs-content/diagrams/spill-classification.dark.svg +1 -0
  300. package/vendor-core/docs-content/diagrams/spill-classification.svg +1 -0
  301. package/vendor-core/docs-content/diagrams/udf-execution-models.dark.svg +1 -0
  302. package/vendor-core/docs-content/diagrams/udf-execution-models.svg +1 -0
  303. package/vendor-core/docs-content/tuning/broadcast-sizing.md +78 -0
  304. package/vendor-core/docs-content/tuning/cold-start.md +81 -0
  305. package/vendor-core/docs-content/tuning/duplicate-plan-subtree.md +45 -0
  306. package/vendor-core/docs-content/tuning/failures.md +124 -0
  307. package/vendor-core/docs-content/tuning/gc.md +110 -0
  308. package/vendor-core/docs-content/tuning/job-failure-rate.md +101 -0
  309. package/vendor-core/docs-content/tuning/memory-utilization.md +58 -0
  310. package/vendor-core/docs-content/tuning/retry-waste.md +90 -0
  311. package/vendor-core/docs-content/tuning/shuffle.md +154 -0
  312. package/vendor-core/docs-content/tuning/skew.md +123 -0
  313. package/vendor-core/docs-content/tuning/slow-host.md +117 -0
  314. package/vendor-core/docs-content/tuning/small-files.md +99 -0
  315. package/vendor-core/docs-content/tuning/spill.md +114 -0
  316. package/vendor-core/docs-content/tuning/straggler.md +103 -0
  317. package/vendor-core/docs-content/tuning/tiny-tasks.md +94 -0
  318. package/vendor-core/docs-content/tuning/utilization.md +90 -0
  319. package/vendor-core/docs-site-config.js +10 -17
  320. package/vendor-core/efficiency-model.js +7 -13
  321. package/vendor-core/etl-phases.js +3 -5
  322. package/vendor-core/event-handlers.js +232 -134
  323. package/vendor-core/event-schemas.js +48 -114
  324. package/vendor-core/evidence-availability.js +5 -10
  325. package/vendor-core/evidence-report.js +73 -123
  326. package/vendor-core/export-data.js +48 -0
  327. package/vendor-core/finding-action-label.js +4 -10
  328. package/vendor-core/finding-filter-predicate.js +3 -7
  329. package/vendor-core/finding-generic-recommendation.js +112 -0
  330. package/vendor-core/finding-names.js +51 -0
  331. package/vendor-core/format-utils.js +112 -38
  332. package/vendor-core/impact-band.js +18 -24
  333. package/vendor-core/impact-estimator.js +38 -74
  334. package/vendor-core/ingest.js +7 -13
  335. package/vendor-core/job-groups.js +3 -6
  336. package/vendor-core/list-runs.js +278 -0
  337. package/vendor-core/load-vendored.js +6 -12
  338. package/vendor-core/log-header-peek.js +81 -0
  339. package/vendor-core/lz4-block.js +4 -6
  340. package/vendor-core/mcp-server-factory.js +38 -8
  341. package/vendor-core/mcp-tools.js +105 -76
  342. package/vendor-core/model-assembler.js +8 -16
  343. package/vendor-core/occupancy.js +5 -9
  344. package/vendor-core/parser-worker.js +20 -29
  345. package/vendor-core/plan-dot.js +2 -5
  346. package/vendor-core/plan-duration-attribution.js +78 -29
  347. package/vendor-core/plan-graph-model.js +126 -69
  348. package/vendor-core/plan-node-detail.js +31 -17
  349. package/vendor-core/plan-summary.js +19 -8
  350. package/vendor-core/recommendation-rollup.js +35 -39
  351. package/vendor-core/redact.js +72 -16
  352. package/vendor-core/rolling-log-reassembly.js +4 -6
  353. package/vendor-core/run-comparison.js +86 -72
  354. package/vendor-core/scaling-sim.js +5 -7
  355. package/vendor-core/session-snapshot.js +1 -1
  356. package/vendor-core/shs-fetch.js +4 -6
  357. package/vendor-core/shs-load.js +9 -13
  358. package/vendor-core/shs-request.js +1 -1
  359. package/vendor-core/stage-quantiles.js +14 -0
  360. package/vendor-core/types.js +78 -18
  361. package/vendor-core/wasted-core-hours.js +7 -12
@@ -0,0 +1,6 @@
1
+ import{_ as a,o,c as s,a5 as t}from"./chunks/framework.DSg0KOwT.js";const r="../assets/udf-execution-models.BUFDICuG.svg",n="../assets/udf-execution-models.dark.YTNS6GDq.svg",k=JSON.parse('{"title":"PySpark Specifics","description":"","frontmatter":{"title":"PySpark Specifics"},"headers":[],"relativePath":"tuning-reference/pyspark.md","filePath":"tuning-reference/pyspark.md"}'),i={name:"tuning-reference/pyspark.md"};function f(c,e,p,h,l,d){return o(),s("div",null,[...e[0]||(e[0]=[t('<h1 id="pyspark" tabindex="-1">PySpark Specifics <a class="header-anchor" href="#pyspark" aria-label="Permalink to &quot;PySpark Specifics {#pyspark}&quot;">​</a></h1><h2 id="crossing-the-jvm-python-boundary" tabindex="-1">Crossing the JVM/Python boundary <a class="header-anchor" href="#crossing-the-jvm-python-boundary" aria-label="Permalink to &quot;Crossing the JVM/Python boundary&quot;">​</a></h2><p>A plain PySpark UDF runs one row at a time in a separate Python process, and getting each row there costs something real: &quot;PySpark UDFs required data movement between the JVM and Python, which was quite expensive,&quot; using pickle to serialize the data across that boundary<sup class="footnote-ref"><a href="#fn1" id="fnref1">[1]</a></sup>. <em>Spark: The Definitive Guide</em> breaks the cost down further. Starting the extra Python process is one expense, but &quot;the real cost is in serializing the data to Python,&quot; and once the data is over there the JVM can no longer manage that worker&#39;s memory, so JVM and Python end up competing for the same machine&#39;s RAM and the worker can fail under pressure<sup class="footnote-ref"><a href="#fn2" id="fnref2">[2]</a></sup>. That&#39;s the underlying reason the same source recommends writing UDFs in Scala or Java and calling them from Python when that&#39;s practical<sup class="footnote-ref"><a href="#fn2" id="fnref2:1">[2:1]</a></sup>.</p><p>Since Spark 3.4 the row-at-a-time path can itself be Arrow-optimized without giving up row semantics: set <code>useArrow=True</code> on <code>udf()</code>, or flip the session-wide <code>spark.sql.execution.pythonUDF.arrow.enabled</code> (default <code>false</code>), and a regular Python UDF becomes an &quot;Arrow Python UDF&quot; that still runs row by row but moves data with Arrow instead of pickle<sup class="footnote-ref"><a href="#fn3" id="fnref3">[3]</a></sup><sup class="footnote-ref"><a href="#fn4" id="fnref4">[4]</a></sup>. The session config only takes effect when <code>useArrow</code> is left unset on the UDF itself<sup class="footnote-ref"><a href="#fn3" id="fnref3:1">[3:1]</a></sup>.</p><img class="light-only" src="'+r+'" alt="Across the JVM to Python boundary a plain Python UDF pickles row-at-a-time, an Arrow-optimized UDF uses Arrow transfer with row semantics, and a pandas UDF passes whole Arrow batches with no per-row handoff."><img class="dark-only" src="'+n+`" alt="Across the JVM to Python boundary a plain Python UDF pickles row-at-a-time, an Arrow-optimized UDF uses Arrow transfer with row semantics, and a pandas UDF passes whole Arrow batches with no per-row handoff."><p>The pandas UDF (vectorized UDF, introduced in Spark 2.3) goes further and drops the per-row JVM↔Python handoff entirely: it hands the Python worker whole Arrow batches, operated on as pandas Series or DataFrames, so there&#39;s nothing to pickle row by row<sup class="footnote-ref"><a href="#fn1" id="fnref1:1">[1:1]</a></sup>. Three shapes cover most cases. <strong>SCALAR</strong> (<code>pandas.Series, ... -&gt; pandas.Series</code>) is the direct vectorized swap-in for a row-at-a-time scalar UDF (computing <code>v + 1</code>, or <code>cubed(x)</code>); PySpark calls the function once per Arrow batch and concatenates the results back into a column<sup class="footnote-ref"><a href="#fn5" id="fnref5">[5]</a></sup><sup class="footnote-ref"><a href="#fn1" id="fnref1:2">[1:2]</a></sup>. <strong>SCALAR_ITER</strong> (<code>Iterator[Series] -&gt; Iterator[Series]</code>) works the same way internally, but takes and yields an iterator instead of a single Series, which lets a function prefetch across batches<sup class="footnote-ref"><a href="#fn3" id="fnref3:2">[3:2]</a></sup>. <strong>MAP_ITER</strong>, exposed as <code>DataFrame.mapInPandas()</code> rather than as a <code>pandas_udf</code> type, maps an iterator of whole <code>pandas.DataFrame</code> partitions to another iterator of <code>pandas.DataFrame</code>s and, unlike the other two, can change the row count<sup class="footnote-ref"><a href="#fn3" id="fnref3:3">[3:3]</a></sup>. A related grouped-map API, <code>DataFrame.groupBy().applyInPandas()</code>, splits the DataFrame into groups and runs a <code>pandas.DataFrame -&gt; pandas.DataFrame</code> function per group: split, apply, combine<sup class="footnote-ref"><a href="#fn5" id="fnref5:1">[5:1]</a></sup><sup class="footnote-ref"><a href="#fn3" id="fnref3:4">[3:4]</a></sup>.</p><p><code>spark.sql.execution.arrow.pyspark.enabled</code> isn&#39;t limited to pandas UDFs, either: it also governs Arrow-based columnar transfer for <code>DataFrame.toPandas()</code> and for <code>SparkSession.createDataFrame()</code> when given a pandas DataFrame or NumPy ndarray<sup class="footnote-ref"><a href="#fn4" id="fnref4:1">[4:1]</a></sup><sup class="footnote-ref"><a href="#fn3" id="fnref3:5">[3:5]</a></sup>. A companion flag, <code>spark.sql.execution.arrow.pyspark.fallback.enabled</code>, silently drops back to the non-Arrow path if an error occurs before computation starts<sup class="footnote-ref"><a href="#fn3" id="fnref3:6">[3:6]</a></sup><sup class="footnote-ref"><a href="#fn4" id="fnref4:2">[4:2]</a></sup>.</p><p>Two worker-level configs round out the picture. <code>spark.python.worker.reuse</code> (default <code>true</code>) keeps a fixed pool of Python worker processes alive across tasks instead of forking a fresh one each time, which also means a large broadcast variable doesn&#39;t have to cross the JVM↔Python boundary again for every task<sup class="footnote-ref"><a href="#fn4" id="fnref4:3">[4:3]</a></sup>. <code>spark.python.worker.memory</code> (default <code>512m</code>) is a per-worker, Spark-managed accounting threshold for in-worker aggregation buffering (not an OS-enforced cap), and Spark <a href="./bottleneck-spill.html">spills to disk</a> once it&#39;s exceeded<sup class="footnote-ref"><a href="#fn4" id="fnref4:4">[4:4]</a></sup>.</p><h2 id="measuring-the-gap" tabindex="-1">Measuring the gap <a class="header-anchor" href="#measuring-the-gap" aria-label="Permalink to &quot;Measuring the gap&quot;">​</a></h2><p>The clearest measurement here is workload-level. Databricks&#39; introductory pandas-UDF post ran three operations (Plus One, Cumulative Probability, Subtract Mean) over a 10M-row, two-column DataFrame on a single-node Databricks Community Edition cluster, and found pandas UDFs &quot;perform much better than row-at-a-time UDFs across the board, ranging from 3x to over 100x&quot;<sup class="footnote-ref"><a href="#fn5" id="fnref5:2">[5:2]</a></sup>. If a job&#39;s Python UDFs are a suspected bottleneck, expect an order-of-magnitude gap between a row-at-a-time UDF and its pandas-UDF equivalent, not a marginal one.</p><p>The comparisons in the corpus consistently favor <code>pyspark.sql.functions</code> and vectorized code over row-at-a-time UDFs. For the &quot;plus one&quot; example, &quot;built-in column operators can perform much faster in this scenario,&quot; and the pandas UDF equivalent is &quot;much faster than the row-at-a-time version&quot; because it&#39;s vectorized over the Series<sup class="footnote-ref"><a href="#fn5" id="fnref5:3">[5:3]</a></sup>. A Python UDF that could instead be expressed with built-in column functions is worth flagging on its own.</p><p>On the memory side, the two failure modes look different and are worth telling apart. A worker that spills is showing up in Spark&#39;s own accounting: it exceeded <code>spark.python.worker.memory</code> during aggregation and wrote to disk, which is expected behavior, not a crash<sup class="footnote-ref"><a href="#fn4" id="fnref4:5">[4:5]</a></sup>. A worker that&#39;s actually killed is a container-level event: if total container memory (JVM heap plus off-heap plus the Python process) exceeds what YARN or Kubernetes allocated, &quot;Kubernetes won&#39;t hesitate&quot; and the process is &quot;OOMKilled&quot;<sup class="footnote-ref"><a href="#fn6" id="fnref6">[6]</a></sup>. That boundary is governed by <a href="./memory-model.html">executor overhead</a> and <code>spark.executor.pyspark.memory</code> settings, not by <code>worker.memory</code>.</p><h2 id="what-the-gap-costs" tabindex="-1">What the gap costs <a class="header-anchor" href="#what-the-gap-costs" aria-label="Permalink to &quot;What the gap costs&quot;">​</a></h2><p>Those numbers point to a structural cost, not a tuning quirk. The JVM↔Python boundary is where a plain Python UDF pays twice: once to start the separate process, and again (the larger cost) to serialize every row across it<sup class="footnote-ref"><a href="#fn2" id="fnref2:2">[2:2]</a></sup>. Because the JVM can&#39;t manage memory inside the Python process once data has crossed over, the two runtimes end up competing for the same machine&#39;s memory, and the Python worker can fail under that pressure<sup class="footnote-ref"><a href="#fn2" id="fnref2:3">[2:3]</a></sup>. That&#39;s a correctness risk as well as a performance one, and it&#39;s why <em>Spark: The Definitive Guide</em> recommends Scala/Java UDFs called from Python over native Python UDFs wherever that&#39;s practical<sup class="footnote-ref"><a href="#fn2" id="fnref2:4">[2:4]</a></sup>.</p><p>The magnitude backs this up: Databricks measured pandas UDFs beating row-at-a-time UDFs by 3x to over 100x depending on the operation<sup class="footnote-ref"><a href="#fn5" id="fnref5:4">[5:4]</a></sup>. Skipping per-row pickling (by moving to Arrow-based row UDFs or, further, to pandas UDFs operating on whole batches) isn&#39;t a marginal tuning knob here; it changes which order of magnitude a job runs at.</p><p>The two memory configs matter for different reasons, too. <code>spark.python.worker.memory</code> only controls when Spark chooses to spill aggregation state to disk; tuning it trades disk I/O for headroom, it doesn&#39;t prevent a crash<sup class="footnote-ref"><a href="#fn4" id="fnref4:6">[4:6]</a></sup>. Actual OOM kills happen at the container boundary, and <code>spark.executor.pyspark.memory</code> (which leans on Python&#39;s <code>resource</code> module and so doesn&#39;t cap memory on macOS and doesn&#39;t exist at all on Windows) is the config that&#39;s actually in that path<sup class="footnote-ref"><a href="#fn4" id="fnref4:7">[4:7]</a></sup>. Conflating the two means tuning the wrong knob when a Python worker gets OOMKilled.</p><h2 id="closing-the-gap" tabindex="-1">Closing the gap <a class="header-anchor" href="#closing-the-gap" aria-label="Permalink to &quot;Closing the gap&quot;">​</a></h2><p>Closing that gap means avoiding the boundary crossing altogether, or crossing it as cheaply as possible. Where the logic allows it, prefer a built-in <code>pyspark.sql.functions</code> expression or a Scala/Java UDF called from Python over a plain row-at-a-time Python UDF<sup class="footnote-ref"><a href="#fn2" id="fnref2:5">[2:5]</a></sup><sup class="footnote-ref"><a href="#fn5" id="fnref5:5">[5:5]</a></sup>; this is the change with the largest documented payoff, 3x to over 100x<sup class="footnote-ref"><a href="#fn5" id="fnref5:6">[5:6]</a></sup>.</p><p>Where a Python UDF is unavoidable, cut the row-at-a-time cost first by turning on Arrow for it:</p><div class="language-python vp-adaptive-theme"><button title="Copy Code" class="copy"></button><span class="lang">python</span><pre class="shiki shiki-themes github-light github-dark vp-code" tabindex="0"><code><span class="line"><span style="--shiki-light:#6F42C1;--shiki-dark:#B392F0;">@udf</span><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;">(</span><span style="--shiki-light:#E36209;--shiki-dark:#FFAB70;">returnType</span><span style="--shiki-light:#D73A49;--shiki-dark:#F97583;">=</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;">&#39;int&#39;</span><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;">, </span><span style="--shiki-light:#E36209;--shiki-dark:#FFAB70;">useArrow</span><span style="--shiki-light:#D73A49;--shiki-dark:#F97583;">=</span><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;">True</span><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;">) </span><span style="--shiki-light:#6A737D;--shiki-dark:#6A737D;"># An Arrow Python UDF</span></span>
2
+ <span class="line"><span style="--shiki-light:#D73A49;--shiki-dark:#F97583;">def</span><span style="--shiki-light:#6F42C1;--shiki-dark:#B392F0;"> arrow_slen</span><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;">(s):</span></span>
3
+ <span class="line"><span style="--shiki-light:#D73A49;--shiki-dark:#F97583;"> return</span><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> len</span><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;">(s)</span></span></code></pre></div><p>or set the session-wide equivalent so existing UDFs pick it up without code changes.</p><blockquote><p><strong>PySpark:</strong> <code>spark.conf.set(&quot;spark.sql.execution.pythonUDF.arrow.enabled&quot;, &quot;true&quot;)</code> turns plain UDFs into Arrow-backed ones, as long as <code>useArrow</code> isn&#39;t explicitly set on the UDF itself<sup class="footnote-ref"><a href="#fn3" id="fnref3:7">[3:7]</a></sup><sup class="footnote-ref"><a href="#fn4" id="fnref4:8">[4:8]</a></sup>.</p></blockquote><p>Beyond that, reach for the vectorized APIs instead of a scalar UDF:</p><ul><li>Use a <strong>SCALAR</strong> pandas UDF (<code>pandas.Series, ... -&gt; pandas.Series</code>) as the default vectorized replacement for a row-at-a-time scalar UDF<sup class="footnote-ref"><a href="#fn5" id="fnref5:7">[5:7]</a></sup><sup class="footnote-ref"><a href="#fn1" id="fnref1:3">[1:3]</a></sup>.</li><li>Use <strong>SCALAR_ITER</strong> (<code>Iterator[Series] -&gt; Iterator[Series]</code>) when the function needs expensive one-time setup. The documented pattern initializes state once, then loops over the batch iterator reusing it, instead of re-initializing per batch<sup class="footnote-ref"><a href="#fn3" id="fnref3:8">[3:8]</a></sup>:</li></ul><div class="language-python vp-adaptive-theme"><button title="Copy Code" class="copy"></button><span class="lang">python</span><pre class="shiki shiki-themes github-light github-dark vp-code" tabindex="0"><code><span class="line"><span style="--shiki-light:#D73A49;--shiki-dark:#F97583;">def</span><span style="--shiki-light:#6F42C1;--shiki-dark:#B392F0;"> apply_with_state</span><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;">(iterator):</span></span>
4
+ <span class="line"><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;"> state </span><span style="--shiki-light:#D73A49;--shiki-dark:#F97583;">=</span><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;"> very_expensive_initialization()</span></span>
5
+ <span class="line"><span style="--shiki-light:#D73A49;--shiki-dark:#F97583;"> for</span><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;"> batch </span><span style="--shiki-light:#D73A49;--shiki-dark:#F97583;">in</span><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;"> iterator:</span></span>
6
+ <span class="line"><span style="--shiki-light:#D73A49;--shiki-dark:#F97583;"> yield</span><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;"> calculate_with_state(batch, state)</span></span></code></pre></div><ul><li>Use <code>DataFrame.mapInPandas()</code> when the transform needs to change row count (filtering, expansion, deduplication) rather than map one-to-one<sup class="footnote-ref"><a href="#fn3" id="fnref3:9">[3:9]</a></sup>.</li><li>Use <code>DataFrame.groupBy().applyInPandas()</code> for per-group logic that needs the whole group as state, such as subtracting a group mean or fitting a per-group regression<sup class="footnote-ref"><a href="#fn5" id="fnref5:8">[5:8]</a></sup><sup class="footnote-ref"><a href="#fn3" id="fnref3:10">[3:10]</a></sup>. Size groups with care: a full group loads into memory before the function runs, and <code>maxRecordsPerBatch</code> doesn&#39;t apply to groups, so a skewed group risks OOM<sup class="footnote-ref"><a href="#fn3" id="fnref3:11">[3:11]</a></sup>.</li></ul><p>For pandas/NumPy conversion at the driver, turn on Arrow explicitly rather than relying on defaults.</p><blockquote><p><strong>PySpark:</strong> with <code>spark.sql.execution.arrow.pyspark.enabled</code> set to <code>&quot;true&quot;</code>, <code>spark.createDataFrame(pdf)</code> and <code>df.select(&quot;*&quot;).toPandas()</code> convert through Arrow and return the same result as the non-Arrow path<sup class="footnote-ref"><a href="#fn4" id="fnref4:9">[4:9]</a></sup><sup class="footnote-ref"><a href="#fn3" id="fnref3:12">[3:12]</a></sup>. <code>spark.sql.execution.arrow.pyspark.fallback.enabled</code> keeps a safety net by falling back silently on pre-computation errors<sup class="footnote-ref"><a href="#fn3" id="fnref3:13">[3:13]</a></sup><sup class="footnote-ref"><a href="#fn4" id="fnref4:10">[4:10]</a></sup>.</p></blockquote><p>Leave <code>spark.python.worker.reuse</code> at its default (<code>true</code>) unless there&#39;s a specific reason not to; it keeps a fixed pool of Python workers alive so a large broadcast variable isn&#39;t re-shipped to Python for every task<sup class="footnote-ref"><a href="#fn4" id="fnref4:11">[4:11]</a></sup>. If a job spills at the Python-worker level, <code>spark.python.worker.memory</code> is the knob for that; if it&#39;s getting OOMKilled at the container level, look at executor overhead and <code>spark.executor.pyspark.memory</code> instead, keeping in mind the latter&#39;s <code>resource</code>-module limitations on macOS and its absence on Windows<sup class="footnote-ref"><a href="#fn4" id="fnref4:12">[4:12]</a></sup><sup class="footnote-ref"><a href="#fn6" id="fnref6:1">[6:1]</a></sup>.</p><h2 id="sources" tabindex="-1">Sources <a class="header-anchor" href="#sources" aria-label="Permalink to &quot;Sources&quot;">​</a></h2><hr class="footnotes-sep"><section class="footnotes"><ol class="footnotes-list"><li id="fn1" class="footnote-item"><p><em>Learning Spark, 2nd Edition</em>, Damji, Wenig, Das &amp; Lee, ch. 5 <a href="#fnref1" class="footnote-backref">↩︎</a> <a href="#fnref1:1" class="footnote-backref">↩︎</a> <a href="#fnref1:2" class="footnote-backref">↩︎</a> <a href="#fnref1:3" class="footnote-backref">↩︎</a></p></li><li id="fn2" class="footnote-item"><p><em>Spark: The Definitive Guide</em>, Chambers &amp; Zaharia, ch. 6 <a href="#fnref2" class="footnote-backref">↩︎</a> <a href="#fnref2:1" class="footnote-backref">↩︎</a> <a href="#fnref2:2" class="footnote-backref">↩︎</a> <a href="#fnref2:3" class="footnote-backref">↩︎</a> <a href="#fnref2:4" class="footnote-backref">↩︎</a> <a href="#fnref2:5" class="footnote-backref">↩︎</a></p></li><li id="fn3" class="footnote-item"><p><a href="https://spark.apache.org/docs/3.5.8/api/python/user_guide/sql/arrow_pandas.html" target="_blank" rel="noreferrer">Apache Arrow in PySpark</a> <a href="#fnref3" class="footnote-backref">↩︎</a> <a href="#fnref3:1" class="footnote-backref">↩︎</a> <a href="#fnref3:2" class="footnote-backref">↩︎</a> <a href="#fnref3:3" class="footnote-backref">↩︎</a> <a href="#fnref3:4" class="footnote-backref">↩︎</a> <a href="#fnref3:5" class="footnote-backref">↩︎</a> <a href="#fnref3:6" class="footnote-backref">↩︎</a> <a href="#fnref3:7" class="footnote-backref">↩︎</a> <a href="#fnref3:8" class="footnote-backref">↩︎</a> <a href="#fnref3:9" class="footnote-backref">↩︎</a> <a href="#fnref3:10" class="footnote-backref">↩︎</a> <a href="#fnref3:11" class="footnote-backref">↩︎</a> <a href="#fnref3:12" class="footnote-backref">↩︎</a> <a href="#fnref3:13" class="footnote-backref">↩︎</a></p></li><li id="fn4" class="footnote-item"><p><a href="https://spark.apache.org/docs/latest/configuration.html" target="_blank" rel="noreferrer">Configuration — Spark</a> <a href="#fnref4" class="footnote-backref">↩︎</a> <a href="#fnref4:1" class="footnote-backref">↩︎</a> <a href="#fnref4:2" class="footnote-backref">↩︎</a> <a href="#fnref4:3" class="footnote-backref">↩︎</a> <a href="#fnref4:4" class="footnote-backref">↩︎</a> <a href="#fnref4:5" class="footnote-backref">↩︎</a> <a href="#fnref4:6" class="footnote-backref">↩︎</a> <a href="#fnref4:7" class="footnote-backref">↩︎</a> <a href="#fnref4:8" class="footnote-backref">↩︎</a> <a href="#fnref4:9" class="footnote-backref">↩︎</a> <a href="#fnref4:10" class="footnote-backref">↩︎</a> <a href="#fnref4:11" class="footnote-backref">↩︎</a> <a href="#fnref4:12" class="footnote-backref">↩︎</a></p></li><li id="fn5" class="footnote-item"><p><a href="https://www.databricks.com/blog/2017/10/30/introducing-vectorized-udfs-for-pyspark.html" target="_blank" rel="noreferrer">Introducing Pandas UDF for PySpark</a> <a href="#fnref5" class="footnote-backref">↩︎</a> <a href="#fnref5:1" class="footnote-backref">↩︎</a> <a href="#fnref5:2" class="footnote-backref">↩︎</a> <a href="#fnref5:3" class="footnote-backref">↩︎</a> <a href="#fnref5:4" class="footnote-backref">↩︎</a> <a href="#fnref5:5" class="footnote-backref">↩︎</a> <a href="#fnref5:6" class="footnote-backref">↩︎</a> <a href="#fnref5:7" class="footnote-backref">↩︎</a> <a href="#fnref5:8" class="footnote-backref">↩︎</a></p></li><li id="fn6" class="footnote-item"><p><a href="https://luminousmen.com/post/dive-into-spark-memory" target="_blank" rel="noreferrer">Dive into Spark memory management</a> <a href="#fnref6" class="footnote-backref">↩︎</a> <a href="#fnref6:1" class="footnote-backref">↩︎</a></p></li></ol></section>`,33)])])}const g=a(i,[["render",f]]);export{k as __pageData,g as default};
@@ -0,0 +1 @@
1
+ import{_ as a,o,c as s,a5 as t}from"./chunks/framework.DSg0KOwT.js";const r="../assets/udf-execution-models.BUFDICuG.svg",n="../assets/udf-execution-models.dark.YTNS6GDq.svg",k=JSON.parse('{"title":"PySpark Specifics","description":"","frontmatter":{"title":"PySpark Specifics"},"headers":[],"relativePath":"tuning-reference/pyspark.md","filePath":"tuning-reference/pyspark.md"}'),i={name:"tuning-reference/pyspark.md"};function f(c,e,p,h,l,d){return o(),s("div",null,[...e[0]||(e[0]=[t("",33)])])}const g=a(i,[["render",f]]);export{k as __pageData,g as default};
@@ -0,0 +1 @@
1
+ import{_ as a,o,c as t,a5 as s}from"./chunks/framework.DSg0KOwT.js";const r="../assets/shuffle-map-reduce.KuOEZVmg.svg",n="../assets/shuffle-map-reduce.dark.BgQZnFSb.svg",g=JSON.parse('{"title":"Shuffle","description":"","frontmatter":{"title":"Shuffle"},"headers":[],"relativePath":"tuning-reference/shuffle.md","filePath":"tuning-reference/shuffle.md"}'),f={name:"tuning-reference/shuffle.md"};function i(c,e,l,d,h,u){return o(),t("div",null,[...e[0]||(e[0]=[s('<h1 id="shuffle" tabindex="-1">Shuffle <a class="header-anchor" href="#shuffle" aria-label="Permalink to &quot;Shuffle {#shuffle}&quot;">​</a></h1><h2 id="how-the-shuffle-works" tabindex="-1">How the shuffle works <a class="header-anchor" href="#how-the-shuffle-works" aria-label="Permalink to &quot;How the shuffle works&quot;">​</a></h2><p>A shuffle is Spark&#39;s mechanism for exchanging, sorting, grouping, and merging data across executors whenever an operation needs rows that share a key to land on the same partition. At the DataFrame/SQL level, the classic wide transformations that trigger one are <code>groupBy()</code>, <code>join()</code>, <code>agg()</code>, <code>sortBy()</code>, and <code>reduceByKey()</code>-style aggregations. Any join implementation that isn&#39;t a <a href="./joins.html">broadcast join</a> (shuffle hash join, shuffle sort-merge join, or shuffle-and-replicated nested loop / Cartesian product join) also requires a shuffle<sup class="footnote-ref"><a href="#fn1" id="fnref1">[1]</a></sup>. At the RDD layer underneath, the operations that can cause a shuffle are <a href="./partitioning.html">repartitioning</a> (<code>repartition</code>, <code>coalesce</code>), <code>&#39;ByKey</code> operations other than counting (<code>groupByKey</code>, <code>reduceByKey</code>), and join-family operations (<code>cogroup</code>, <code>join</code>)<sup class="footnote-ref"><a href="#fn2" id="fnref2">[2]</a></sup>.</p><p>Spark can skip the shuffle outright in a few well-defined cases where it already knows the data layout. Storage Partition Join (SPJ) avoids the shuffle phase entirely when Spark can use the partitioning already reported by a compatible V2 data source (<code>spark.sql.sources.v2.bucketing.enabled</code>, default <code>true</code> since 3.3.0), generalizing bucket joins to functions registered in a <code>FunctionCatalog</code><sup class="footnote-ref"><a href="#fn3" id="fnref3">[3]</a></sup>. Classic Hive-style bucketing gets the same effect: once both sides of a join are bucketed and sorted the same way, the physical plan shows no <code>Exchange</code> operator at all<sup class="footnote-ref"><a href="#fn4" id="fnref4">[4]</a></sup>. <code>DataFrameWriter.partitionBy</code>, by contrast, does not trigger a shuffle by itself on write, though it can still leave you with a large number of small output files<sup class="footnote-ref"><a href="#fn5" id="fnref5">[5]</a></sup>.</p><p>Modern Spark ships a single shuffle manager, <code>SortShuffleManager</code> (3.5 source). Incoming records are sorted by their target partition id and written to one map output file per task; reducers then fetch contiguous regions of that file for their share of the map output, spilling sorted subsets to disk and merging them if the data doesn&#39;t fit in memory<sup class="footnote-ref"><a href="#fn6" id="fnref6">[6]</a></sup>. Internally it has two write paths: a &quot;serialized sorting&quot; path used when there&#39;s no map-side combine, the serializer supports relocation of serialized values (Kryo or Spark SQL&#39;s custom serializers), and the shuffle produces at most 16,777,216 output partitions; and a &quot;deserialized sorting&quot; path used for everything else<sup class="footnote-ref"><a href="#fn6" id="fnref6:1">[6:1]</a></sup>. Framed against the older hash-based approach (where each map task produces a separate output file per reduce task rather than one sorted file), sort-based shuffle pays a sorting cost but scales better once shuffle data gets large, which is why Spark and Flink both moved to it and pair it with an external shuffle service<sup class="footnote-ref"><a href="#fn7" id="fnref7">[7]</a></sup>.</p><img class="light-only" src="'+r+'" alt="Map tasks sorting records by target partition id into one output file each while reduce tasks fetch their partition blocks from every map output."><img class="dark-only" src="'+n+'" alt="Map tasks sorting records by target partition id into one output file each while reduce tasks fetch their partition blocks from every map output."><p>Two pieces of supporting infrastructure sit alongside the core shuffle manager. The external shuffle service (<code>spark.shuffle.service.enabled</code>, default <code>false</code>) preserves the shuffle files written by executors so they can be safely removed, or so shuffle fetches continue even after an executor failure; enabling it requires standing up the service separately, since the flag alone isn&#39;t enough<sup class="footnote-ref"><a href="#fn8" id="fnref8">[8]</a></sup>. It&#39;s a long-running process on each cluster node, independent of any particular Spark application or its executors: once enabled, executors fetch shuffle files from this service instead of from each other, so shuffle state written by an executor keeps being served after that executor&#39;s own lifetime ends<sup class="footnote-ref"><a href="#fn9" id="fnref9">[9]</a></sup>. Push-based shuffle, designed under SPARK-30602 and based on LinkedIn&#39;s &quot;Magnet&quot; shuffle service, goes further: it changes the reduce side from a pull model (reducers requesting many small, randomly-ordered blocks from wherever each mapper wrote them) into a push-and-merge model, where mapper-generated blocks are pushed to remote Magnet shuffle services that opportunistically merge them into large per-partition chunks before the reduce stage starts<sup class="footnote-ref"><a href="#fn7" id="fnref7:1">[7:1]</a></sup>.</p><h2 id="reading-exchange-nodes" tabindex="-1">Reading Exchange nodes <a class="header-anchor" href="#reading-exchange-nodes" aria-label="Permalink to &quot;Reading Exchange nodes&quot;">​</a></h2><p>Whether an operation shuffles doesn&#39;t have to be guesswork: shuffles show up in the physical plan as <code>Exchange</code> nodes, and reading the plan is the most direct way to check. <code>groupBy()</code> on a DataFrame is a wide transformation and normally requires an <code>Exchange</code> so all rows for a key land on the same partition<sup class="footnote-ref"><a href="#fn1" id="fnref1:1">[1:1]</a></sup>; that node disappears only when Spark already knows the data is co-partitioned by the grouping key. The clearest documented case is a table bucketed and sorted on the join/group key, where the physical plan drops the <code>Exchange</code> because the data was already shuffled at write time<sup class="footnote-ref"><a href="#fn10" id="fnref10">[10]</a></sup>. SPJ produces the same absence of an <code>Exchange</code> for join queries against compatible V2 sources<sup class="footnote-ref"><a href="#fn3" id="fnref3:1">[3:1]</a></sup>.</p><p><code>DataFrame.repartition(key)</code> is worth watching for separately: it&#39;s a one-time, explicit shuffle, but if you cache the result, subsequent joins on that same key skip their own shuffle because the data is already partitioned that way<sup class="footnote-ref"><a href="#fn5" id="fnref5:1">[5:1]</a></sup>. Don&#39;t mistake <code>partitionBy</code> on write for the same thing: it changes the on-disk directory layout but is &quot;not the equivalent&quot; of an in-memory repartition, so <code>df.groupBy(&#39;key&#39;).sum()</code> over data written with <code>partitionBy</code> still requires an <code>Exchange</code><sup class="footnote-ref"><a href="#fn5" id="fnref5:2">[5:2]</a></sup>.</p><p>Deduplication is a case where there&#39;s no avoiding the shuffle regardless of how you write it: Spark SQL&#39;s <code>dropDuplicates</code> selects unique rows the same way RDD-level <code>distinct</code> does, and both &quot;can require a shuffle&quot;<sup class="footnote-ref"><a href="#fn11" id="fnref11">[11]</a></sup>. The only documented difference between them is key flexibility, not shuffle volume: unlike RDD <code>distinct</code>, DataFrame <code>dropDuplicates</code> can optionally drop rows based on only a subset of columns (e.g., <code>dropDuplicates(List(&quot;id&quot;))</code>) rather than requiring uniqueness across the whole row<sup class="footnote-ref"><a href="#fn11" id="fnref11:1">[11:1]</a></sup>.</p><h2 id="where-the-cost-accumulates" tabindex="-1">Where the cost accumulates <a class="header-anchor" href="#where-the-cost-accumulates" aria-label="Permalink to &quot;Where the cost accumulates&quot;">​</a></h2><p>Confirming a shuffle happened is the easy part; pricing it out is the harder one. Shuffle cost shows up in several places at once, and Spark&#39;s shuffle-tuning knobs largely exist to manage it. Inside <code>SortShuffleManager</code>, when there are fewer than <code>spark.shuffle.sort.bypassMergeThreshold</code> reduce partitions (default 200)<sup class="footnote-ref"><a href="#fn8" id="fnref8:1">[8:1]</a></sup> and no map-side aggregation is needed, Spark takes a bypass path: it writes <code>numPartitions</code> files directly per map task and concatenates them at the end, rather than merge-sorting spilled files. This avoids serializing and deserializing the data twice, which the normal merge-sort path would otherwise pay; the trade-off is having multiple files open at once, and more memory allocated to buffers<sup class="footnote-ref"><a href="#fn6" id="fnref6:2">[6:2]</a></sup>. That trade stays cheap only while the reduce-side partition count is small.</p><p>Reducers also pay a fixed memory cost for every map output fetched in parallel, since each needs a buffer to receive it; that&#39;s what <code>spark.reducer.maxSizeInFlight</code> (default 48m) caps<sup class="footnote-ref"><a href="#fn8" id="fnref8:2">[8:2]</a></sup>. And map tasks pay an I/O cost writing shuffle files: a larger <code>spark.shuffle.file.buffer</code> (default 32k) reduces the number of disk seeks and system calls needed to create those intermediate files<sup class="footnote-ref"><a href="#fn8" id="fnref8:3">[8:3]</a></sup>.</p><p>External shuffle infrastructure matters most once <a href="./cluster-config.html">dynamic allocation</a> is in play: without the external shuffle service, an executor decommissioned mid-shuffle (say, because of stragglers) would force its shuffle files to be recomputed from scratch once it&#39;s removed<sup class="footnote-ref"><a href="#fn9" id="fnref9:1">[9:1]</a></sup>. Push-based shuffle exists because, as shuffle data volume grows, the number of blocks a classic shuffle produces grows quadratically (mappers × reducers) while individual block sizes shrink to only tens of KB, which is inefficient for HDD-backed random reads<sup class="footnote-ref"><a href="#fn7" id="fnref7:2">[7:2]</a></sup>. Magnet&#39;s push-and-merge model improves disk I/O efficiency by converting many small random reads into large sequential reads of pre-merged chunks, and improves reducer data locality since merged output can be co-located with the reduce tasks that consume it; the push itself is decoupled from the mappers, so it doesn&#39;t add to map task runtime or fail map tasks if a push fails, and it&#39;s best-effort, so reducers can fetch a mix of merged and unmerged blocks<sup class="footnote-ref"><a href="#fn7" id="fnref7:3">[7:3]</a></sup>. LinkedIn&#39;s own production rollout, on a cluster running 30K+ Spark applications shuffling roughly 5 PB/day, reported a large reduction in shuffle read wait time and a corresponding drop in total executor runtime after enabling it on complex production flows<sup class="footnote-ref"><a href="#fn7" id="fnref7:4">[7:4]</a></sup>.</p><h2 id="tuning-or-avoiding-the-shuffle" tabindex="-1">Tuning or avoiding the shuffle <a class="header-anchor" href="#tuning-or-avoiding-the-shuffle" aria-label="Permalink to &quot;Tuning or avoiding the shuffle&quot;">​</a></h2><p>The levers split into two groups: avoiding a shuffle, and tuning the one you can&#39;t avoid. Where possible, avoid the shuffle rather than tune it: bucket and sort tables on the join/group key so the physical plan drops the <code>Exchange</code> entirely<sup class="footnote-ref"><a href="#fn10" id="fnref10:1">[10:1]</a></sup><sup class="footnote-ref"><a href="#fn4" id="fnref4:1">[4:1]</a></sup>, or lean on Storage Partition Join for compatible V2 sources<sup class="footnote-ref"><a href="#fn3" id="fnref3:2">[3:2]</a></sup>. If you need to reuse a partitioning scheme across several joins, <code>repartition(key)</code> once and cache the result rather than relying on <code>partitionBy</code>, which only affects the on-disk layout and doesn&#39;t exempt later aggregations from their own <code>Exchange</code><sup class="footnote-ref"><a href="#fn5" id="fnref5:3">[5:3]</a></sup>.</p><blockquote><p><strong>PySpark:</strong> <code>dropDuplicates(subset)</code> still shuffles like <code>distinct()</code>, but lets you dedupe on a subset of columns instead of the whole row; use <code>df.dropDuplicates([&quot;id&quot;])</code> when uniqueness only needs to hold on a few key columns<sup class="footnote-ref"><a href="#fn11" id="fnref11:2">[11:2]</a></sup>.</p></blockquote><p>For shuffles you can&#39;t avoid, the main tuning levers are:</p><ul><li><code>spark.shuffle.sort.bypassMergeThreshold</code> (default 200): raise or lower the reduce-partition-count cutoff for the bypass (hash-style) write path described above<sup class="footnote-ref"><a href="#fn8" id="fnref8:4">[8:4]</a></sup>.</li><li><code>spark.shuffle.compress</code> (default <code>true</code>): compresses map output files using whatever codec <code>spark.io.compression.codec</code> names (default <code>lz4</code>; <code>lzf</code>, <code>snappy</code>, and <code>zstd</code> are also available)<sup class="footnote-ref"><a href="#fn8" id="fnref8:5">[8:5]</a></sup>. <code>spark.shuffle.spill.compress</code> (also default <code>true</code>) separately governs compression of spilled shuffle data, using the same codec setting<sup class="footnote-ref"><a href="#fn8" id="fnref8:6">[8:6]</a></sup>.</li><li><code>spark.reducer.maxSizeInFlight</code> (default 48m): raising it lets more data move in flight at the cost of per-task memory; lowering it saves memory at the cost of more fetch rounds<sup class="footnote-ref"><a href="#fn8" id="fnref8:7">[8:7]</a></sup>.</li><li><code>spark.shuffle.file.buffer</code> (default 32k): Learning Spark&#39;s tuning guidance recommends bumping it to 1 MB for large jobs, trading some per-task memory for fewer disk seeks and syscalls during the shuffle-write phase<sup class="footnote-ref"><a href="#fn1" id="fnref1:2">[1:2]</a></sup>.</li><li><code>spark.shuffle.service.enabled</code> (default <code>false</code>): turn on the external shuffle service so executors can be safely removed under dynamic allocation without losing shuffle state. Dynamic allocation&#39;s documentation lists shuffle tracking (<code>spark.dynamicAllocation.shuffleTracking.enabled</code>), shuffle-block decommissioning, and a custom <code>ShuffleDataIO</code> plugin backed by reliable storage as alternatives to enabling the service outright<sup class="footnote-ref"><a href="#fn8" id="fnref8:8">[8:8]</a></sup>.</li><li>Push-based shuffle: enable it with the paired server/client flags added in Spark 3.2.0: <code>spark.shuffle.push.server.mergedShuffleFileManagerImpl</code> on the server side, and <code>spark.shuffle.push.enabled=true</code> on the client side (both disabled by default; the client flag only takes effect together with the server-side one). It&#39;s currently only supported for Spark on YARN with the external shuffle service enabled<sup class="footnote-ref"><a href="#fn8" id="fnref8:9">[8:9]</a></sup>.</li></ul><h2 id="sources" tabindex="-1">Sources <a class="header-anchor" href="#sources" aria-label="Permalink to &quot;Sources&quot;">​</a></h2><hr class="footnotes-sep"><section class="footnotes"><ol class="footnotes-list"><li id="fn1" class="footnote-item"><p><em>Learning Spark, 2nd Edition</em>, Damji, Wenig, Das, Lee, ch. 7: Optimizing and Tuning Spark Applications <a href="#fnref1" class="footnote-backref">↩︎</a> <a href="#fnref1:1" class="footnote-backref">↩︎</a> <a href="#fnref1:2" class="footnote-backref">↩︎</a></p></li><li id="fn2" class="footnote-item"><p><a href="https://spark.apache.org/docs/latest/rdd-programming-guide.html" target="_blank" rel="noreferrer">RDD Programming Guide</a> <a href="#fnref2" class="footnote-backref">↩︎</a></p></li><li id="fn3" class="footnote-item"><p><a href="https://spark.apache.org/docs/latest/sql-performance-tuning.html" target="_blank" rel="noreferrer">Performance Tuning: Spark SQL, DataFrames and Datasets Guide</a> <a href="#fnref3" class="footnote-backref">↩︎</a> <a href="#fnref3:1" class="footnote-backref">↩︎</a> <a href="#fnref3:2" class="footnote-backref">↩︎</a></p></li><li id="fn4" class="footnote-item"><p><a href="https://books.japila.pl/spark-sql-internals/bucketing/" target="_blank" rel="noreferrer">Bucketing (The Internals of Spark SQL)</a> <a href="#fnref4" class="footnote-backref">↩︎</a> <a href="#fnref4:1" class="footnote-backref">↩︎</a></p></li><li id="fn5" class="footnote-item"><p><a href="https://luminousmen.com/post/spark-tips-partition-tuning" target="_blank" rel="noreferrer">Spark Tips: Partition Tuning</a> <a href="#fnref5" class="footnote-backref">↩︎</a> <a href="#fnref5:1" class="footnote-backref">↩︎</a> <a href="#fnref5:2" class="footnote-backref">↩︎</a> <a href="#fnref5:3" class="footnote-backref">↩︎</a></p></li><li id="fn6" class="footnote-item"><p><a href="https://raw.githubusercontent.com/apache/spark/v3.5.0/core/src/main/scala/org/apache/spark/shuffle/sort/SortShuffleManager.scala" target="_blank" rel="noreferrer">SortShuffleManager.scala</a> <a href="#fnref6" class="footnote-backref">↩︎</a> <a href="#fnref6:1" class="footnote-backref">↩︎</a> <a href="#fnref6:2" class="footnote-backref">↩︎</a></p></li><li id="fn7" class="footnote-item"><p><a href="https://issues.apache.org/jira/browse/SPARK-30602" target="_blank" rel="noreferrer">SPARK-30602: Support push-based shuffle to improve shuffle efficiency (Magnet)</a> <a href="#fnref7" class="footnote-backref">↩︎</a> <a href="#fnref7:1" class="footnote-backref">↩︎</a> <a href="#fnref7:2" class="footnote-backref">↩︎</a> <a href="#fnref7:3" class="footnote-backref">↩︎</a> <a href="#fnref7:4" class="footnote-backref">↩︎</a></p></li><li id="fn8" class="footnote-item"><p><a href="https://spark.apache.org/docs/latest/configuration.html" target="_blank" rel="noreferrer">Configuration (Spark)</a> <a href="#fnref8" class="footnote-backref">↩︎</a> <a href="#fnref8:1" class="footnote-backref">↩︎</a> <a href="#fnref8:2" class="footnote-backref">↩︎</a> <a href="#fnref8:3" class="footnote-backref">↩︎</a> <a href="#fnref8:4" class="footnote-backref">↩︎</a> <a href="#fnref8:5" class="footnote-backref">↩︎</a> <a href="#fnref8:6" class="footnote-backref">↩︎</a> <a href="#fnref8:7" class="footnote-backref">↩︎</a> <a href="#fnref8:8" class="footnote-backref">↩︎</a> <a href="#fnref8:9" class="footnote-backref">↩︎</a></p></li><li id="fn9" class="footnote-item"><p><a href="https://spark.apache.org/docs/latest/job-scheduling.html" target="_blank" rel="noreferrer">Job Scheduling</a> <a href="#fnref9" class="footnote-backref">↩︎</a> <a href="#fnref9:1" class="footnote-backref">↩︎</a></p></li><li id="fn10" class="footnote-item"><p><a href="https://www.taboola.com/engineering/bucket-the-shuffle-out-of-here/" target="_blank" rel="noreferrer">Bucket the Shuffle Out of Here (Taboola Engineering)</a> <a href="#fnref10" class="footnote-backref">↩︎</a> <a href="#fnref10:1" class="footnote-backref">↩︎</a></p></li><li id="fn11" class="footnote-item"><p><em>High Performance Spark, 2nd Edition</em>, Karau, Polak &amp; Warren, ch. 5: DataFrames, Datasets, and Spark SQL <a href="#fnref11" class="footnote-backref">↩︎</a> <a href="#fnref11:1" class="footnote-backref">↩︎</a> <a href="#fnref11:2" class="footnote-backref">↩︎</a></p></li></ol></section>',24)])])}const m=a(f,[["render",i]]);export{g as __pageData,m as default};
@@ -0,0 +1 @@
1
+ import{_ as a,o,c as t,a5 as s}from"./chunks/framework.DSg0KOwT.js";const r="../assets/shuffle-map-reduce.KuOEZVmg.svg",n="../assets/shuffle-map-reduce.dark.BgQZnFSb.svg",g=JSON.parse('{"title":"Shuffle","description":"","frontmatter":{"title":"Shuffle"},"headers":[],"relativePath":"tuning-reference/shuffle.md","filePath":"tuning-reference/shuffle.md"}'),f={name:"tuning-reference/shuffle.md"};function i(c,e,l,d,h,u){return o(),t("div",null,[...e[0]||(e[0]=[s("",24)])])}const m=a(f,[["render",i]]);export{g as __pageData,m as default};
@@ -0,0 +1 @@
1
+ import{_ as a,o as t,c as s,a5 as o}from"./chunks/framework.DSg0KOwT.js";const r="../assets/driver-executor.D5pQ7YN1.svg",n="../assets/driver-executor.dark.BmX9cPvh.svg",i="../assets/dag-stages.DSz_S937.svg",f="../assets/dag-stages.dark.F72UzxH4.svg",k=JSON.parse('{"title":"Spark Execution Model","description":"","frontmatter":{"title":"Spark Execution Model"},"headers":[],"relativePath":"tuning-reference/spark-architecture.md","filePath":"tuning-reference/spark-architecture.md"}'),c={name:"tuning-reference/spark-architecture.md"};function l(h,e,d,p,u,g){return t(),s("div",null,[...e[0]||(e[0]=[o('<h1 id="spark-architecture" tabindex="-1">Spark Execution Model <a class="header-anchor" href="#spark-architecture" aria-label="Permalink to &quot;Spark Execution Model {#spark-architecture}&quot;">​</a></h1><h2 id="the-driver-and-the-dag" tabindex="-1">The driver and the DAG <a class="header-anchor" href="#the-driver-and-the-dag" aria-label="Permalink to &quot;The driver and the DAG&quot;">​</a></h2><p>The Driver is the process &quot;in the driver seat&quot; of a Spark application: it controls execution and maintains all state of the cluster, including the state and tasks of the executors, and it interfaces with the cluster manager to obtain physical resources and launch executors<sup class="footnote-ref"><a href="#fn1" id="fnref1">[1]</a></sup>. It runs as a single JVM process (on the submission machine in client mode, or on a dedicated cluster node in cluster mode), and if it dies, the application dies with it<sup class="footnote-ref"><a href="#fn2" id="fnref2">[2]</a></sup>.</p><p>As application code executes, the Driver incrementally builds a logical DAG of RDD/DataFrame transformations. Because Spark is lazy, nothing actually computes until an action is called; only then does the Driver convert the logical DAG into a physical plan, cut it into stages at wide-dependency (<a href="./shuffle.html">shuffle</a>) boundaries, break each stage into one task per partition, and take on the scheduler&#39;s job: talking to the cluster resource manager to request executors, handing out tasks, tracking their progress, and resubmitting tasks whose executor died<sup class="footnote-ref"><a href="#fn2" id="fnref2:1">[2:1]</a></sup>.</p><p>Three components live only on the Driver:</p><ul><li>The <strong>DAGScheduler</strong> reads RDD lineage, decides stage boundaries at wide dependencies, emits per-partition TaskSets, tracks lineage so lost partitions can be recomputed, and reacts dynamically to stage completions rather than pre-scheduling the whole DAG up front; it also declares a job failed if a stage can&#39;t make progress<sup class="footnote-ref"><a href="#fn2" id="fnref2:2">[2:2]</a></sup>.</li><li>The <strong>BlockManagerMaster</strong> is the Driver-side counterpart to each executor&#39;s own BlockManager. It keeps the global map of block locations (RDD partitions, shuffle files, broadcast variables) across the whole cluster, so a task&#39;s local BlockManager knows where to fetch a remote block from<sup class="footnote-ref"><a href="#fn2" id="fnref2:3">[2:3]</a></sup>.</li><li>The <strong>SparkContext</strong> (wrapped today by SparkSession) is the entry point that lets the Driver talk to the cluster manager, request executors, create RDDs, and manage shared variables. It&#39;s instantiated once per application and lives for the application&#39;s whole lifetime<sup class="footnote-ref"><a href="#fn2" id="fnref2:4">[2:4]</a></sup>.</li></ul><p>Executors, by contrast, hold no cluster-wide state: they only run the tasks assigned to them and report back success, failure, and results<sup class="footnote-ref"><a href="#fn1" id="fnref1:1">[1:1]</a></sup>. The split even shows up at the configuration level: <code>spark.driver.host</code>/<code>spark.driver.port</code> and <code>spark.driver.blockManager.port</code> are Driver-specific listening endpoints, distinct from the equivalent executor settings<sup class="footnote-ref"><a href="#fn3" id="fnref3">[3]</a></sup>.</p><img class="light-only" src="'+r+'" alt="A Spark driver with its scheduler components dispatching tasks to executors through the cluster manager and receiving status and results back."><img class="dark-only" src="'+n+'" alt="A Spark driver with its scheduler components dispatching tasks to executors through the cluster manager and receiving status and results back."><p>A Spark job corresponds to one action, and each job breaks down into a series of stages: how many depends on how many shuffle operations need to happen<sup class="footnote-ref"><a href="#fn1" id="fnref1:2">[1:2]</a></sup>. A stage is a group of tasks that can execute together to compute the same operation across machines: work one executor can do without communicating with other executors or the Driver. A new stage begins whenever data has to move across the network, that is, at a shuffle<sup class="footnote-ref"><a href="#fn4" id="fnref4">[4]</a></sup>. Wide transformations, such as <code>groupByKey</code>, <code>join</code>, and <code>sortByKey</code>, create <code>ShuffleDependency</code> objects, and it&#39;s these <code>ShuffleDependency</code>s that mark the stage boundary; several narrow transformations can be grouped into the same stage<sup class="footnote-ref"><a href="#fn4" id="fnref4:1">[4:1]</a></sup>. Within a stage, the number of tasks equals the number of partitions in that stage&#39;s output RDD: one task per partition, each running the same code on a different slice of data<sup class="footnote-ref"><a href="#fn4" id="fnref4:2">[4:2]</a></sup>.</p><p>The rule is always &quot;a shuffle dependency creates the boundary,&quot; and that isn&#39;t limited to an explicit <code>.repartition()</code> or <code>.groupBy()</code> call: <code>sortByKey</code>/<code>sortBy</code> on an RDD is itself a wide transformation, not just <code>groupByKey</code>-style aggregations<sup class="footnote-ref"><a href="#fn4" id="fnref4:3">[4:3]</a></sup>. The original Spark paper defines stage boundaries generically, as the shuffle operations required for wide dependencies, or already-computed partitions that can short-circuit the computation of a parent RDD<sup class="footnote-ref"><a href="#fn5" id="fnref5">[5]</a></sup>. A wide dependency of any kind is what matters, not a specific API call.</p><p>That wide/narrow split has a formal definition behind it. Conceptually, a narrow transformation is one where each partition in the child RDD has simple, finite dependencies on partitions in the parent RDD, determinable at design time regardless of the values of the records; narrow dependencies allow pipelined, one-node execution, while wide dependencies require data from all parent partitions to be shuffled across nodes<sup class="footnote-ref"><a href="#fn5" id="fnref5:1">[5:1]</a></sup>. The formal 2012 definition is stated from the parent&#39;s side rather than the child&#39;s: a transformation is narrow if &quot;each partition of the parent RDD is used by at most one partition of the child RDD,&quot; and wide if &quot;multiple child partitions may depend on&quot; a given parent partition<sup class="footnote-ref"><a href="#fn4" id="fnref4:4">[4:4]</a></sup>. That parent-centric framing is more precise, because Spark&#39;s DAG scheduler builds the execution plan backward from the action to the input RDD, and it correctly rules out the case of one parent partition feeding multiple children under &quot;narrow&quot;<sup class="footnote-ref"><a href="#fn4" id="fnref4:5">[4:5]</a></sup>.</p><p><code>mapPartitions</code> is narrow under this definition: each child partition depends on exactly one parent partition, the same as <code>map</code> and <code>filter</code><sup class="footnote-ref"><a href="#fn4" id="fnref4:6">[4:6]</a></sup>. <code>coalesce</code> is narrow too, even though it changes the number of partitions and a child partition can depend on multiple parent partitions: the rule only requires that each parent partition be used by at most one child partition, and which parent partitions merge into which child is fixed at design time, independent of the data&#39;s values. That holds specifically when <code>coalesce</code> reduces the partition count; when it increases the count, it behaves like <code>repartition</code>, a shuffle<sup class="footnote-ref"><a href="#fn4" id="fnref4:7">[4:7]</a></sup>.</p><p>Pipelining is what makes narrow dependencies pay off. Spark performs as many steps as possible in one pass before writing data to memory or disk: any sequence of operations that feed data directly into each other, without moving data across nodes, collapses into a single stage of tasks that execute all the operations together. <code>map → filter → map</code> becomes one stage whose tasks read each record and pass it through all three operations in sequence, rather than materializing intermediate results after each step; the same collapsing happens for a DataFrame/SQL computation doing <code>select → filter → select</code><sup class="footnote-ref"><a href="#fn1" id="fnref1:3">[1:3]</a></sup>. The original paper phrases the mechanism at the scheduler level: to run an action, Spark builds stages at wide dependencies and pipelines narrow transformations inside each stage<sup class="footnote-ref"><a href="#fn5" id="fnref5:2">[5:2]</a></sup>.</p><p>Two more components take over once a job&#39;s stages exist. The <strong>DAGScheduler</strong> is Spark&#39;s high-level scheduling layer: it takes RDD dependencies, builds a DAG of stages for each job, determines where each task should run, and passes that to the TaskScheduler<sup class="footnote-ref"><a href="#fn4" id="fnref4:8">[4:8]</a></sup>. The <strong>TaskScheduler</strong> takes it from there and stops thinking in terms of RDDs, lineage, or shuffles: only machines and slots. It works with the SchedulerBackend to request executors, assigns tasks to executors while respecting data-locality preferences, and retries failed tasks, including resubmitting a task elsewhere if its executor died<sup class="footnote-ref"><a href="#fn2" id="fnref2:5">[2:5]</a></sup>. The TaskScheduler doesn&#39;t launch tasks directly; it hands them to the SchedulerBackend, the layer that actually talks to the cluster manager to request, launch, and kill executors<sup class="footnote-ref"><a href="#fn2" id="fnref2:6">[2:6]</a></sup>.</p><p>Finally, at the RDD API level <code>spark.default.parallelism</code> sets the default partition count for distributed shuffle operations like <code>reduceByKey</code> and <code>join</code>. It defaults to the largest number of partitions in the parent RDD, and for operations with no parent RDD, such as <code>parallelize</code>, the default depends on the cluster manager<sup class="footnote-ref"><a href="#fn3" id="fnref3:1">[3:1]</a></sup>. <code>spark.sql.shuffle.partitions</code> is the equivalent knob for the Spark SQL/DataFrame engine: it configures how many partitions Spark uses when shuffling data for joins or aggregations, and it defaults to 200 regardless of data size<sup class="footnote-ref"><a href="#fn6" id="fnref6">[6]</a></sup>, applying whenever a DataFrame/SQL operation triggers a shuffle: join, groupBy, distinct, orderBy, and so on<sup class="footnote-ref"><a href="#fn7" id="fnref7">[7]</a></sup>.</p><h2 id="reading-stages-and-tasks" tabindex="-1">Reading stages and tasks <a class="header-anchor" href="#reading-stages-and-tasks" aria-label="Permalink to &quot;Reading stages and tasks&quot;">​</a></h2><p>Once a job runs, stage and task counts show up directly in the Spark UI, and a worked example makes the mechanics concrete: a job that reads a range with 8 partitions produces a stage with 8 tasks; repartitioning to 6 and then 5 partitions produces stages with 6 and 5 tasks; a subsequent join shuffles into the default 200 shuffle partitions, producing a 200-task stage<sup class="footnote-ref"><a href="#fn1" id="fnref1:4">[1:4]</a></sup>. The same logic explains stage counts for RDD pipelines: the chain <code>filter → map → groupByKey → map → sortByKey → count</code> produces three stages, bounded by the <code>groupByKey</code> and <code>sortByKey</code> operations, since both are wide<sup class="footnote-ref"><a href="#fn4" id="fnref4:9">[4:9]</a></sup>.</p><img class="light-only" src="'+i+'" alt="A three stage DAG for the pipeline filter, map, groupByKey, map, sortByKey, count, with stage boundaries at the wide groupByKey and sortByKey operations."><img class="dark-only" src="'+f+'" alt="A three stage DAG for the pipeline filter, map, groupByKey, map, sortByKey, count, with stage boundaries at the wide groupByKey and sortByKey operations."><p>Pipelining is invisible to the application and only shows up in the Spark UI or logs, where multiple chained narrow operations appear collapsed into a single stage instead of one stage per operation<sup class="footnote-ref"><a href="#fn1" id="fnref1:5">[1:5]</a></sup>. A related signal is the &quot;skipped stage&quot; marking: because Spark always writes shuffle output to stable storage regardless of any <code>persist</code>/<code>checkpoint</code> call, if the Driver reuses an RDD that was already shuffled, Spark can skip recomputing everything up to that shuffle and read the shuffle files directly. The Spark UI shows this as a skipped stage, not as an added boundary<sup class="footnote-ref"><a href="#fn4" id="fnref4:10">[4:10]</a></sup>.</p><p>When diagnosing partition counts, checking which knob is in play matters: <code>spark.default.parallelism</code> governs the legacy RDD API&#39;s shuffle and <code>parallelize</code> operations, while <code>spark.sql.shuffle.partitions</code> governs DataFrame/Dataset/SQL shuffles<sup class="footnote-ref"><a href="#fn3" id="fnref3:2">[3:2]</a></sup>. Since Spark 3.0, <a href="./aqe.html">Adaptive Query Execution</a> (AQE) can also override the static <code>spark.sql.shuffle.partitions</code> value after the fact. It can coalesce many small post-shuffle partitions and split skewed ones at runtime, but only after the first shuffle has already happened, so it never explains the initial input partitioning<sup class="footnote-ref"><a href="#fn7" id="fnref7:1">[7:1]</a></sup>.</p><h2 id="why-the-boundary-matters" tabindex="-1">Why the boundary matters <a class="header-anchor" href="#why-the-boundary-matters" aria-label="Permalink to &quot;Why the boundary matters&quot;">​</a></h2><p>Those stage boundaries aren&#39;t just a UI detail. Because a new stage begins only at a shuffle, the boundary is exactly where Spark pays the cost of moving data across the network. Everything inside a stage runs without that cost, pipelined on a single executor<sup class="footnote-ref"><a href="#fn4" id="fnref4:11">[4:11]</a></sup>.</p><p>That&#39;s also why the wide/narrow distinction determines whether pipelining is even possible: once a shuffle is required, downstream computation can&#39;t proceed until the shuffle completes, because the records landing on each partition may change as a result of it, so narrow transformations following a wide one belong to a new stage and can&#39;t be pipelined across that boundary<sup class="footnote-ref"><a href="#fn4" id="fnref4:12">[4:12]</a></sup>.</p><p>Whether pipelining happens can also depend on partitioning state Spark already knows about. Because Spark tracks how an RDD is already partitioned, the same operation can land in different stage boundaries depending on whether the input RDD already has a known partitioner: no shuffle, and hence no new stage, is needed if one is already in place<sup class="footnote-ref"><a href="#fn4" id="fnref4:13">[4:13]</a></sup>.</p><p><a href="./caching.html">Checkpointing</a> is a separate mechanism from all of this: it breaks RDD lineage and writes data to disk so later transformations start from a fresh, trimmed plan, rather than being pipelined against the original computation<sup class="footnote-ref"><a href="#fn8" id="fnref8">[8]</a></sup>.</p><p>The choice of <a href="./partitioning.html">shuffle-partition count</a> matters for the same reason: there&#39;s no fixed formula for the right number, since it depends on data set size, number of cores, and available executor memory, and the default of 200 is called out as too high for smaller or streaming workloads, where it&#39;s better reduced toward the number of executor cores or less<sup class="footnote-ref"><a href="#fn9" id="fnref9">[9]</a></sup>. <code>coalesce</code>&#39;s narrowness carries its own tradeoff: reducing partition count without a shuffle forces all upstream partitions in that stage to run at the coalesced parallelism level, which can be undesirable<sup class="footnote-ref"><a href="#fn4" id="fnref4:14">[4:14]</a></sup>.</p><p>On the scheduling side, Spark&#39;s scheduler is fully thread-safe and supports running multiple jobs concurrently if they&#39;re submitted from separate Driver threads, common when an application serves multiple concurrent requests over the network<sup class="footnote-ref"><a href="#fn10" id="fnref10">[10]</a></sup>. By default Spark schedules jobs FIFO, which means an application running many jobs from multiple threads can have later jobs wait behind earlier ones unless a fairer policy is configured<sup class="footnote-ref"><a href="#fn4" id="fnref4:15">[4:15]</a></sup>.</p><h2 id="controlling-parallelism-and-scheduling" tabindex="-1">Controlling parallelism and scheduling <a class="header-anchor" href="#controlling-parallelism-and-scheduling" aria-label="Permalink to &quot;Controlling parallelism and scheduling&quot;">​</a></h2><p>Parallelism and scheduling are both tunable from here. When <code>coalesce</code>&#39;s single-parent-per-child constraint forces upstream parallelism lower than you want, <code>repartition</code> trades that limitation for an explicit shuffle, restoring full control over the resulting partition count<sup class="footnote-ref"><a href="#fn4" id="fnref4:16">[4:16]</a></sup>. For the SQL/DataFrame engine, don&#39;t leave <code>spark.sql.shuffle.partitions</code> at its 200 default for small or streaming workloads: reduce it toward the number of executor cores or less, since the right value depends on data size, core count, and executor memory rather than a fixed rule<sup class="footnote-ref"><a href="#fn9" id="fnref9:1">[9:1]</a></sup>. Where the data volume is unpredictable, Adaptive Query Execution (available since Spark 3.0) can take over after the fact: it coalesces many small post-shuffle partitions and splits skewed ones at runtime, though it only acts after the first shuffle has already happened and won&#39;t fix the initial input partitioning<sup class="footnote-ref"><a href="#fn7" id="fnref7:2">[7:2]</a></sup>.</p><p>For concurrent workloads, Spark also offers a fair scheduler as an alternative to the FIFO default, assigning tasks to concurrent jobs round-robin so each job gets a more even share of cluster resources<sup class="footnote-ref"><a href="#fn4" id="fnref4:17">[4:17]</a></sup>; job pools and weights for finer-grained sharing are configured via <code>spark.scheduler.mode=FAIR</code> and the <code>spark.scheduler.pool</code> local property set on the submitting thread<sup class="footnote-ref"><a href="#fn1" id="fnref1:6">[1:6]</a></sup>. Submitting jobs from separate Driver threads is what lets them run concurrently in the first place, rather than queuing behind each other<sup class="footnote-ref"><a href="#fn10" id="fnref10:1">[10:1]</a></sup>.</p><h2 id="sources" tabindex="-1">Sources <a class="header-anchor" href="#sources" aria-label="Permalink to &quot;Sources&quot;">​</a></h2><hr class="footnotes-sep"><section class="footnotes"><ol class="footnotes-list"><li id="fn1" class="footnote-item"><p><em>Spark: The Definitive Guide</em>, Chambers &amp; Zaharia, ch. 15–16 <a href="#fnref1" class="footnote-backref">↩︎</a> <a href="#fnref1:1" class="footnote-backref">↩︎</a> <a href="#fnref1:2" class="footnote-backref">↩︎</a> <a href="#fnref1:3" class="footnote-backref">↩︎</a> <a href="#fnref1:4" class="footnote-backref">↩︎</a> <a href="#fnref1:5" class="footnote-backref">↩︎</a> <a href="#fnref1:6" class="footnote-backref">↩︎</a></p></li><li id="fn2" class="footnote-item"><p><a href="https://luminousmen.com/post/spark-anatomy-of-spark-application" target="_blank" rel="noreferrer">Anatomy of Spark Application</a> <a href="#fnref2" class="footnote-backref">↩︎</a> <a href="#fnref2:1" class="footnote-backref">↩︎</a> <a href="#fnref2:2" class="footnote-backref">↩︎</a> <a href="#fnref2:3" class="footnote-backref">↩︎</a> <a href="#fnref2:4" class="footnote-backref">↩︎</a> <a href="#fnref2:5" class="footnote-backref">↩︎</a> <a href="#fnref2:6" class="footnote-backref">↩︎</a></p></li><li id="fn3" class="footnote-item"><p><a href="https://spark.apache.org/docs/latest/configuration.html" target="_blank" rel="noreferrer">Configuration — Spark</a> <a href="#fnref3" class="footnote-backref">↩︎</a> <a href="#fnref3:1" class="footnote-backref">↩︎</a> <a href="#fnref3:2" class="footnote-backref">↩︎</a></p></li><li id="fn4" class="footnote-item"><p><em>High Performance Spark, 2nd Edition</em>, Karau, Polak &amp; Warren, ch. 2, 7–8 <a href="#fnref4" class="footnote-backref">↩︎</a> <a href="#fnref4:1" class="footnote-backref">↩︎</a> <a href="#fnref4:2" class="footnote-backref">↩︎</a> <a href="#fnref4:3" class="footnote-backref">↩︎</a> <a href="#fnref4:4" class="footnote-backref">↩︎</a> <a href="#fnref4:5" class="footnote-backref">↩︎</a> <a href="#fnref4:6" class="footnote-backref">↩︎</a> <a href="#fnref4:7" class="footnote-backref">↩︎</a> <a href="#fnref4:8" class="footnote-backref">↩︎</a> <a href="#fnref4:9" class="footnote-backref">↩︎</a> <a href="#fnref4:10" class="footnote-backref">↩︎</a> <a href="#fnref4:11" class="footnote-backref">↩︎</a> <a href="#fnref4:12" class="footnote-backref">↩︎</a> <a href="#fnref4:13" class="footnote-backref">↩︎</a> <a href="#fnref4:14" class="footnote-backref">↩︎</a> <a href="#fnref4:15" class="footnote-backref">↩︎</a> <a href="#fnref4:16" class="footnote-backref">↩︎</a> <a href="#fnref4:17" class="footnote-backref">↩︎</a></p></li><li id="fn5" class="footnote-item"><p><a href="https://www.usenix.org/system/files/conference/nsdi12/nsdi12-final138.pdf" target="_blank" rel="noreferrer">Resilient Distributed Datasets: A Fault-Tolerant Abstraction for In-Memory Cluster Computing</a> <a href="#fnref5" class="footnote-backref">↩︎</a> <a href="#fnref5:1" class="footnote-backref">↩︎</a> <a href="#fnref5:2" class="footnote-backref">↩︎</a></p></li><li id="fn6" class="footnote-item"><p><a href="https://spark.apache.org/docs/latest/sql-performance-tuning.html" target="_blank" rel="noreferrer">Performance Tuning — Spark SQL, DataFrames and Datasets Guide</a> <a href="#fnref6" class="footnote-backref">↩︎</a></p></li><li id="fn7" class="footnote-item"><p><a href="https://luminousmen.com/post/spark-partitions" target="_blank" rel="noreferrer">Spark Partitions</a> <a href="#fnref7" class="footnote-backref">↩︎</a> <a href="#fnref7:1" class="footnote-backref">↩︎</a> <a href="#fnref7:2" class="footnote-backref">↩︎</a></p></li><li id="fn8" class="footnote-item"><p><a href="https://luminousmen.com/post/spark-tips-caching" target="_blank" rel="noreferrer">Spark Tips: Caching</a> <a href="#fnref8" class="footnote-backref">↩︎</a></p></li><li id="fn9" class="footnote-item"><p><em>Learning Spark, 2nd Edition</em>, Damji, Wenig, Das &amp; Lee, ch. 7 <a href="#fnref9" class="footnote-backref">↩︎</a> <a href="#fnref9:1" class="footnote-backref">↩︎</a></p></li><li id="fn10" class="footnote-item"><p><a href="https://spark.apache.org/docs/latest/job-scheduling.html" target="_blank" rel="noreferrer">Job Scheduling</a> <a href="#fnref10" class="footnote-backref">↩︎</a> <a href="#fnref10:1" class="footnote-backref">↩︎</a></p></li></ol></section>',35)])])}const b=a(c,[["render",l]]);export{k as __pageData,b as default};
@@ -0,0 +1 @@
1
+ import{_ as a,o as t,c as s,a5 as o}from"./chunks/framework.DSg0KOwT.js";const r="../assets/driver-executor.D5pQ7YN1.svg",n="../assets/driver-executor.dark.BmX9cPvh.svg",i="../assets/dag-stages.DSz_S937.svg",f="../assets/dag-stages.dark.F72UzxH4.svg",k=JSON.parse('{"title":"Spark Execution Model","description":"","frontmatter":{"title":"Spark Execution Model"},"headers":[],"relativePath":"tuning-reference/spark-architecture.md","filePath":"tuning-reference/spark-architecture.md"}'),c={name:"tuning-reference/spark-architecture.md"};function l(h,e,d,p,u,g){return t(),s("div",null,[...e[0]||(e[0]=[o("",35)])])}const b=a(c,[["render",l]]);export{k as __pageData,b as default};
@@ -0,0 +1 @@
1
+ import{_ as a,o as t,c as o,a5 as s}from"./chunks/framework.DSg0KOwT.js";const p=JSON.parse('{"title":"Table Formats","description":"","frontmatter":{"title":"Table Formats"},"headers":[],"relativePath":"tuning-reference/table-formats.md","filePath":"tuning-reference/table-formats.md"}'),r={name:"tuning-reference/table-formats.md"};function n(f,e,i,l,c,d){return t(),o("div",null,[...e[0]||(e[0]=[s('<h1 id="table-formats" tabindex="-1">Table Formats <a class="header-anchor" href="#table-formats" aria-label="Permalink to &quot;Table Formats {#table-formats}&quot;">​</a></h1><h2 id="the-knobs-these-formats-share" tabindex="-1">The knobs these formats share <a class="header-anchor" href="#the-knobs-these-formats-share" aria-label="Permalink to &quot;The knobs these formats share&quot;">​</a></h2><p><a href="./data-formats.html">Parquet and ORC</a> stop at the file. A lakehouse <em>table format</em> wraps a directory of Parquet files with a transactional metadata layer that tracks which files belong to the table right now, so engines get ACID commits, time travel, schema evolution, and row-level updates over plain columnar files. The three in wide use are Delta Lake, Apache Iceberg, and Apache Hudi. They solve the same problems and differ mostly in mechanism, and the tuning knobs that matter fall into a handful of dimensions.</p><p><strong>File sizing.</strong> All three fight the small-files problem, but at different points in the write. Delta Lake does not size files during the write by default; its <code>OPTIMIZE</code> command bin-packs already-written small files into larger ones, available since Delta Lake 1.2.0<sup class="footnote-ref"><a href="#fn1" id="fnref1">[1]</a></sup>. Iceberg rewrites files after the fact through its <code>rewrite_data_files</code> procedure, whose default <code>binpack</code> strategy coalesces small files<sup class="footnote-ref"><a href="#fn2" id="fnref2">[2]</a></sup>. Hudi treats correct sizing at write time as &quot;a critical design decision&quot;: auto-sizing targets a 120 MB Parquet base file (<code>hoodie.parquet.max.file.size</code>) and pads any existing file at or below the 100 MB small-file limit (<code>hoodie.parquet.small.file.limit</code>) with new records rather than opening a fresh file<sup class="footnote-ref"><a href="#fn3" id="fnref3">[3]</a></sup>.</p><p><strong>Compaction.</strong> Delta&#39;s <code>OPTIMIZE</code> bin-packing is the compaction path, run manually or, since Delta Lake 3.1.0, as auto compaction that runs synchronously right after a write succeeds<sup class="footnote-ref"><a href="#fn1" id="fnref1:1">[1:1]</a></sup>. Iceberg uses <code>rewrite_data_files</code> as a stored procedure invoked from Spark SQL<sup class="footnote-ref"><a href="#fn2" id="fnref2:1">[2:1]</a></sup>. Hudi splits the job in two: <em>compaction</em> applies only to Merge-on-Read tables and merges the row-based delta logs back into base files (async by default, or inline via <code>hoodie.compact.inline = true</code>), while <em>clustering</em> is a separate data-layout service that stitches small files together<sup class="footnote-ref"><a href="#fn4" id="fnref4">[4]</a></sup><sup class="footnote-ref"><a href="#fn5" id="fnref5">[5]</a></sup>.</p><p><strong>Clustering.</strong> To co-locate rows that are queried together, Delta Lake offers Z-Ordering via <code>OPTIMIZE table ZORDER BY (cols)</code>; its effectiveness drops with each added column and it is not idempotent, since each run reclusters all files in the partition<sup class="footnote-ref"><a href="#fn1" id="fnref1:2">[1:2]</a></sup>. Delta also documents liquid clustering as a separate feature<sup class="footnote-ref"><a href="#fn1" id="fnref1:3">[1:3]</a></sup>. Iceberg clusters through <code>rewrite_data_files</code> with <code>strategy =&gt; &#39;sort&#39;</code> and a <code>sort_order</code> argument, including a <code>zorder(c1,c2)</code> form<sup class="footnote-ref"><a href="#fn2" id="fnref2:2">[2:2]</a></sup>. Hudi clustering rewrites file groups sorting by <code>hoodie.clustering.plan.strategy.sort.columns</code><sup class="footnote-ref"><a href="#fn5" id="fnref5:1">[5:1]</a></sup>.</p><p><strong>Metadata and manifest overhead.</strong> Delta records each change as a JSON commit in the transaction log and periodically compacts those commits into a Parquet checkpoint so readers reconstruct state without replaying every commit; checkpoints can be split multi-part (default 50,000 actions per part), and since Delta 3.0 log-compaction files aggregate a commit range to cut checkpoint frequency<sup class="footnote-ref"><a href="#fn1" id="fnref1:4">[1:4]</a></sup>. Iceberg writes a new metadata JSON file per change, tracks the manifests for each snapshot in a manifest list written fresh on every commit, and splits large tables across multiple manifests so query planning parallelizes<sup class="footnote-ref"><a href="#fn6" id="fnref6">[6]</a></sup>. Hudi records every write and table-service action as an instant on a timeline, with clustering and compaction writing their plans there before executing<sup class="footnote-ref"><a href="#fn5" id="fnref5:2">[5:2]</a></sup><sup class="footnote-ref"><a href="#fn4" id="fnref4:1">[4:1]</a></sup>.</p><p><strong>Snapshot and version expiry.</strong> Delta reclaims storage with <code>VACUUM</code>, which deletes data files past a retention threshold (default 7 days) but never log files; log files are pruned automatically after checkpoints, with a 30-day default set by <code>delta.logRetentionDuration</code><sup class="footnote-ref"><a href="#fn7" id="fnref7">[7]</a></sup>. Iceberg&#39;s <code>expire_snapshots</code> removes old snapshots and the files only they referenced (<code>older_than</code> default 5 days, <code>retain_last</code> default 1), and <code>remove_orphan_files</code> sweeps unreferenced files (<code>older_than</code> default 3 days)<sup class="footnote-ref"><a href="#fn2" id="fnref2:3">[2:3]</a></sup>. Hudi&#39;s cleaner runs automatically after each commit; the default <code>KEEP_LATEST_COMMITS</code> policy retains <code>hoodie.clean.commits.retained</code> commits (default 10)<sup class="footnote-ref"><a href="#fn8" id="fnref8">[8]</a></sup>.</p><p><strong>Row-level deletes: merge-on-read vs copy-on-write.</strong> By default, deleting one row in a Delta table rewrites the whole Parquet file that holds it. Deletion vectors avoid that by marking rows removed in a side file and applying the marks at read time; support landed incrementally (DELETE in 2.4.0, UPDATE in 3.0.0, on by default since 3.1.0)<sup class="footnote-ref"><a href="#fn9" id="fnref9">[9]</a></sup>. Iceberg took a spec-versioned route: v2 adds position and equality delete files, and v3 replaces position deletes with per-file deletion-vector bitmaps stored in the Puffin format<sup class="footnote-ref"><a href="#fn6" id="fnref6:1">[6:1]</a></sup>. Hudi frames the same trade-off as two table types: Copy-on-Write rewrites a base file on every change (fast reads, slower writes), while Merge-on-Read appends changes to log files merged at query time and compacted later (fast writes, some read cost)<sup class="footnote-ref"><a href="#fn10" id="fnref10">[10]</a></sup>.</p><h2 id="where-these-problems-show-up" tabindex="-1">Where these problems show up <a class="header-anchor" href="#where-these-problems-show-up" aria-label="Permalink to &quot;Where these problems show up&quot;">​</a></h2><p>Each of those knobs fails in its own recognizable way. The symptoms are the same ones the <a href="./data-formats.html">Data Formats</a> and <a href="./bottleneck-small-files.html">Small Files</a> pages describe, read through the table&#39;s own metadata. A table accumulating many sub-target files is a sizing problem: Hudi&#39;s own guidance ties small files to more tasks, more per-file open/close cost, and cloud object-store request-rate limits that trip because at least one request is issued per file regardless of size<sup class="footnote-ref"><a href="#fn3" id="fnref3:1">[3:1]</a></sup>.</p><p>Growing metadata is the second signal. Iceberg snapshots and metadata JSON files accumulate until expiry runs, and Iceberg recommends <code>expire_snapshots</code> specifically to keep metadata size small on top of freeing data files<sup class="footnote-ref"><a href="#fn6" id="fnref6:2">[6:2]</a></sup>. Delta&#39;s commit log grows until checkpoints and log compaction absorb it<sup class="footnote-ref"><a href="#fn1" id="fnref1:5">[1:5]</a></sup>. Hudi&#39;s timeline lengthens with every commit, clean, cluster, and compaction<sup class="footnote-ref"><a href="#fn5" id="fnref5:3">[5:3]</a></sup>. Delta&#39;s <code>deltaTable.history()</code> surfaces the per-commit version, timestamp, and operation for inspecting that growth<sup class="footnote-ref"><a href="#fn1" id="fnref1:6">[1:6]</a></sup>.</p><p>Read amplification is the third. On a Merge-on-Read Hudi table, uncompacted delta logs are merged at query time, so a table that has not compacted recently reads more slowly<sup class="footnote-ref"><a href="#fn4" id="fnref4:2">[4:2]</a></sup>. The Iceberg equivalent is a data file with many stacked position/equality delete files that a scan must apply<sup class="footnote-ref"><a href="#fn6" id="fnref6:3">[6:3]</a></sup>.</p><h2 id="what-each-dimension-costs" tabindex="-1">What each dimension costs <a class="header-anchor" href="#what-each-dimension-costs" aria-label="Permalink to &quot;What each dimension costs&quot;">​</a></h2><p>Those symptoms aren&#39;t cosmetic. Small files cost more in metadata and scheduling than their bytes suggest: a query scans many files for the same data, each file adds fixed overhead, and on object storage the per-file request pattern raises the odds of hitting per-prefix rate limits<sup class="footnote-ref"><a href="#fn3" id="fnref3:2">[3:2]</a></sup>. That is exactly the tension the compaction and clustering services exist to resolve, since ingestion favors many small files for low latency while queries favor fewer large ones<sup class="footnote-ref"><a href="#fn3" id="fnref3:3">[3:3]</a></sup>.</p><p>Metadata overhead is a planning tax. More data files mean more manifest entries, and small files inflate that metadata disproportionately and slow query planning, which is why Iceberg exposes <code>rewriteManifests</code> and compaction to cut it<sup class="footnote-ref"><a href="#fn6" id="fnref6:4">[6:4]</a></sup>. Unbounded snapshot and metadata retention keeps that footprint growing until expiry is run<sup class="footnote-ref"><a href="#fn6" id="fnref6:5">[6:5]</a></sup>.</p><p>Clustering decides how much data a query skips. Delta&#39;s Z-Ordering co-locates related values so data-skipping can prune more files, dramatically reducing bytes read<sup class="footnote-ref"><a href="#fn1" id="fnref1:7">[1:7]</a></sup>. The retention settings carry a correctness edge too: once Delta <code>VACUUM</code> runs, time travel to a version older than the retention window is gone<sup class="footnote-ref"><a href="#fn7" id="fnref7:1">[7:1]</a></sup>, and Iceberg warns that running <code>remove_orphan_files</code> with too short an interval can delete in-flight files and corrupt the table<sup class="footnote-ref"><a href="#fn6" id="fnref6:6">[6:6]</a></sup>.</p><h2 id="tuning-each-dimension" tabindex="-1">Tuning each dimension <a class="header-anchor" href="#tuning-each-dimension" aria-label="Permalink to &quot;Tuning each dimension&quot;">​</a></h2><p>Each dimension has a matching maintenance habit that keeps its cost down. Compact on a schedule. For Delta, run <code>OPTIMIZE</code> (optionally scoped with a <code>WHERE</code> partition predicate) or enable auto compaction on 3.1.0+<sup class="footnote-ref"><a href="#fn1" id="fnref1:8">[1:8]</a></sup>. For Iceberg, call <code>rewrite_data_files</code><sup class="footnote-ref"><a href="#fn2" id="fnref2:4">[2:4]</a></sup>. For Merge-on-Read Hudi, let async compaction run or force it inline with <code>hoodie.compact.inline = true</code>, and use clustering to consolidate small files independently of it<sup class="footnote-ref"><a href="#fn4" id="fnref4:3">[4:3]</a></sup><sup class="footnote-ref"><a href="#fn5" id="fnref5:4">[5:4]</a></sup>.</p><p>Cluster the columns you filter on. Use Delta <code>OPTIMIZE ... ZORDER BY (cols)</code> on a few high-cardinality predicate columns<sup class="footnote-ref"><a href="#fn1" id="fnref1:9">[1:9]</a></sup>, Iceberg <code>rewrite_data_files(strategy =&gt; &#39;sort&#39;, sort_order =&gt; &#39;...&#39;)</code> or its <code>zorder(...)</code> form<sup class="footnote-ref"><a href="#fn2" id="fnref2:5">[2:5]</a></sup>, or Hudi clustering with <code>hoodie.clustering.plan.strategy.sort.columns</code><sup class="footnote-ref"><a href="#fn5" id="fnref5:5">[5:5]</a></sup>.</p><p>Expire aggressively but safely. Run Delta <code>VACUUM</code> (keeping the 7-day floor unless you have a reason to override it, since the safety check exists to protect concurrent readers)<sup class="footnote-ref"><a href="#fn7" id="fnref7:2">[7:2]</a></sup>, Iceberg <code>expire_snapshots</code> plus periodic <code>remove_orphan_files</code> with an interval longer than your longest in-flight write<sup class="footnote-ref"><a href="#fn6" id="fnref6:7">[6:7]</a></sup>, and tune Hudi&#39;s cleaner via <code>hoodie.clean.commits.retained</code><sup class="footnote-ref"><a href="#fn8" id="fnref8:1">[8:1]</a></sup>.</p><p>Match the delete strategy to the workload. Enable Delta deletion vectors (<code>ALTER TABLE ... SET TBLPROPERTIES(&#39;delta.enableDeletionVectors&#39; = true)</code>) so DML marks rows instead of rewriting files, remembering the marks are applied physically only on <code>OPTIMIZE</code> or <code>REORG TABLE ... APPLY (PURGE)</code><sup class="footnote-ref"><a href="#fn9" id="fnref9:1">[9:1]</a></sup>. On Iceberg, prefer merge-on-read deletes for update-heavy tables and compact the delete files<sup class="footnote-ref"><a href="#fn6" id="fnref6:8">[6:8]</a></sup>. On Hudi, pick Copy-on-Write for read-heavy tables and Merge-on-Read for write-heavy or near-real-time ingestion<sup class="footnote-ref"><a href="#fn10" id="fnref10:1">[10:1]</a></sup>.</p><h2 id="sources" tabindex="-1">Sources <a class="header-anchor" href="#sources" aria-label="Permalink to &quot;Sources&quot;">​</a></h2><hr class="footnotes-sep"><section class="footnotes"><ol class="footnotes-list"><li id="fn1" class="footnote-item"><p><a href="https://docs.delta.io/latest/optimizations-oss.html" target="_blank" rel="noreferrer">Delta Lake: Optimizations (OSS)</a> <a href="#fnref1" class="footnote-backref">↩︎</a> <a href="#fnref1:1" class="footnote-backref">↩︎</a> <a href="#fnref1:2" class="footnote-backref">↩︎</a> <a href="#fnref1:3" class="footnote-backref">↩︎</a> <a href="#fnref1:4" class="footnote-backref">↩︎</a> <a href="#fnref1:5" class="footnote-backref">↩︎</a> <a href="#fnref1:6" class="footnote-backref">↩︎</a> <a href="#fnref1:7" class="footnote-backref">↩︎</a> <a href="#fnref1:8" class="footnote-backref">↩︎</a> <a href="#fnref1:9" class="footnote-backref">↩︎</a></p></li><li id="fn2" class="footnote-item"><p><a href="https://iceberg.apache.org/docs/latest/spark-procedures/" target="_blank" rel="noreferrer">Apache Iceberg: Spark Procedures</a> <a href="#fnref2" class="footnote-backref">↩︎</a> <a href="#fnref2:1" class="footnote-backref">↩︎</a> <a href="#fnref2:2" class="footnote-backref">↩︎</a> <a href="#fnref2:3" class="footnote-backref">↩︎</a> <a href="#fnref2:4" class="footnote-backref">↩︎</a> <a href="#fnref2:5" class="footnote-backref">↩︎</a></p></li><li id="fn3" class="footnote-item"><p><a href="https://hudi.apache.org/docs/file_sizing/" target="_blank" rel="noreferrer">Apache Hudi: File Sizing</a> <a href="#fnref3" class="footnote-backref">↩︎</a> <a href="#fnref3:1" class="footnote-backref">↩︎</a> <a href="#fnref3:2" class="footnote-backref">↩︎</a> <a href="#fnref3:3" class="footnote-backref">↩︎</a></p></li><li id="fn4" class="footnote-item"><p><a href="https://hudi.apache.org/docs/compaction/" target="_blank" rel="noreferrer">Apache Hudi: Compaction</a> <a href="#fnref4" class="footnote-backref">↩︎</a> <a href="#fnref4:1" class="footnote-backref">↩︎</a> <a href="#fnref4:2" class="footnote-backref">↩︎</a> <a href="#fnref4:3" class="footnote-backref">↩︎</a></p></li><li id="fn5" class="footnote-item"><p><a href="https://hudi.apache.org/docs/clustering/" target="_blank" rel="noreferrer">Apache Hudi: Clustering</a> <a href="#fnref5" class="footnote-backref">↩︎</a> <a href="#fnref5:1" class="footnote-backref">↩︎</a> <a href="#fnref5:2" class="footnote-backref">↩︎</a> <a href="#fnref5:3" class="footnote-backref">↩︎</a> <a href="#fnref5:4" class="footnote-backref">↩︎</a> <a href="#fnref5:5" class="footnote-backref">↩︎</a></p></li><li id="fn6" class="footnote-item"><p><a href="https://iceberg.apache.org/docs/latest/maintenance/" target="_blank" rel="noreferrer">Apache Iceberg: Maintenance</a> and <a href="https://iceberg.apache.org/spec/" target="_blank" rel="noreferrer">Table Spec</a> <a href="#fnref6" class="footnote-backref">↩︎</a> <a href="#fnref6:1" class="footnote-backref">↩︎</a> <a href="#fnref6:2" class="footnote-backref">↩︎</a> <a href="#fnref6:3" class="footnote-backref">↩︎</a> <a href="#fnref6:4" class="footnote-backref">↩︎</a> <a href="#fnref6:5" class="footnote-backref">↩︎</a> <a href="#fnref6:6" class="footnote-backref">↩︎</a> <a href="#fnref6:7" class="footnote-backref">↩︎</a> <a href="#fnref6:8" class="footnote-backref">↩︎</a></p></li><li id="fn7" class="footnote-item"><p><a href="https://docs.delta.io/latest/delta-utility.html" target="_blank" rel="noreferrer">Delta Lake: Table Utility Commands</a> <a href="#fnref7" class="footnote-backref">↩︎</a> <a href="#fnref7:1" class="footnote-backref">↩︎</a> <a href="#fnref7:2" class="footnote-backref">↩︎</a></p></li><li id="fn8" class="footnote-item"><p><a href="https://hudi.apache.org/docs/cleaning/" target="_blank" rel="noreferrer">Apache Hudi: Cleaning</a> <a href="#fnref8" class="footnote-backref">↩︎</a> <a href="#fnref8:1" class="footnote-backref">↩︎</a></p></li><li id="fn9" class="footnote-item"><p><a href="https://docs.delta.io/latest/delta-deletion-vectors.html" target="_blank" rel="noreferrer">Delta Lake: Deletion Vectors</a> <a href="#fnref9" class="footnote-backref">↩︎</a> <a href="#fnref9:1" class="footnote-backref">↩︎</a></p></li><li id="fn10" class="footnote-item"><p><a href="https://hudi.apache.org/docs/table_types/" target="_blank" rel="noreferrer">Apache Hudi: Table Types</a> <a href="#fnref10" class="footnote-backref">↩︎</a> <a href="#fnref10:1" class="footnote-backref">↩︎</a></p></li></ol></section>',25)])])}const u=a(r,[["render",n]]);export{p as __pageData,u as default};
@@ -0,0 +1 @@
1
+ import{_ as a,o as t,c as o,a5 as s}from"./chunks/framework.DSg0KOwT.js";const p=JSON.parse('{"title":"Table Formats","description":"","frontmatter":{"title":"Table Formats"},"headers":[],"relativePath":"tuning-reference/table-formats.md","filePath":"tuning-reference/table-formats.md"}'),r={name:"tuning-reference/table-formats.md"};function n(f,e,i,l,c,d){return t(),o("div",null,[...e[0]||(e[0]=[s("",25)])])}const u=a(r,[["render",n]]);export{p as __pageData,u as default};
@@ -0,0 +1 @@
1
+ <svg id="gd33" width="100%" xmlns="http://www.w3.org/2000/svg" class="flowchart" viewBox="-8 -17.3125 730.3125 506.7875" role="graphics-document document" aria-roledescription="flowchart-v2" style="max-width: 730.3125px;"><style>#gd33{font-family:Recursive,ui-sans-serif,system-ui,-apple-system,sans-serif;font-size:16px;fill:#151b1e;}@keyframes edge-animation-frame{from{stroke-dashoffset:0;}}@keyframes dash{to{stroke-dashoffset:0;}}#gd33 .edge-animation-slow{stroke-dasharray:9,5!important;stroke-dashoffset:900;animation:dash 50s linear infinite;stroke-linecap:round;}#gd33 .edge-animation-fast{stroke-dasharray:9,5!important;stroke-dashoffset:900;animation:dash 20s linear infinite;stroke-linecap:round;}#gd33 .error-icon{fill:#eef1f6;}#gd33 .error-text{fill:#110e09;stroke:#110e09;}#gd33 .edge-thickness-normal{stroke-width:1px;}#gd33 .edge-thickness-thick{stroke-width:3.5px;}#gd33 .edge-pattern-solid{stroke-dasharray:0;}#gd33 .edge-thickness-invisible{stroke-width:0;fill:none;}#gd33 .edge-pattern-dashed{stroke-dasharray:3;}#gd33 .edge-pattern-dotted{stroke-dasharray:2;}#gd33 .marker{fill:#4f5f60;stroke:#4f5f60;}#gd33 .marker.cross{stroke:#4f5f60;}#gd33 svg{font-family:Recursive,ui-sans-serif,system-ui,-apple-system,sans-serif;font-size:16px;}#gd33 p{margin:0;}#gd33 .label{font-family:Recursive,ui-sans-serif,system-ui,-apple-system,sans-serif;color:#151b1e;}#gd33 .cluster-label text{fill:#151b1e;}#gd33 .cluster-label span{color:#151b1e;}#gd33 .cluster-label span p{background-color:transparent;}#gd33 .label text,#gd33 span{fill:#151b1e;color:#151b1e;}#gd33 .node rect,#gd33 .node circle,#gd33 .node ellipse,#gd33 .node polygon,#gd33 .node path{fill:#ffffff;stroke:#9aa8a3;stroke-width:1px;}#gd33 .rough-node .label text,#gd33 .node .label text,#gd33 .image-shape .label,#gd33 .icon-shape .label{text-anchor:middle;}#gd33 .node .katex path{fill:#000;stroke:#000;stroke-width:1px;}#gd33 .rough-node .label,#gd33 .node .label,#gd33 .image-shape .label,#gd33 .icon-shape .label{text-align:center;}#gd33 .node.clickable{cursor:pointer;}#gd33 .root .anchor path{fill:#4f5f60!important;stroke-width:0;stroke:#4f5f60;}#gd33 .arrowheadPath{fill:rgba(255, 255, 255, 0);}#gd33 .edgePath .path{stroke:#4f5f60;stroke-width:1px;}#gd33 .flowchart-link{stroke:#4f5f60;fill:none;}#gd33 .edgeLabel{background-color:#eef1f6;text-align:center;}#gd33 .edgeLabel p{background-color:#eef1f6;}#gd33 .edgeLabel rect{opacity:0.5;background-color:#eef1f6;fill:#eef1f6;}#gd33 .labelBkg{background-color:rgba(238, 241, 246, 0.5);}#gd33 .cluster rect{fill:#f2f5f9;stroke:#dbe2df;stroke-width:1px;}#gd33 .cluster text{fill:#151b1e;}#gd33 .cluster span{color:#151b1e;}#gd33 div.mermaidTooltip{position:absolute;text-align:center;max-width:200px;padding:2px;font-family:Recursive,ui-sans-serif,system-ui,-apple-system,sans-serif;font-size:12px;background:#eef1f6;border:1px solid hsl(217.5, 0%, 84.9019607843%);border-radius:2px;pointer-events:none;z-index:100;}#gd33 .flowchartTitleText{text-anchor:middle;font-size:18px;fill:#151b1e;}#gd33 rect.text{fill:none;stroke-width:0;}#gd33 .icon-shape,#gd33 .image-shape{background-color:#eef1f6;text-align:center;}#gd33 .icon-shape p,#gd33 .image-shape p{background-color:#eef1f6;padding:2px;}#gd33 .icon-shape .label rect,#gd33 .image-shape .label rect{opacity:0.5;background-color:#eef1f6;fill:#eef1f6;}#gd33 .label-icon{display:inline-block;height:1em;overflow:visible;vertical-align:-0.125em;}#gd33 .node .label-icon path{fill:currentColor;stroke:revert;stroke-width:revert;}#gd33 .node .neo-node{stroke:#9aa8a3;}#gd33 [data-look=&quot;neo&quot;].node rect,#gd33 [data-look=&quot;neo&quot;].cluster rect,#gd33 [data-look=&quot;neo&quot;].node polygon{stroke:url(#gd33-gradient);filter:drop-shadow( 1px 2px 2px rgba(185,185,185,1));}#gd33 [data-look=&quot;neo&quot;].swimlane.cluster rect{filter:none;}#gd33 [data-look=&quot;neo&quot;].node path{stroke:url(#gd33-gradient);stroke-width:1px;}#gd33 [data-look=&quot;neo&quot;].node .outer-path{filter:drop-shadow( 1px 2px 2px rgba(185,185,185,1));}#gd33 [data-look=&quot;neo&quot;].node .neo-line path{stroke:#9aa8a3;filter:none;}#gd33 [data-look=&quot;neo&quot;].node circle{stroke:url(#gd33-gradient);filter:drop-shadow( 1px 2px 2px rgba(185,185,185,1));}#gd33 [data-look=&quot;neo&quot;].node circle .state-start{fill:#000000;}#gd33 [data-look=&quot;neo&quot;].icon-shape .icon{fill:url(#gd33-gradient);filter:drop-shadow( 1px 2px 2px rgba(185,185,185,1));}#gd33 [data-look=&quot;neo&quot;].icon-shape .icon-neo path{stroke:url(#gd33-gradient);filter:drop-shadow( 1px 2px 2px rgba(185,185,185,1));}#gd33 :root{--mermaid-font-family:Recursive,ui-sans-serif,system-ui,-apple-system,sans-serif;}</style><g><marker id="gd33_flowchart-v2-pointEnd" class="marker flowchart-v2" viewBox="0 0 10 10" refX="5" refY="5" markerUnits="userSpaceOnUse" markerWidth="8" markerHeight="8" orient="auto"><path d="M 0 0 L 10 5 L 0 10 z" class="arrowMarkerPath" style="stroke-width:1;stroke-dasharray:1,0"></path></marker><marker id="gd33_flowchart-v2-pointStart" class="marker flowchart-v2" viewBox="0 0 10 10" refX="4.5" refY="5" markerUnits="userSpaceOnUse" markerWidth="8" markerHeight="8" orient="auto"><path d="M 0 5 L 10 10 L 10 0 z" class="arrowMarkerPath" style="stroke-width:1;stroke-dasharray:1,0"></path></marker><marker id="gd33_flowchart-v2-pointEnd-margin" class="marker flowchart-v2" viewBox="0 0 11.5 14" refX="11.5" refY="7" markerUnits="userSpaceOnUse" markerWidth="10.5" markerHeight="14" orient="auto"><path d="M 0 0 L 11.5 7 L 0 14 z" class="arrowMarkerPath" style="stroke-width:0;stroke-dasharray:1,0"></path></marker><marker id="gd33_flowchart-v2-pointStart-margin" class="marker flowchart-v2" viewBox="0 0 11.5 14" refX="1" refY="7" markerUnits="userSpaceOnUse" markerWidth="11.5" markerHeight="14" orient="auto"><polygon points="0,7 11.5,14 11.5,0" class="arrowMarkerPath" style="stroke-width:0;stroke-dasharray:1,0"></polygon></marker><marker id="gd33_flowchart-v2-circleEnd" class="marker flowchart-v2" viewBox="0 0 10 10" refX="11" refY="5" markerUnits="userSpaceOnUse" markerWidth="11" markerHeight="11" orient="auto"><circle cx="5" cy="5" r="5" class="arrowMarkerPath" style="stroke-width:1;stroke-dasharray:1,0"></circle></marker><marker id="gd33_flowchart-v2-circleStart" class="marker flowchart-v2" viewBox="0 0 10 10" refX="-1" refY="5" markerUnits="userSpaceOnUse" markerWidth="11" markerHeight="11" orient="auto"><circle cx="5" cy="5" r="5" class="arrowMarkerPath" style="stroke-width:1;stroke-dasharray:1,0"></circle></marker><marker id="gd33_flowchart-v2-circleEnd-margin" class="marker flowchart-v2" viewBox="0 0 10 10" refY="5" refX="12.25" markerUnits="userSpaceOnUse" markerWidth="14" markerHeight="14" orient="auto"><circle cx="5" cy="5" r="5" class="arrowMarkerPath" style="stroke-width:0;stroke-dasharray:1,0"></circle></marker><marker id="gd33_flowchart-v2-circleStart-margin" class="marker flowchart-v2" viewBox="0 0 10 10" refX="-2" refY="5" markerUnits="userSpaceOnUse" markerWidth="14" markerHeight="14" orient="auto"><circle cx="5" cy="5" r="5" class="arrowMarkerPath" style="stroke-width:0;stroke-dasharray:1,0"></circle></marker><marker id="gd33_flowchart-v2-crossEnd" class="marker cross flowchart-v2" viewBox="0 0 11 11" refX="12" refY="5.2" markerUnits="userSpaceOnUse" markerWidth="11" markerHeight="11" orient="auto"><path d="M 1,1 l 9,9 M 10,1 l -9,9" class="arrowMarkerPath" style="stroke-width:2;stroke-dasharray:1,0"></path></marker><marker id="gd33_flowchart-v2-crossStart" class="marker cross flowchart-v2" viewBox="0 0 11 11" refX="-1" refY="5.2" markerUnits="userSpaceOnUse" markerWidth="11" markerHeight="11" orient="auto"><path d="M 1,1 l 9,9 M 10,1 l -9,9" class="arrowMarkerPath" style="stroke-width:2;stroke-dasharray:1,0"></path></marker><marker id="gd33_flowchart-v2-crossEnd-margin" class="marker cross flowchart-v2" viewBox="0 0 15 15" refX="17.7" refY="7.5" markerUnits="userSpaceOnUse" markerWidth="12" markerHeight="12" orient="auto"><path d="M 1,1 L 14,14 M 1,14 L 14,1" class="arrowMarkerPath" style="stroke-width:2.5"></path></marker><marker id="gd33_flowchart-v2-crossStart-margin" class="marker cross flowchart-v2" viewBox="0 0 15 15" refX="-3.5" refY="7.5" markerUnits="userSpaceOnUse" markerWidth="12" markerHeight="12" orient="auto"><path d="M 1,1 L 14,14 M 1,14 L 14,1" class="arrowMarkerPath" style="stroke-width:2.5;stroke-dasharray:1,0"></path></marker><g class="root"><g class="clusters"></g><g class="edgePaths"><path d="M361.156,126.625L361.156,130.792C361.156,134.958,361.156,143.292,361.156,151.625C361.156,159.958,361.156,168.292,361.156,172.458L361.156,176.625" id="gd33-L_PLAIN_ARROW_0" class=" edge-thickness-invisible edge-pattern-solid" data-edge="true" data-et="edge" data-id="L_PLAIN_ARROW_0" data-points="W3sieCI6MzYxLjE1NjI1LCJ5IjoxMjYuNjI1fSx7IngiOjM2MS4xNTYyNSwieSI6MTUxLjYyNX0seyJ4IjozNjEuMTU2MjUsInkiOjE3Ni42MjV9XQ==" data-look="classic" style=";"></path><path d="M361.156,295.25L361.156,299.417C361.156,303.583,361.156,311.917,361.156,320.25C361.156,328.583,361.156,336.917,361.156,341.083L361.156,345.25" id="gd33-L_ARROW_PANDAS_0" class=" edge-thickness-invisible edge-pattern-solid" data-edge="true" data-et="edge" data-id="L_ARROW_PANDAS_0" data-points="W3sieCI6MzYxLjE1NjI1LCJ5IjoyOTUuMjV9LHsieCI6MzYxLjE1NjI1LCJ5IjozMjAuMjV9LHsieCI6MzYxLjE1NjI1LCJ5IjozNDUuMjV9XQ==" data-look="classic" style=";"></path></g><g class="edgeLabels"><g class="edgeLabel"><g class="label" data-id="L_PLAIN_ARROW_0" transform="translate(0, -10.4609375)"><text y="-10.1" text-anchor="middle"><tspan class="text-outer-tspan row" x="0" y="-0.1em" dy="1.1em" text-anchor="middle"></tspan></text></g></g><g><rect class="background" style="stroke: none"></rect></g><g class="edgeLabel"><g class="label" data-id="L_ARROW_PANDAS_0" transform="translate(0, -10.4609375)"><text y="-10.1" text-anchor="middle"><tspan class="text-outer-tspan row" x="0" y="-0.1em" dy="1.1em" text-anchor="middle"></tspan></text></g></g><g><rect class="background" style="stroke: none"></rect></g></g><g class="nodes"><g class="root" transform="translate(0, 337.25)"><g class="clusters"><g class="cluster " id="gd33-PANDAS" data-look="classic"><rect x="8" y="8" width="706.3125" height="136.225"></rect><g class="cluster-label " transform="translate(262.21484375, 8)"><g><rect class="background" style="stroke: none"></rect><text y="-10.1"><tspan class="text-outer-tspan row" x="0" y="-0.1em" dy="1.1em"><tspan font-style="normal" class="text-inner-tspan" font-weight="normal">pandas</tspan><tspan font-style="normal" class="text-inner-tspan" font-weight="normal"> UDF</tspan><tspan font-style="normal" class="text-inner-tspan" font-weight="normal"> (vectorized)</tspan></tspan></text></g></g></g></g><g class="edgePaths"><path d="M134.969,76.113L155.96,76.113C176.951,76.113,218.932,76.113,260.247,76.113C301.563,76.113,342.211,76.113,362.535,76.113L382.859,76.113" id="gd33-L_JVM3_PY3_0" class=" edge-thickness-normal edge-pattern-solid edge-thickness-normal edge-pattern-solid flowchart-link" data-edge="true" data-et="edge" data-id="L_JVM3_PY3_0" data-points="W3sieCI6MTM0Ljk2ODc1LCJ5Ijo3Ni4xMTI1fSx7IngiOjI2MC45MTQwNjI1LCJ5Ijo3Ni4xMTI1fSx7IngiOjM4Ni44NTkzNzUsInkiOjc2LjExMjV9XQ==" data-look="classic" marker-end="url(#gd33_flowchart-v2-pointEnd)" style=";"></path></g><g class="edgeLabels"><g class="edgeLabel" transform="translate(260.9140625, 76.1125)"><g class="label" data-id="L_JVM3_PY3_0" transform="translate(0, -19.2609375)"><g><rect class="background" x="-88.4453125" y="-0.8515625" width="176.890625" height="40.225"></rect><text y="-10.1" text-anchor="middle"><tspan class="text-outer-tspan row" x="0" y="-0.1em" dy="1.1em" text-anchor="middle"><tspan font-style="normal" class="text-inner-tspan" font-weight="normal">whole</tspan><tspan font-style="normal" class="text-inner-tspan" font-weight="normal"> Arrow</tspan><tspan font-style="normal" class="text-inner-tspan" font-weight="normal"> batches,</tspan></tspan><tspan class="text-outer-tspan row" x="0" y="1em" dy="1.1em" text-anchor="middle"><tspan font-style="normal" class="text-inner-tspan" font-weight="normal">no</tspan><tspan font-style="normal" class="text-inner-tspan" font-weight="normal"> per-row</tspan><tspan font-style="normal" class="text-inner-tspan" font-weight="normal"> handoff</tspan></tspan></text></g></g></g></g><g class="nodes"><g class="node default " id="gd33-flowchart-JVM3-4" data-look="classic" transform="translate(90.234375, 76.1125)"><rect class="basic label-container" x="-44.734375" y="-24.3125" width="89.46875" height="48.625"></rect><g class="label" transform="translate(0, -9.3125)"><rect></rect><g><rect class="background" style="stroke: none"></rect><text y="-10.1"><tspan class="text-outer-tspan row" x="0" y="-0.1em" dy="1.1em"><tspan font-style="normal" class="text-inner-tspan" font-weight="normal">JVM</tspan></tspan></text></g></g></g><g class="node default " id="gd33-flowchart-PY3-5" data-look="classic" transform="translate(531.8359375, 76.1125)"><rect class="basic label-container" x="-144.9765625" y="-33.1125" width="289.953125" height="66.225"></rect><g class="label" transform="translate(0, -18.1125)"><rect></rect><g><rect class="background" style="stroke: none"></rect><text y="-10.1"><tspan class="text-outer-tspan row" x="0" y="-0.1em" dy="1.1em"><tspan font-style="normal" class="text-inner-tspan" font-weight="normal">Python</tspan></tspan><tspan class="text-outer-tspan row" x="0" y="1em" dy="1.1em"><tspan font-style="normal" class="text-inner-tspan" font-weight="normal">(pandas</tspan><tspan font-style="normal" class="text-inner-tspan" font-weight="normal"> Series</tspan><tspan font-style="normal" class="text-inner-tspan" font-weight="normal"> /</tspan><tspan font-style="normal" class="text-inner-tspan" font-weight="normal"> DataFrame)</tspan></tspan></text></g></g></g></g></g><g class="root" transform="translate(99.125, 168.625)"><g class="clusters"><g class="cluster " id="gd33-ARROW" data-look="classic"><rect x="8" y="8" width="508.0625" height="118.625"></rect><g class="cluster-label " transform="translate(146.67578125, 8)"><g><rect class="background" style="stroke: none"></rect><text y="-10.1"><tspan class="text-outer-tspan row" x="0" y="-0.1em" dy="1.1em"><tspan font-style="normal" class="text-inner-tspan" font-weight="normal">Arrow-optimized</tspan><tspan font-style="normal" class="text-inner-tspan" font-weight="normal"> Python</tspan><tspan font-style="normal" class="text-inner-tspan" font-weight="normal"> UDF</tspan></tspan></text></g></g></g></g><g class="edgePaths"><path d="M134.969,67.313L153.98,67.313C172.991,67.313,211.013,67.313,248.368,67.313C285.724,67.313,322.413,67.313,340.757,67.313L359.102,67.313" id="gd33-L_JVM2_PY2_0" class=" edge-thickness-normal edge-pattern-solid edge-thickness-normal edge-pattern-solid flowchart-link" data-edge="true" data-et="edge" data-id="L_JVM2_PY2_0" data-points="W3sieCI6MTM0Ljk2ODc1LCJ5Ijo2Ny4zMTI1fSx7IngiOjI0OS4wMzUxNTYyNSwieSI6NjcuMzEyNX0seyJ4IjozNjMuMTAxNTYyNSwieSI6NjcuMzEyNX1d" data-look="classic" marker-end="url(#gd33_flowchart-v2-pointEnd)" style=";"></path></g><g class="edgeLabels"><g class="edgeLabel" transform="translate(249.03515625, 67.3125)"><g class="label" data-id="L_JVM2_PY2_0" transform="translate(0, -19.2609375)"><g><rect class="background" x="-76.56640625" y="-0.8515625" width="153.1328125" height="40.225"></rect><text y="-10.1" text-anchor="middle"><tspan class="text-outer-tspan row" x="0" y="-0.1em" dy="1.1em" text-anchor="middle"><tspan font-style="normal" class="text-inner-tspan" font-weight="normal">Arrow</tspan><tspan font-style="normal" class="text-inner-tspan" font-weight="normal"> transfer,</tspan></tspan><tspan class="text-outer-tspan row" x="0" y="1em" dy="1.1em" text-anchor="middle"><tspan font-style="normal" class="text-inner-tspan" font-weight="normal">still</tspan><tspan font-style="normal" class="text-inner-tspan" font-weight="normal"> row</tspan><tspan font-style="normal" class="text-inner-tspan" font-weight="normal"> semantics</tspan></tspan></text></g></g></g></g><g class="nodes"><g class="node default " id="gd33-flowchart-JVM2-2" data-look="classic" transform="translate(90.234375, 67.3125)"><rect class="basic label-container" x="-44.734375" y="-24.3125" width="89.46875" height="48.625"></rect><g class="label" transform="translate(0, -9.3125)"><rect></rect><g><rect class="background" style="stroke: none"></rect><text y="-10.1"><tspan class="text-outer-tspan row" x="0" y="-0.1em" dy="1.1em"><tspan font-style="normal" class="text-inner-tspan" font-weight="normal">JVM</tspan></tspan></text></g></g></g><g class="node default " id="gd33-flowchart-PY2-3" data-look="classic" transform="translate(420.83203125, 67.3125)"><rect class="basic label-container" x="-57.73046875" y="-24.3125" width="115.4609375" height="48.625"></rect><g class="label" transform="translate(0, -9.3125)"><rect></rect><g><rect class="background" style="stroke: none"></rect><text y="-10.1"><tspan class="text-outer-tspan row" x="0" y="-0.1em" dy="1.1em"><tspan font-style="normal" class="text-inner-tspan" font-weight="normal">Python</tspan></tspan></text></g></g></g></g></g><g class="root" transform="translate(87.22265625, 0)"><g class="clusters"><g class="cluster " id="gd33-PLAIN" data-look="classic"><rect x="8" y="8" width="531.8671875" height="118.625"></rect><g class="cluster-label " transform="translate(205.2578125, 8)"><g><rect class="background" style="stroke: none"></rect><text y="-10.1"><tspan class="text-outer-tspan row" x="0" y="-0.1em" dy="1.1em"><tspan font-style="normal" class="text-inner-tspan" font-weight="normal">Plain</tspan><tspan font-style="normal" class="text-inner-tspan" font-weight="normal"> Python</tspan><tspan font-style="normal" class="text-inner-tspan" font-weight="normal"> UDF</tspan></tspan></text></g></g></g></g><g class="edgePaths"><path d="M134.969,67.313L155.964,67.313C176.958,67.313,218.948,67.313,260.271,67.313C301.594,67.313,342.25,67.313,362.578,67.313L382.906,67.313" id="gd33-L_JVM1_PY1_0" class=" edge-thickness-normal edge-pattern-solid edge-thickness-normal edge-pattern-solid flowchart-link" data-edge="true" data-et="edge" data-id="L_JVM1_PY1_0" data-points="W3sieCI6MTM0Ljk2ODc1LCJ5Ijo2Ny4zMTI1fSx7IngiOjI2MC45Mzc1LCJ5Ijo2Ny4zMTI1fSx7IngiOjM4Ni45MDYyNSwieSI6NjcuMzEyNX1d" data-look="classic" marker-end="url(#gd33_flowchart-v2-pointEnd)" style=";"></path></g><g class="edgeLabels"><g class="edgeLabel" transform="translate(260.9375, 67.3125)"><g class="label" data-id="L_JVM1_PY1_0" transform="translate(0, -28.0609375)"><g><rect class="background" x="-88.46875" y="-0.8515625" width="176.9375" height="57.825"></rect><text y="-10.1" text-anchor="middle"><tspan class="text-outer-tspan row" x="0" y="-0.1em" dy="1.1em" text-anchor="middle"><tspan font-style="normal" class="text-inner-tspan" font-weight="normal">pickle,</tspan><tspan font-style="normal" class="text-inner-tspan" font-weight="normal"> row-at-a-time</tspan></tspan><tspan class="text-outer-tspan row" x="0" y="1em" dy="1.1em" text-anchor="middle"><tspan font-style="normal" class="text-inner-tspan" font-weight="normal">(serialization</tspan><tspan font-style="normal" class="text-inner-tspan" font-weight="normal"> cost</tspan><tspan font-style="normal" class="text-inner-tspan" font-weight="normal"> per</tspan></tspan><tspan class="text-outer-tspan row" x="0" y="2.1em" dy="1.1em" text-anchor="middle"><tspan font-style="normal" class="text-inner-tspan" font-weight="normal">row)</tspan></tspan></text></g></g></g></g><g class="nodes"><g class="node default " id="gd33-flowchart-JVM1-0" data-look="classic" transform="translate(90.234375, 67.3125)"><rect class="basic label-container" x="-44.734375" y="-24.3125" width="89.46875" height="48.625"></rect><g class="label" transform="translate(0, -9.3125)"><rect></rect><g><rect class="background" style="stroke: none"></rect><text y="-10.1"><tspan class="text-outer-tspan row" x="0" y="-0.1em" dy="1.1em"><tspan font-style="normal" class="text-inner-tspan" font-weight="normal">JVM</tspan></tspan></text></g></g></g><g class="node default " id="gd33-flowchart-PY1-1" data-look="classic" transform="translate(444.63671875, 67.3125)"><rect class="basic label-container" x="-57.73046875" y="-24.3125" width="115.4609375" height="48.625"></rect><g class="label" transform="translate(0, -9.3125)"><rect></rect><g><rect class="background" style="stroke: none"></rect><text y="-10.1"><tspan class="text-outer-tspan row" x="0" y="-0.1em" dy="1.1em"><tspan font-style="normal" class="text-inner-tspan" font-weight="normal">Python</tspan></tspan></text></g></g></g></g></g></g></g></g><defs><filter id="gd33-drop-shadow" height="130%" width="130%"><feDropShadow dx="4" dy="4" stdDeviation="0" flood-opacity="0.06" flood-color="#000000"></feDropShadow></filter></defs><defs><filter id="gd33-drop-shadow-small" height="150%" width="150%"><feDropShadow dx="2" dy="2" stdDeviation="0" flood-opacity="0.06" flood-color="#000000"></feDropShadow></filter></defs><linearGradient id="gd33-gradient" gradientUnits="objectBoundingBox" x1="0%" y1="0%" x2="100%" y2="0%"><stop offset="0%" stop-color="#9aa8a3" stop-opacity="1"></stop><stop offset="100%" stop-color="hsl(214.2857142857, 0%, 86.2745098039%)" stop-opacity="1"></stop></linearGradient></svg>
@@ -0,0 +1 @@
1
+ <svg id="gd34" width="100%" xmlns="http://www.w3.org/2000/svg" class="flowchart" viewBox="-8 -17.3125 730.3125 506.7875" role="graphics-document document" aria-roledescription="flowchart-v2" style="max-width: 730.3125px;"><style>#gd34{font-family:Recursive,ui-sans-serif,system-ui,-apple-system,sans-serif;font-size:16px;fill:#f4f4f5;}@keyframes edge-animation-frame{from{stroke-dashoffset:0;}}@keyframes dash{to{stroke-dashoffset:0;}}#gd34 .edge-animation-slow{stroke-dasharray:9,5!important;stroke-dashoffset:900;animation:dash 50s linear infinite;stroke-linecap:round;}#gd34 .edge-animation-fast{stroke-dasharray:9,5!important;stroke-dashoffset:900;animation:dash 20s linear infinite;stroke-linecap:round;}#gd34 .error-icon{fill:#1a232f;}#gd34 .error-text{fill:#e5dcd0;stroke:#e5dcd0;}#gd34 .edge-thickness-normal{stroke-width:1px;}#gd34 .edge-thickness-thick{stroke-width:3.5px;}#gd34 .edge-pattern-solid{stroke-dasharray:0;}#gd34 .edge-thickness-invisible{stroke-width:0;fill:none;}#gd34 .edge-pattern-dashed{stroke-dasharray:3;}#gd34 .edge-pattern-dotted{stroke-dasharray:2;}#gd34 .marker{fill:#a9b5b4;stroke:#a9b5b4;}#gd34 .marker.cross{stroke:#a9b5b4;}#gd34 svg{font-family:Recursive,ui-sans-serif,system-ui,-apple-system,sans-serif;font-size:16px;}#gd34 p{margin:0;}#gd34 .label{font-family:Recursive,ui-sans-serif,system-ui,-apple-system,sans-serif;color:#f4f4f5;}#gd34 .cluster-label text{fill:#f4f4f5;}#gd34 .cluster-label span{color:#f4f4f5;}#gd34 .cluster-label span p{background-color:transparent;}#gd34 .label text,#gd34 span{fill:#f4f4f5;color:#f4f4f5;}#gd34 .node rect,#gd34 .node circle,#gd34 .node ellipse,#gd34 .node polygon,#gd34 .node path{fill:#17202c;stroke:#3a4a52;stroke-width:1px;}#gd34 .rough-node .label text,#gd34 .node .label text,#gd34 .image-shape .label,#gd34 .icon-shape .label{text-anchor:middle;}#gd34 .node .katex path{fill:#000;stroke:#000;stroke-width:1px;}#gd34 .rough-node .label,#gd34 .node .label,#gd34 .image-shape .label,#gd34 .icon-shape .label{text-align:center;}#gd34 .node.clickable{cursor:pointer;}#gd34 .root .anchor path{fill:#a9b5b4!important;stroke-width:0;stroke:#a9b5b4;}#gd34 .arrowheadPath{fill:rgba(255, 255, 255, 0);}#gd34 .edgePath .path{stroke:#a9b5b4;stroke-width:1px;}#gd34 .flowchart-link{stroke:#a9b5b4;fill:none;}#gd34 .edgeLabel{background-color:#11171a;text-align:center;}#gd34 .edgeLabel p{background-color:#11171a;}#gd34 .edgeLabel rect{opacity:0.5;background-color:#11171a;fill:#11171a;}#gd34 .labelBkg{background-color:rgba(17, 23, 26, 0.5);}#gd34 .cluster rect{fill:#11171a;stroke:#233035;stroke-width:1px;}#gd34 .cluster text{fill:#f4f4f5;}#gd34 .cluster span{color:#f4f4f5;}#gd34 div.mermaidTooltip{position:absolute;text-align:center;max-width:200px;padding:2px;font-family:Recursive,ui-sans-serif,system-ui,-apple-system,sans-serif;font-size:12px;background:#1a232f;border:1px solid hsl(214.2857142857, 0%, 4.3137254902%);border-radius:2px;pointer-events:none;z-index:100;}#gd34 .flowchartTitleText{text-anchor:middle;font-size:18px;fill:#f4f4f5;}#gd34 rect.text{fill:none;stroke-width:0;}#gd34 .icon-shape,#gd34 .image-shape{background-color:#11171a;text-align:center;}#gd34 .icon-shape p,#gd34 .image-shape p{background-color:#11171a;padding:2px;}#gd34 .icon-shape .label rect,#gd34 .image-shape .label rect{opacity:0.5;background-color:#11171a;fill:#11171a;}#gd34 .label-icon{display:inline-block;height:1em;overflow:visible;vertical-align:-0.125em;}#gd34 .node .label-icon path{fill:currentColor;stroke:revert;stroke-width:revert;}#gd34 .node .neo-node{stroke:#3a4a52;}#gd34 [data-look=&quot;neo&quot;].node rect,#gd34 [data-look=&quot;neo&quot;].cluster rect,#gd34 [data-look=&quot;neo&quot;].node polygon{stroke:url(#gd34-gradient);filter:drop-shadow( 1px 2px 2px rgba(185,185,185,1));}#gd34 [data-look=&quot;neo&quot;].swimlane.cluster rect{filter:none;}#gd34 [data-look=&quot;neo&quot;].node path{stroke:url(#gd34-gradient);stroke-width:1px;}#gd34 [data-look=&quot;neo&quot;].node .outer-path{filter:drop-shadow( 1px 2px 2px rgba(185,185,185,1));}#gd34 [data-look=&quot;neo&quot;].node .neo-line path{stroke:#3a4a52;filter:none;}#gd34 [data-look=&quot;neo&quot;].node circle{stroke:url(#gd34-gradient);filter:drop-shadow( 1px 2px 2px rgba(185,185,185,1));}#gd34 [data-look=&quot;neo&quot;].node circle .state-start{fill:#000000;}#gd34 [data-look=&quot;neo&quot;].icon-shape .icon{fill:url(#gd34-gradient);filter:drop-shadow( 1px 2px 2px rgba(185,185,185,1));}#gd34 [data-look=&quot;neo&quot;].icon-shape .icon-neo path{stroke:url(#gd34-gradient);filter:drop-shadow( 1px 2px 2px rgba(185,185,185,1));}#gd34 :root{--mermaid-font-family:Recursive,ui-sans-serif,system-ui,-apple-system,sans-serif;}</style><g><marker id="gd34_flowchart-v2-pointEnd" class="marker flowchart-v2" viewBox="0 0 10 10" refX="5" refY="5" markerUnits="userSpaceOnUse" markerWidth="8" markerHeight="8" orient="auto"><path d="M 0 0 L 10 5 L 0 10 z" class="arrowMarkerPath" style="stroke-width:1;stroke-dasharray:1,0"></path></marker><marker id="gd34_flowchart-v2-pointStart" class="marker flowchart-v2" viewBox="0 0 10 10" refX="4.5" refY="5" markerUnits="userSpaceOnUse" markerWidth="8" markerHeight="8" orient="auto"><path d="M 0 5 L 10 10 L 10 0 z" class="arrowMarkerPath" style="stroke-width:1;stroke-dasharray:1,0"></path></marker><marker id="gd34_flowchart-v2-pointEnd-margin" class="marker flowchart-v2" viewBox="0 0 11.5 14" refX="11.5" refY="7" markerUnits="userSpaceOnUse" markerWidth="10.5" markerHeight="14" orient="auto"><path d="M 0 0 L 11.5 7 L 0 14 z" class="arrowMarkerPath" style="stroke-width:0;stroke-dasharray:1,0"></path></marker><marker id="gd34_flowchart-v2-pointStart-margin" class="marker flowchart-v2" viewBox="0 0 11.5 14" refX="1" refY="7" markerUnits="userSpaceOnUse" markerWidth="11.5" markerHeight="14" orient="auto"><polygon points="0,7 11.5,14 11.5,0" class="arrowMarkerPath" style="stroke-width:0;stroke-dasharray:1,0"></polygon></marker><marker id="gd34_flowchart-v2-circleEnd" class="marker flowchart-v2" viewBox="0 0 10 10" refX="11" refY="5" markerUnits="userSpaceOnUse" markerWidth="11" markerHeight="11" orient="auto"><circle cx="5" cy="5" r="5" class="arrowMarkerPath" style="stroke-width:1;stroke-dasharray:1,0"></circle></marker><marker id="gd34_flowchart-v2-circleStart" class="marker flowchart-v2" viewBox="0 0 10 10" refX="-1" refY="5" markerUnits="userSpaceOnUse" markerWidth="11" markerHeight="11" orient="auto"><circle cx="5" cy="5" r="5" class="arrowMarkerPath" style="stroke-width:1;stroke-dasharray:1,0"></circle></marker><marker id="gd34_flowchart-v2-circleEnd-margin" class="marker flowchart-v2" viewBox="0 0 10 10" refY="5" refX="12.25" markerUnits="userSpaceOnUse" markerWidth="14" markerHeight="14" orient="auto"><circle cx="5" cy="5" r="5" class="arrowMarkerPath" style="stroke-width:0;stroke-dasharray:1,0"></circle></marker><marker id="gd34_flowchart-v2-circleStart-margin" class="marker flowchart-v2" viewBox="0 0 10 10" refX="-2" refY="5" markerUnits="userSpaceOnUse" markerWidth="14" markerHeight="14" orient="auto"><circle cx="5" cy="5" r="5" class="arrowMarkerPath" style="stroke-width:0;stroke-dasharray:1,0"></circle></marker><marker id="gd34_flowchart-v2-crossEnd" class="marker cross flowchart-v2" viewBox="0 0 11 11" refX="12" refY="5.2" markerUnits="userSpaceOnUse" markerWidth="11" markerHeight="11" orient="auto"><path d="M 1,1 l 9,9 M 10,1 l -9,9" class="arrowMarkerPath" style="stroke-width:2;stroke-dasharray:1,0"></path></marker><marker id="gd34_flowchart-v2-crossStart" class="marker cross flowchart-v2" viewBox="0 0 11 11" refX="-1" refY="5.2" markerUnits="userSpaceOnUse" markerWidth="11" markerHeight="11" orient="auto"><path d="M 1,1 l 9,9 M 10,1 l -9,9" class="arrowMarkerPath" style="stroke-width:2;stroke-dasharray:1,0"></path></marker><marker id="gd34_flowchart-v2-crossEnd-margin" class="marker cross flowchart-v2" viewBox="0 0 15 15" refX="17.7" refY="7.5" markerUnits="userSpaceOnUse" markerWidth="12" markerHeight="12" orient="auto"><path d="M 1,1 L 14,14 M 1,14 L 14,1" class="arrowMarkerPath" style="stroke-width:2.5"></path></marker><marker id="gd34_flowchart-v2-crossStart-margin" class="marker cross flowchart-v2" viewBox="0 0 15 15" refX="-3.5" refY="7.5" markerUnits="userSpaceOnUse" markerWidth="12" markerHeight="12" orient="auto"><path d="M 1,1 L 14,14 M 1,14 L 14,1" class="arrowMarkerPath" style="stroke-width:2.5;stroke-dasharray:1,0"></path></marker><g class="root"><g class="clusters"></g><g class="edgePaths"><path d="M361.156,126.625L361.156,130.792C361.156,134.958,361.156,143.292,361.156,151.625C361.156,159.958,361.156,168.292,361.156,172.458L361.156,176.625" id="gd34-L_PLAIN_ARROW_0" class=" edge-thickness-invisible edge-pattern-solid" data-edge="true" data-et="edge" data-id="L_PLAIN_ARROW_0" data-points="W3sieCI6MzYxLjE1NjI1LCJ5IjoxMjYuNjI1fSx7IngiOjM2MS4xNTYyNSwieSI6MTUxLjYyNX0seyJ4IjozNjEuMTU2MjUsInkiOjE3Ni42MjV9XQ==" data-look="classic" style=";"></path><path d="M361.156,295.25L361.156,299.417C361.156,303.583,361.156,311.917,361.156,320.25C361.156,328.583,361.156,336.917,361.156,341.083L361.156,345.25" id="gd34-L_ARROW_PANDAS_0" class=" edge-thickness-invisible edge-pattern-solid" data-edge="true" data-et="edge" data-id="L_ARROW_PANDAS_0" data-points="W3sieCI6MzYxLjE1NjI1LCJ5IjoyOTUuMjV9LHsieCI6MzYxLjE1NjI1LCJ5IjozMjAuMjV9LHsieCI6MzYxLjE1NjI1LCJ5IjozNDUuMjV9XQ==" data-look="classic" style=";"></path></g><g class="edgeLabels"><g class="edgeLabel"><g class="label" data-id="L_PLAIN_ARROW_0" transform="translate(0, -10.4609375)"><text y="-10.1" text-anchor="middle"><tspan class="text-outer-tspan row" x="0" y="-0.1em" dy="1.1em" text-anchor="middle"></tspan></text></g></g><g><rect class="background" style="stroke: none"></rect></g><g class="edgeLabel"><g class="label" data-id="L_ARROW_PANDAS_0" transform="translate(0, -10.4609375)"><text y="-10.1" text-anchor="middle"><tspan class="text-outer-tspan row" x="0" y="-0.1em" dy="1.1em" text-anchor="middle"></tspan></text></g></g><g><rect class="background" style="stroke: none"></rect></g></g><g class="nodes"><g class="root" transform="translate(0, 337.25)"><g class="clusters"><g class="cluster " id="gd34-PANDAS" data-look="classic"><rect x="8" y="8" width="706.3125" height="136.225"></rect><g class="cluster-label " transform="translate(262.21484375, 8)"><g><rect class="background" style="stroke: none"></rect><text y="-10.1"><tspan class="text-outer-tspan row" x="0" y="-0.1em" dy="1.1em"><tspan font-style="normal" class="text-inner-tspan" font-weight="normal">pandas</tspan><tspan font-style="normal" class="text-inner-tspan" font-weight="normal"> UDF</tspan><tspan font-style="normal" class="text-inner-tspan" font-weight="normal"> (vectorized)</tspan></tspan></text></g></g></g></g><g class="edgePaths"><path d="M134.969,76.113L155.96,76.113C176.951,76.113,218.932,76.113,260.247,76.113C301.563,76.113,342.211,76.113,362.535,76.113L382.859,76.113" id="gd34-L_JVM3_PY3_0" class=" edge-thickness-normal edge-pattern-solid edge-thickness-normal edge-pattern-solid flowchart-link" data-edge="true" data-et="edge" data-id="L_JVM3_PY3_0" data-points="W3sieCI6MTM0Ljk2ODc1LCJ5Ijo3Ni4xMTI1fSx7IngiOjI2MC45MTQwNjI1LCJ5Ijo3Ni4xMTI1fSx7IngiOjM4Ni44NTkzNzUsInkiOjc2LjExMjV9XQ==" data-look="classic" marker-end="url(#gd34_flowchart-v2-pointEnd)" style=";"></path></g><g class="edgeLabels"><g class="edgeLabel" transform="translate(260.9140625, 76.1125)"><g class="label" data-id="L_JVM3_PY3_0" transform="translate(0, -19.2609375)"><g><rect class="background" x="-88.4453125" y="-0.8515625" width="176.890625" height="40.225"></rect><text y="-10.1" text-anchor="middle"><tspan class="text-outer-tspan row" x="0" y="-0.1em" dy="1.1em" text-anchor="middle"><tspan font-style="normal" class="text-inner-tspan" font-weight="normal">whole</tspan><tspan font-style="normal" class="text-inner-tspan" font-weight="normal"> Arrow</tspan><tspan font-style="normal" class="text-inner-tspan" font-weight="normal"> batches,</tspan></tspan><tspan class="text-outer-tspan row" x="0" y="1em" dy="1.1em" text-anchor="middle"><tspan font-style="normal" class="text-inner-tspan" font-weight="normal">no</tspan><tspan font-style="normal" class="text-inner-tspan" font-weight="normal"> per-row</tspan><tspan font-style="normal" class="text-inner-tspan" font-weight="normal"> handoff</tspan></tspan></text></g></g></g></g><g class="nodes"><g class="node default " id="gd34-flowchart-JVM3-4" data-look="classic" transform="translate(90.234375, 76.1125)"><rect class="basic label-container" x="-44.734375" y="-24.3125" width="89.46875" height="48.625"></rect><g class="label" transform="translate(0, -9.3125)"><rect></rect><g><rect class="background" style="stroke: none"></rect><text y="-10.1"><tspan class="text-outer-tspan row" x="0" y="-0.1em" dy="1.1em"><tspan font-style="normal" class="text-inner-tspan" font-weight="normal">JVM</tspan></tspan></text></g></g></g><g class="node default " id="gd34-flowchart-PY3-5" data-look="classic" transform="translate(531.8359375, 76.1125)"><rect class="basic label-container" x="-144.9765625" y="-33.1125" width="289.953125" height="66.225"></rect><g class="label" transform="translate(0, -18.1125)"><rect></rect><g><rect class="background" style="stroke: none"></rect><text y="-10.1"><tspan class="text-outer-tspan row" x="0" y="-0.1em" dy="1.1em"><tspan font-style="normal" class="text-inner-tspan" font-weight="normal">Python</tspan></tspan><tspan class="text-outer-tspan row" x="0" y="1em" dy="1.1em"><tspan font-style="normal" class="text-inner-tspan" font-weight="normal">(pandas</tspan><tspan font-style="normal" class="text-inner-tspan" font-weight="normal"> Series</tspan><tspan font-style="normal" class="text-inner-tspan" font-weight="normal"> /</tspan><tspan font-style="normal" class="text-inner-tspan" font-weight="normal"> DataFrame)</tspan></tspan></text></g></g></g></g></g><g class="root" transform="translate(99.125, 168.625)"><g class="clusters"><g class="cluster " id="gd34-ARROW" data-look="classic"><rect x="8" y="8" width="508.0625" height="118.625"></rect><g class="cluster-label " transform="translate(146.67578125, 8)"><g><rect class="background" style="stroke: none"></rect><text y="-10.1"><tspan class="text-outer-tspan row" x="0" y="-0.1em" dy="1.1em"><tspan font-style="normal" class="text-inner-tspan" font-weight="normal">Arrow-optimized</tspan><tspan font-style="normal" class="text-inner-tspan" font-weight="normal"> Python</tspan><tspan font-style="normal" class="text-inner-tspan" font-weight="normal"> UDF</tspan></tspan></text></g></g></g></g><g class="edgePaths"><path d="M134.969,67.313L153.98,67.313C172.991,67.313,211.013,67.313,248.368,67.313C285.724,67.313,322.413,67.313,340.757,67.313L359.102,67.313" id="gd34-L_JVM2_PY2_0" class=" edge-thickness-normal edge-pattern-solid edge-thickness-normal edge-pattern-solid flowchart-link" data-edge="true" data-et="edge" data-id="L_JVM2_PY2_0" data-points="W3sieCI6MTM0Ljk2ODc1LCJ5Ijo2Ny4zMTI1fSx7IngiOjI0OS4wMzUxNTYyNSwieSI6NjcuMzEyNX0seyJ4IjozNjMuMTAxNTYyNSwieSI6NjcuMzEyNX1d" data-look="classic" marker-end="url(#gd34_flowchart-v2-pointEnd)" style=";"></path></g><g class="edgeLabels"><g class="edgeLabel" transform="translate(249.03515625, 67.3125)"><g class="label" data-id="L_JVM2_PY2_0" transform="translate(0, -19.2609375)"><g><rect class="background" x="-76.56640625" y="-0.8515625" width="153.1328125" height="40.225"></rect><text y="-10.1" text-anchor="middle"><tspan class="text-outer-tspan row" x="0" y="-0.1em" dy="1.1em" text-anchor="middle"><tspan font-style="normal" class="text-inner-tspan" font-weight="normal">Arrow</tspan><tspan font-style="normal" class="text-inner-tspan" font-weight="normal"> transfer,</tspan></tspan><tspan class="text-outer-tspan row" x="0" y="1em" dy="1.1em" text-anchor="middle"><tspan font-style="normal" class="text-inner-tspan" font-weight="normal">still</tspan><tspan font-style="normal" class="text-inner-tspan" font-weight="normal"> row</tspan><tspan font-style="normal" class="text-inner-tspan" font-weight="normal"> semantics</tspan></tspan></text></g></g></g></g><g class="nodes"><g class="node default " id="gd34-flowchart-JVM2-2" data-look="classic" transform="translate(90.234375, 67.3125)"><rect class="basic label-container" x="-44.734375" y="-24.3125" width="89.46875" height="48.625"></rect><g class="label" transform="translate(0, -9.3125)"><rect></rect><g><rect class="background" style="stroke: none"></rect><text y="-10.1"><tspan class="text-outer-tspan row" x="0" y="-0.1em" dy="1.1em"><tspan font-style="normal" class="text-inner-tspan" font-weight="normal">JVM</tspan></tspan></text></g></g></g><g class="node default " id="gd34-flowchart-PY2-3" data-look="classic" transform="translate(420.83203125, 67.3125)"><rect class="basic label-container" x="-57.73046875" y="-24.3125" width="115.4609375" height="48.625"></rect><g class="label" transform="translate(0, -9.3125)"><rect></rect><g><rect class="background" style="stroke: none"></rect><text y="-10.1"><tspan class="text-outer-tspan row" x="0" y="-0.1em" dy="1.1em"><tspan font-style="normal" class="text-inner-tspan" font-weight="normal">Python</tspan></tspan></text></g></g></g></g></g><g class="root" transform="translate(87.22265625, 0)"><g class="clusters"><g class="cluster " id="gd34-PLAIN" data-look="classic"><rect x="8" y="8" width="531.8671875" height="118.625"></rect><g class="cluster-label " transform="translate(205.2578125, 8)"><g><rect class="background" style="stroke: none"></rect><text y="-10.1"><tspan class="text-outer-tspan row" x="0" y="-0.1em" dy="1.1em"><tspan font-style="normal" class="text-inner-tspan" font-weight="normal">Plain</tspan><tspan font-style="normal" class="text-inner-tspan" font-weight="normal"> Python</tspan><tspan font-style="normal" class="text-inner-tspan" font-weight="normal"> UDF</tspan></tspan></text></g></g></g></g><g class="edgePaths"><path d="M134.969,67.313L155.964,67.313C176.958,67.313,218.948,67.313,260.271,67.313C301.594,67.313,342.25,67.313,362.578,67.313L382.906,67.313" id="gd34-L_JVM1_PY1_0" class=" edge-thickness-normal edge-pattern-solid edge-thickness-normal edge-pattern-solid flowchart-link" data-edge="true" data-et="edge" data-id="L_JVM1_PY1_0" data-points="W3sieCI6MTM0Ljk2ODc1LCJ5Ijo2Ny4zMTI1fSx7IngiOjI2MC45Mzc1LCJ5Ijo2Ny4zMTI1fSx7IngiOjM4Ni45MDYyNSwieSI6NjcuMzEyNX1d" data-look="classic" marker-end="url(#gd34_flowchart-v2-pointEnd)" style=";"></path></g><g class="edgeLabels"><g class="edgeLabel" transform="translate(260.9375, 67.3125)"><g class="label" data-id="L_JVM1_PY1_0" transform="translate(0, -28.0609375)"><g><rect class="background" x="-88.46875" y="-0.8515625" width="176.9375" height="57.825"></rect><text y="-10.1" text-anchor="middle"><tspan class="text-outer-tspan row" x="0" y="-0.1em" dy="1.1em" text-anchor="middle"><tspan font-style="normal" class="text-inner-tspan" font-weight="normal">pickle,</tspan><tspan font-style="normal" class="text-inner-tspan" font-weight="normal"> row-at-a-time</tspan></tspan><tspan class="text-outer-tspan row" x="0" y="1em" dy="1.1em" text-anchor="middle"><tspan font-style="normal" class="text-inner-tspan" font-weight="normal">(serialization</tspan><tspan font-style="normal" class="text-inner-tspan" font-weight="normal"> cost</tspan><tspan font-style="normal" class="text-inner-tspan" font-weight="normal"> per</tspan></tspan><tspan class="text-outer-tspan row" x="0" y="2.1em" dy="1.1em" text-anchor="middle"><tspan font-style="normal" class="text-inner-tspan" font-weight="normal">row)</tspan></tspan></text></g></g></g></g><g class="nodes"><g class="node default " id="gd34-flowchart-JVM1-0" data-look="classic" transform="translate(90.234375, 67.3125)"><rect class="basic label-container" x="-44.734375" y="-24.3125" width="89.46875" height="48.625"></rect><g class="label" transform="translate(0, -9.3125)"><rect></rect><g><rect class="background" style="stroke: none"></rect><text y="-10.1"><tspan class="text-outer-tspan row" x="0" y="-0.1em" dy="1.1em"><tspan font-style="normal" class="text-inner-tspan" font-weight="normal">JVM</tspan></tspan></text></g></g></g><g class="node default " id="gd34-flowchart-PY1-1" data-look="classic" transform="translate(444.63671875, 67.3125)"><rect class="basic label-container" x="-57.73046875" y="-24.3125" width="115.4609375" height="48.625"></rect><g class="label" transform="translate(0, -9.3125)"><rect></rect><g><rect class="background" style="stroke: none"></rect><text y="-10.1"><tspan class="text-outer-tspan row" x="0" y="-0.1em" dy="1.1em"><tspan font-style="normal" class="text-inner-tspan" font-weight="normal">Python</tspan></tspan></text></g></g></g></g></g></g></g></g><defs><filter id="gd34-drop-shadow" height="130%" width="130%"><feDropShadow dx="4" dy="4" stdDeviation="0" flood-opacity="0.06" flood-color="#000000"></feDropShadow></filter></defs><defs><filter id="gd34-drop-shadow-small" height="150%" width="150%"><feDropShadow dx="2" dy="2" stdDeviation="0" flood-opacity="0.06" flood-color="#000000"></feDropShadow></filter></defs><linearGradient id="gd34-gradient" gradientUnits="objectBoundingBox" x1="0%" y1="0%" x2="100%" y2="0%"><stop offset="0%" stop-color="#3a4a52" stop-opacity="1"></stop><stop offset="100%" stop-color="hsl(200, 0%, 0%)" stop-opacity="1"></stop></linearGradient></svg>
@@ -0,0 +1 @@
1
+ import{_ as t,o as a,c as s,a5 as o}from"./chunks/framework.DSg0KOwT.js";const g=JSON.parse('{"title":"Alternative ways to get the logs","description":"","frontmatter":{},"headers":[],"relativePath":"user-guide/alternative-log-retrieval.md","filePath":"user-guide/alternative-log-retrieval.md"}'),i={name:"user-guide/alternative-log-retrieval.md"};function r(l,e,h,n,d,p){return a(),s("div",null,[...e[0]||(e[0]=[o('<h1 id="alternative-ways-to-get-the-logs" tabindex="-1">Alternative ways to get the logs <a class="header-anchor" href="#alternative-ways-to-get-the-logs" aria-label="Permalink to &quot;Alternative ways to get the logs&quot;">​</a></h1><p><code>--shs-base-url</code> is the direct route: point SparkForensics at the History Server and it fetches the event log itself. This page covers what to do when that direct route isn&#39;t available.</p><h2 id="when-you-need-this" tabindex="-1">When you need this <a class="header-anchor" href="#when-you-need-this" aria-label="Permalink to &quot;When you need this&quot;">​</a></h2><p>The most common case is a Spark History Server (SHS) reachable only through an SSH bastion or jump host (the recipe below is the same regardless of what&#39;s issuing the SSH session), with the event logs themselves living on Kerberized HDFS and no direct HTTP path from your machine to SHS.</p><h2 id="why-shs-base-url-won-t-work-here" tabindex="-1">Why <code>--shs-base-url</code> won&#39;t work here <a class="header-anchor" href="#why-shs-base-url-won-t-work-here" aria-label="Permalink to &quot;Why `--shs-base-url` won&#39;t work here&quot;">​</a></h2><p>The CLI&#39;s <code>--shs-base-url</code>/<code>--app-id</code> flags and the MCP tools&#39; <code>{ shsBaseUrl, appId }</code> source both make a direct HTTP request from wherever SparkForensics runs to the History Server&#39;s REST API. If that address is only reachable from inside a bastion, that request never lands: there&#39;s no <code>--via-ssh</code> flag or other way to route it through an SSH session.</p><h2 id="the-recipe" tabindex="-1">The recipe <a class="header-anchor" href="#the-recipe" aria-label="Permalink to &quot;The recipe&quot;">​</a></h2><ol><li><p>SSH into the edge node (through the bastion, or whatever gets you there).</p></li><li><p>Pull the event log off HDFS onto local disk on the edge node:</p><div class="language-bash vp-adaptive-theme"><button title="Copy Code" class="copy"></button><span class="lang">bash</span><pre class="shiki shiki-themes github-light github-dark vp-code" tabindex="0"><code><span class="line"><span style="--shiki-light:#6F42C1;--shiki-dark:#B392F0;">hdfs</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> dfs</span><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> -get</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> /path/to/spark-events/application_XXXX_XXXX</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> /tmp/application_XXXX_XXXX</span></span></code></pre></div></li><li><p>Copy that file back to your own machine through the same bastion, e.g. with <code>scp</code> or <code>sftp</code>:</p><div class="language-bash vp-adaptive-theme"><button title="Copy Code" class="copy"></button><span class="lang">bash</span><pre class="shiki shiki-themes github-light github-dark vp-code" tabindex="0"><code><span class="line"><span style="--shiki-light:#6F42C1;--shiki-dark:#B392F0;">scp</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> edge-node:/tmp/application_XXXX_XXXX</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> ./application_XXXX_XXXX</span></span></code></pre></div></li><li><p>Point SparkForensics at the local copy instead of a <code>shsBaseUrl</code>:</p><ul><li>Browser: drop the file onto the landing page, same as any other run (see <a href="./getting-started.html">Getting started</a>).</li><li>CLI: <code>npx sparkforensics-analyze ./application_XXXX_XXXX</code> (full flag list in <a href="./getting-started.html#ci-and-automation">Getting started</a>).</li><li>MCP: call a tool with <code>{ &quot;source&quot;: { &quot;path&quot;: &quot;./application_XXXX_XXXX&quot; } }</code> (tool reference in <a href="./mcp-tools.html">MCP tools reference</a>).</li></ul></li></ol><p>The file is the app&#39;s native input format either way: a single event-log file, optionally <code>.gz</code>/<code>.zstd</code>/<code>.lz4</code>/<code>.snappy</code>-compressed, or a rolling <code>eventlog_v2_*</code> directory pulled the same way. No conversion step needed.</p>',9)])])}const u=t(i,[["render",r]]);export{g as __pageData,u as default};
@@ -0,0 +1 @@
1
+ import{_ as t,o as a,c as s,a5 as o}from"./chunks/framework.DSg0KOwT.js";const g=JSON.parse('{"title":"Alternative ways to get the logs","description":"","frontmatter":{},"headers":[],"relativePath":"user-guide/alternative-log-retrieval.md","filePath":"user-guide/alternative-log-retrieval.md"}'),i={name:"user-guide/alternative-log-retrieval.md"};function r(l,e,h,n,d,p){return a(),s("div",null,[...e[0]||(e[0]=[o("",9)])])}const u=t(i,[["render",r]]);export{g as __pageData,u as default};
@@ -0,0 +1,3 @@
1
+ import{_ as a,o as t,c as o,a5 as s}from"./chunks/framework.DSg0KOwT.js";const g=JSON.parse('{"title":"Getting started","description":"","frontmatter":{},"headers":[],"relativePath":"user-guide/getting-started.md","filePath":"user-guide/getting-started.md"}'),r={name:"user-guide/getting-started.md"};function n(i,e,d,l,h,c){return t(),o("div",null,[...e[0]||(e[0]=[s(`<h1 id="getting-started" tabindex="-1">Getting started <a class="header-anchor" href="#getting-started" aria-label="Permalink to &quot;Getting started&quot;">​</a></h1><p>SparkForensics reads a Spark History Server event log and turns it into a dashboard of flagged bottlenecks. No install and no account: open the app in a browser and drop a log in.</p><h2 id="load-a-run" tabindex="-1">Load a run <a class="header-anchor" href="#load-a-run" aria-label="Permalink to &quot;Load a run&quot;">​</a></h2><p>Drop a log file onto the landing page, or click <strong>Choose file</strong> to pick one. Parsing runs in a background worker, off the browser&#39;s main thread, so a multi-hundred-megabyte event stream doesn&#39;t freeze the tab.</p><p>No log of your own yet? Click <strong>Try a sample run</strong> on the landing page to load a bundled example run and see a populated dashboard right away.</p><p>The app takes a newline-delimited JSON event log (one JSON event per line, the format Spark writes to <code>spark.eventLog.dir</code>), either plain or gzip/Zstandard/LZ4/Snappy-compressed.</p><p>Click <strong>Other sources</strong> on the landing page for two more ways in:</p><ul><li><strong>Choose rolling-log folder</strong>, for an <code>eventlog_v2_*</code> rolling directory. Drop the folder or point the picker at it; the app reassembles its parts in order before parsing.</li><li><strong>Fetch from Spark History Server</strong>, to pull an application straight from a reachable History Server. This needs <strong>local-server mode</strong> (see below) and an application ID in one of the forms Spark itself uses: <code>application_&lt;timestamp&gt;_&lt;id&gt;</code>, <code>local-&lt;timestamp&gt;</code>, <code>app-&lt;id&gt;</code>, <code>spark-&lt;id&gt;</code>, or <code>driver-&lt;id&gt;</code>. If the fetch fails, the panel names the problem (server unreachable, application not found, local server not running, and so on) and suggests what to try next.</li></ul><p>Can&#39;t reach the History Server directly (it&#39;s only reachable through an SSH bastion)? See <a href="./alternative-log-retrieval.html">Alternative ways to get the logs</a>.</p><p>A History Server export needs one manual step. <code>GET /api/v1/applications/&lt;appId&gt;/logs</code> hands back a zip, and the drop zone does not unwrap zip containers. Extract the <code>.zstd</code> event log from that zip yourself and drop the extracted file in. It needs no further decompressing.</p><p>Files you have already loaded stay listed under <strong>Recent files</strong> on the landing page.</p><h3 id="local-server-mode" tabindex="-1">Local-server mode <a class="header-anchor" href="#local-server-mode" aria-label="Permalink to &quot;Local-server mode&quot;">​</a></h3><p>The plain browser app can&#39;t fetch from a History Server itself: the browser&#39;s CORS policy blocks a cross-origin request like that, and there&#39;s no server on the other end to proxy it. Local-server mode adds that server. Run it with:</p><div class="language-sh vp-adaptive-theme"><button title="Copy Code" class="copy"></button><span class="lang">sh</span><pre class="shiki shiki-themes github-light github-dark vp-code" tabindex="0"><code><span class="line"><span style="--shiki-light:#6F42C1;--shiki-dark:#B392F0;">npx</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> sparkforensics-server</span></span></code></pre></div><p>It listens on <code>http://127.0.0.1:4173</code> by default and binds to localhost only. Open that URL instead of the static build, and the <strong>Fetch from Spark History Server</strong> panel works because requests now go server-to-server. Full setup and its port/environment-variable overrides are in the <a href="https://github.com/shuffle-works/sparkforensics#readme" target="_blank" rel="noreferrer">project README</a>, which also covers what changes for a static (no-server) deploy.</p><h2 id="reading-the-dashboard" tabindex="-1">Reading the dashboard <a class="header-anchor" href="#reading-the-dashboard" aria-label="Permalink to &quot;Reading the dashboard&quot;">​</a></h2><p>When parsing finishes, the dashboard shows a board of widgets, each in its own card with a title. Every widget that flags a problem uses the same convention: a colored impact dot (critical / warning / info) and an ALL-CAPS tag for the bottleneck category (<code>SKEW</code>, <code>SPILL</code>, <code>GC</code>, and so on: see <a href="./understanding-findings.html">Understanding findings</a>). Widgets list every affected stage, not just the worst one. The dot&#39;s color tracks how much run time the finding could save you rather than how unusual the metric looks, so a small-looking anomaly with a big payoff can outrank a dramatic one that would barely move your run time.</p><p>A run scorecard (wall-clock, efficiency, wastage) always shows at the top. Below it, two tabs split the rest of the board:</p><ol><li><strong>Findings</strong>: every flagged finding and its detail widget, grouped by impact band (Critical, Warning, Info). Within a band, a recommendation row for a bottleneck type collapses into a summary row when it fires more than once; clicking the summary row expands its full list, and clicking any single row or widget jumps straight to that finding. Below the impact-band groups, memory and core-usage utilization always show, even on a clean run. Widgets that found nothing fold away into a &quot;Clean checks&quot; disclosure. This is the tab you land on.</li><li><strong>Full app report</strong>: the wall-clock and executor timelines, the stage table, and the reference-only cards.</li></ol><p>Click a finding&#39;s documentation link (or the topbar&#39;s <strong>Docs</strong> button) to open the reference material in a slide-in panel beside the dashboard: the dashboard stays visible and interactive, so you can check a metric against the reference without losing your place.</p><h3 id="advanced-view" tabindex="-1">Advanced view <a class="header-anchor" href="#advanced-view" aria-label="Permalink to &quot;Advanced view&quot;">​</a></h3><p>The topbar has an <strong>Advanced view</strong> toggle. It&#39;s off by default, which keeps each widget to the finding itself and what to do about it. Turn it on to also show confidence levels, supporting evidence, and documentation links for each finding, plus a few extra table columns. Your choice is remembered across runs.</p><h3 id="the-rest-of-the-topbar" tabindex="-1">The rest of the topbar <a class="header-anchor" href="#the-rest-of-the-topbar" aria-label="Permalink to &quot;The rest of the topbar&quot;">​</a></h3><p>Once a run is loaded, the topbar also carries a few more controls. <strong>New analysis</strong> goes back to the landing page to load another run. <strong>Plan graph</strong> opens an interactive node-and-edge view of the run&#39;s SQL execution plan, filterable down to I/O operators (scan, exchange), a broader &quot;basic&quot; set, or every operator. <strong>Export evidence</strong> downloads the current run&#39;s findings as a portable Markdown or JSON report, the same shape the CLI and MCP tools produce; turn on <strong>Redact identifiers</strong> first if the report is headed outside the environment that produced it, since that pseudonymizes the app id and any host/IP tokens. <strong>Keyboard shortcuts</strong> (press <code>?</code> from anywhere) lists every shortcut. Rounding it out: a <strong>Docs</strong> link and a theme toggle.</p><h2 id="compare-two-runs" tabindex="-1">Compare two runs <a class="header-anchor" href="#compare-two-runs" aria-label="Permalink to &quot;Compare two runs&quot;">​</a></h2><p>To compare a baseline run against a candidate, say to check whether a tuning change helped, see <a href="./run-comparison.html">Run comparison mode</a>.</p><h2 id="ci-and-automation" tabindex="-1">CI and automation <a class="header-anchor" href="#ci-and-automation" aria-label="Permalink to &quot;CI and automation&quot;">​</a></h2><p>In a pipeline or a script, use the <code>analyze</code> CLI instead of the browser dashboard. It parses the same event logs and exits non-zero when a run crosses a threshold you set, so it can gate a build:</p><div class="language-sh vp-adaptive-theme"><button title="Copy Code" class="copy"></button><span class="lang">sh</span><pre class="shiki shiki-themes github-light github-dark vp-code" tabindex="0"><code><span class="line"><span style="--shiki-light:#6F42C1;--shiki-dark:#B392F0;">npx</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;"> sparkforensics-analyze</span><span style="--shiki-light:#D73A49;--shiki-dark:#F97583;"> &lt;</span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;">file</span><span style="--shiki-light:#D73A49;--shiki-dark:#F97583;">|</span><span style="--shiki-light:#6F42C1;--shiki-dark:#B392F0;">dir</span><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;">&gt; [--max-runtime </span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;">ms]</span><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;"> [--max-skew </span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;">ratio]</span><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> \\</span></span>
2
+ <span class="line"><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;"> [--max-spill </span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;">gb]</span><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;"> [--max-failed-task-rate </span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;">pct]</span><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;"> [--min-efficiency </span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;">pct]</span><span style="--shiki-light:#005CC5;--shiki-dark:#79B8FF;"> \\</span></span>
3
+ <span class="line"><span style="--shiki-light:#24292E;--shiki-dark:#E1E4E8;"> [--out </span><span style="--shiki-light:#032F62;--shiki-dark:#9ECBFF;">path]</span></span></code></pre></div><p>Point it at a single event log or a directory of them. Pass <code>--out</code> to also write the findings to a file.</p><p>Pass <code>--export-html &lt;dir&gt;</code> to write a self-contained HTML dashboard for the run into <code>&lt;dir&gt;</code> (which must not already exist or must be empty). Open <code>&lt;dir&gt;/index.html</code> directly in a browser over <code>file://</code>, with no server, to get the same interactive dashboard offline. This makes it easy to archive or share a run. <code>--redact</code> applies to the exported report too.</p><p>The CLI also supports fetching a run directly from a reachable Spark History Server (<code>--shs-base-url</code>/<code>--app-id</code>/<code>--attempt-id</code>) instead of a local file, comparing a candidate run against a baseline with regression gating (<code>--baseline</code>/<code>--max-regression-pct</code>/<code>--regression-metric</code>/ <code>--fail-on-introduced</code>), redacting the app id and any host/IP tokens before sharing output (<code>--redact</code>), and narrowing the findings to certain impact bands, types, or a stage (<code>--impact</code>/<code>--type</code>/<code>--stage</code>). Run it with <code>--help</code> for the full flag list.</p><p>Same caveat as above: if the History Server is only reachable through an SSH bastion, <code>--shs-base-url</code> can&#39;t reach it either: see <a href="./alternative-log-retrieval.html">Alternative ways to get the logs</a>.</p><p>Running in Airflow instead of a plain CI pipeline? See <a href="https://github.com/shuffle-works/sparkforensics-operator" target="_blank" rel="noreferrer">sparkforensics-operator</a>, an Airflow operator that wraps the CLI and acts on the result after each Spark job, so you don&#39;t have to wire up the call yourself.</p><p>Want an AI assistant to diagnose a run directly, without the dashboard or a CI gate? See <a href="./mcp-tools.html">MCP tools reference</a>.</p>`,35)])])}const u=a(r,[["render",n]]);export{g as __pageData,u as default};
@@ -0,0 +1 @@
1
+ import{_ as a,o as t,c as o,a5 as s}from"./chunks/framework.DSg0KOwT.js";const g=JSON.parse('{"title":"Getting started","description":"","frontmatter":{},"headers":[],"relativePath":"user-guide/getting-started.md","filePath":"user-guide/getting-started.md"}'),r={name:"user-guide/getting-started.md"};function n(i,e,d,l,h,c){return t(),o("div",null,[...e[0]||(e[0]=[s("",35)])])}const u=a(r,[["render",n]]);export{g as __pageData,u as default};