sparkforensics-cli 0.1.0 → 0.2.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (361) hide show
  1. package/README.md +6 -0
  2. package/bin/sparkforensics-analyze.mjs +113 -48
  3. package/export-template/docs/404.html +25 -0
  4. package/export-template/docs/assets/app.DQTZyGL1.js +1 -0
  5. package/export-template/docs/assets/aqe-loop.IwQSATHw.svg +1 -0
  6. package/export-template/docs/assets/aqe-loop.dark.DGbaxqJE.svg +1 -0
  7. package/export-template/docs/assets/broadcast-vs-shuffle.Db4WY1XK.svg +1 -0
  8. package/export-template/docs/assets/broadcast-vs-shuffle.dark.C7Bxs0mG.svg +1 -0
  9. package/export-template/docs/assets/cache-lifecycle.dark.B-hS7AgU.svg +1 -0
  10. package/export-template/docs/assets/cache-lifecycle.rEOVYQNU.svg +1 -0
  11. package/export-template/docs/assets/chunks/@localSearchIndexroot.DppnXnDE.js +1 -0
  12. package/export-template/docs/assets/chunks/VPLocalSearchBox.BkBIPFs6.js +9 -0
  13. package/export-template/docs/assets/chunks/duplicate-plan-subtree.dark.Cdp70QhV.js +1 -0
  14. package/export-template/docs/assets/chunks/framework.DSg0KOwT.js +20 -0
  15. package/export-template/docs/assets/chunks/retry-escalation-ladder.dark.DHipdJgZ.js +1 -0
  16. package/export-template/docs/assets/chunks/theme.DP0u1AUq.js +2 -0
  17. package/export-template/docs/assets/cold-start-timeline.DxC_Sc7w.svg +1 -0
  18. package/export-template/docs/assets/cold-start-timeline.dark.CZ17YcAG.svg +1 -0
  19. package/export-template/docs/assets/columnar-layout.PghGeOEA.svg +1 -0
  20. package/export-template/docs/assets/columnar-layout.dark.BVNlz0ff.svg +1 -0
  21. package/export-template/docs/assets/container-memory.DIO0AnIm.svg +1 -0
  22. package/export-template/docs/assets/container-memory.dark.CP-5zuCl.svg +1 -0
  23. package/export-template/docs/assets/contributor-guide_architecture_board-widgets.md.CWpj01WU.js +1 -0
  24. package/export-template/docs/assets/contributor-guide_architecture_board-widgets.md.CWpj01WU.lean.js +1 -0
  25. package/export-template/docs/assets/contributor-guide_architecture_detector-contract.md.CgzUsQ6W.js +1 -0
  26. package/export-template/docs/assets/contributor-guide_architecture_detector-contract.md.CgzUsQ6W.lean.js +1 -0
  27. package/export-template/docs/assets/contributor-guide_architecture_drill-down.md.BtPdlM7r.js +1 -0
  28. package/export-template/docs/assets/contributor-guide_architecture_drill-down.md.BtPdlM7r.lean.js +1 -0
  29. package/export-template/docs/assets/contributor-guide_architecture_impact-estimation.md.CooslVJt.js +1 -0
  30. package/export-template/docs/assets/contributor-guide_architecture_impact-estimation.md.CooslVJt.lean.js +1 -0
  31. package/export-template/docs/assets/contributor-guide_architecture_index.md.3TO9ic6w.js +1 -0
  32. package/export-template/docs/assets/contributor-guide_architecture_index.md.3TO9ic6w.lean.js +1 -0
  33. package/export-template/docs/assets/contributor-guide_architecture_overview.md.CehiRmGn.js +1 -0
  34. package/export-template/docs/assets/contributor-guide_architecture_overview.md.CehiRmGn.lean.js +1 -0
  35. package/export-template/docs/assets/contributor-guide_architecture_state-and-history.md.C-xxn0q7.js +1 -0
  36. package/export-template/docs/assets/contributor-guide_architecture_state-and-history.md.C-xxn0q7.lean.js +1 -0
  37. package/export-template/docs/assets/contributor-guide_architecture_widget-rendering.md.R27gQrgY.js +1 -0
  38. package/export-template/docs/assets/contributor-guide_architecture_widget-rendering.md.R27gQrgY.lean.js +1 -0
  39. package/export-template/docs/assets/contributor-guide_architecture_worker-protocol.md.IbnfNrV3.js +6 -0
  40. package/export-template/docs/assets/contributor-guide_architecture_worker-protocol.md.IbnfNrV3.lean.js +1 -0
  41. package/export-template/docs/assets/contributor-guide_contributing.md.CvRsdr6J.js +1 -0
  42. package/export-template/docs/assets/contributor-guide_contributing.md.CvRsdr6J.lean.js +1 -0
  43. package/export-template/docs/assets/contributor-guide_development-setup.md.DvAN_9mK.js +12 -0
  44. package/export-template/docs/assets/contributor-guide_development-setup.md.DvAN_9mK.lean.js +1 -0
  45. package/export-template/docs/assets/contributor-guide_testing.md.6rIKqSyY.js +1 -0
  46. package/export-template/docs/assets/contributor-guide_testing.md.6rIKqSyY.lean.js +1 -0
  47. package/export-template/docs/assets/dag-stages.DSz_S937.svg +1 -0
  48. package/export-template/docs/assets/dag-stages.dark.F72UzxH4.svg +1 -0
  49. package/export-template/docs/assets/driver-executor.D5pQ7YN1.svg +1 -0
  50. package/export-template/docs/assets/driver-executor.dark.BmX9cPvh.svg +1 -0
  51. package/export-template/docs/assets/duplicate-plan-subtree.B4cvN6fj.svg +1 -0
  52. package/export-template/docs/assets/duplicate-plan-subtree.dark.Dw8wS0Ag.svg +1 -0
  53. package/export-template/docs/assets/index.md.CHJVslga.js +1 -0
  54. package/export-template/docs/assets/index.md.CHJVslga.lean.js +1 -0
  55. package/export-template/docs/assets/inter-italic-cyrillic-ext.r48I6akx.woff2 +0 -0
  56. package/export-template/docs/assets/inter-italic-cyrillic.By2_1cv3.woff2 +0 -0
  57. package/export-template/docs/assets/inter-italic-greek-ext.1u6EdAuj.woff2 +0 -0
  58. package/export-template/docs/assets/inter-italic-greek.DJ8dCoTZ.woff2 +0 -0
  59. package/export-template/docs/assets/inter-italic-latin-ext.CN1xVJS-.woff2 +0 -0
  60. package/export-template/docs/assets/inter-italic-latin.C2AdPX0b.woff2 +0 -0
  61. package/export-template/docs/assets/inter-italic-vietnamese.BSbpV94h.woff2 +0 -0
  62. package/export-template/docs/assets/inter-roman-cyrillic-ext.BBPuwvHQ.woff2 +0 -0
  63. package/export-template/docs/assets/inter-roman-cyrillic.C5lxZ8CY.woff2 +0 -0
  64. package/export-template/docs/assets/inter-roman-greek-ext.CqjqNYQ-.woff2 +0 -0
  65. package/export-template/docs/assets/inter-roman-greek.BBVDIX6e.woff2 +0 -0
  66. package/export-template/docs/assets/inter-roman-latin-ext.4ZJIpNVo.woff2 +0 -0
  67. package/export-template/docs/assets/inter-roman-latin.Di8DUHzh.woff2 +0 -0
  68. package/export-template/docs/assets/inter-roman-vietnamese.BjW4sHH5.woff2 +0 -0
  69. package/export-template/docs/assets/join-strategy.C_FvrCEo.svg +1 -0
  70. package/export-template/docs/assets/join-strategy.dark.ChMLnNII.svg +1 -0
  71. package/export-template/docs/assets/memory-borrowing.BqQRJg0u.svg +1 -0
  72. package/export-template/docs/assets/memory-borrowing.dark.Yhh20O9C.svg +1 -0
  73. package/export-template/docs/assets/memory-regions.XHvO7jHG.svg +1 -0
  74. package/export-template/docs/assets/memory-regions.dark.D4TP9_08.svg +1 -0
  75. package/export-template/docs/assets/repartition-vs-coalesce.BovLRrpj.svg +1 -0
  76. package/export-template/docs/assets/repartition-vs-coalesce.dark.BhAczKZQ.svg +1 -0
  77. package/export-template/docs/assets/retry-escalation-ladder.DyTKJJmZ.svg +1 -0
  78. package/export-template/docs/assets/retry-escalation-ladder.dark.BdsabtU3.svg +1 -0
  79. package/export-template/docs/assets/shuffle-map-reduce.KuOEZVmg.svg +1 -0
  80. package/export-template/docs/assets/shuffle-map-reduce.dark.BgQZnFSb.svg +1 -0
  81. package/export-template/docs/assets/spill-classification.BU2euYDO.svg +1 -0
  82. package/export-template/docs/assets/spill-classification.dark.D7i1M40d.svg +1 -0
  83. package/export-template/docs/assets/style.DSixAiZE.css +1 -0
  84. package/export-template/docs/assets/tuning-reference_anti-patterns.md.Df1YMIHu.js +1 -0
  85. package/export-template/docs/assets/tuning-reference_anti-patterns.md.Df1YMIHu.lean.js +1 -0
  86. package/export-template/docs/assets/tuning-reference_aqe.md.BIsCtLzm.js +1 -0
  87. package/export-template/docs/assets/tuning-reference_aqe.md.BIsCtLzm.lean.js +1 -0
  88. package/export-template/docs/assets/tuning-reference_bottleneck-broadcast-sizing.md.CEstB3Ia.js +1 -0
  89. package/export-template/docs/assets/tuning-reference_bottleneck-broadcast-sizing.md.CEstB3Ia.lean.js +1 -0
  90. package/export-template/docs/assets/tuning-reference_bottleneck-cold-start.md.CEuy-72y.js +7 -0
  91. package/export-template/docs/assets/tuning-reference_bottleneck-cold-start.md.CEuy-72y.lean.js +1 -0
  92. package/export-template/docs/assets/tuning-reference_bottleneck-duplicate-plan-subtree.md.CIohQDfn.js +1 -0
  93. package/export-template/docs/assets/tuning-reference_bottleneck-duplicate-plan-subtree.md.CIohQDfn.lean.js +1 -0
  94. package/export-template/docs/assets/tuning-reference_bottleneck-failures.md.4z5BXGJ2.js +6 -0
  95. package/export-template/docs/assets/tuning-reference_bottleneck-failures.md.4z5BXGJ2.lean.js +1 -0
  96. package/export-template/docs/assets/tuning-reference_bottleneck-gc.md.DSxzZRK7.js +6 -0
  97. package/export-template/docs/assets/tuning-reference_bottleneck-gc.md.DSxzZRK7.lean.js +1 -0
  98. package/export-template/docs/assets/tuning-reference_bottleneck-job-failure-rate.md.BaJl__1W.js +8 -0
  99. package/export-template/docs/assets/tuning-reference_bottleneck-job-failure-rate.md.BaJl__1W.lean.js +1 -0
  100. package/export-template/docs/assets/tuning-reference_bottleneck-memory-utilization.md.DbP-SJZc.js +7 -0
  101. package/export-template/docs/assets/tuning-reference_bottleneck-memory-utilization.md.DbP-SJZc.lean.js +1 -0
  102. package/export-template/docs/assets/tuning-reference_bottleneck-retry-waste.md.D5JMjVOt.js +1 -0
  103. package/export-template/docs/assets/tuning-reference_bottleneck-retry-waste.md.D5JMjVOt.lean.js +1 -0
  104. package/export-template/docs/assets/tuning-reference_bottleneck-shuffle.md.CM-nTmIH.js +12 -0
  105. package/export-template/docs/assets/tuning-reference_bottleneck-shuffle.md.CM-nTmIH.lean.js +1 -0
  106. package/export-template/docs/assets/tuning-reference_bottleneck-skew.md.BdUwiDhn.js +14 -0
  107. package/export-template/docs/assets/tuning-reference_bottleneck-skew.md.BdUwiDhn.lean.js +1 -0
  108. package/export-template/docs/assets/tuning-reference_bottleneck-slow-host.md.BlIo6UDW.js +7 -0
  109. package/export-template/docs/assets/tuning-reference_bottleneck-slow-host.md.BlIo6UDW.lean.js +1 -0
  110. package/export-template/docs/assets/tuning-reference_bottleneck-small-files.md.B8kloyx8.js +5 -0
  111. package/export-template/docs/assets/tuning-reference_bottleneck-small-files.md.B8kloyx8.lean.js +1 -0
  112. package/export-template/docs/assets/tuning-reference_bottleneck-spill.md.PNH7mITt.js +6 -0
  113. package/export-template/docs/assets/tuning-reference_bottleneck-spill.md.PNH7mITt.lean.js +1 -0
  114. package/export-template/docs/assets/tuning-reference_bottleneck-straggler.md.DY36fHN5.js +7 -0
  115. package/export-template/docs/assets/tuning-reference_bottleneck-straggler.md.DY36fHN5.lean.js +1 -0
  116. package/export-template/docs/assets/tuning-reference_bottleneck-tiny-tasks.md.QTV7O8kU.js +8 -0
  117. package/export-template/docs/assets/tuning-reference_bottleneck-tiny-tasks.md.QTV7O8kU.lean.js +1 -0
  118. package/export-template/docs/assets/tuning-reference_bottleneck-utilization.md.DTiueZC3.js +5 -0
  119. package/export-template/docs/assets/tuning-reference_bottleneck-utilization.md.DTiueZC3.lean.js +1 -0
  120. package/export-template/docs/assets/tuning-reference_caching.md.B7aQ8asB.js +1 -0
  121. package/export-template/docs/assets/tuning-reference_caching.md.B7aQ8asB.lean.js +1 -0
  122. package/export-template/docs/assets/tuning-reference_cluster-config.md.ZVmDGsQ3.js +1 -0
  123. package/export-template/docs/assets/tuning-reference_cluster-config.md.ZVmDGsQ3.lean.js +1 -0
  124. package/export-template/docs/assets/tuning-reference_config.md.UvveiWG3.js +1 -0
  125. package/export-template/docs/assets/tuning-reference_config.md.UvveiWG3.lean.js +1 -0
  126. package/export-template/docs/assets/tuning-reference_data-formats.md.bjCAWH3N.js +1 -0
  127. package/export-template/docs/assets/tuning-reference_data-formats.md.bjCAWH3N.lean.js +1 -0
  128. package/export-template/docs/assets/tuning-reference_index.md.BQ_NooMV.js +1 -0
  129. package/export-template/docs/assets/tuning-reference_index.md.BQ_NooMV.lean.js +1 -0
  130. package/export-template/docs/assets/tuning-reference_intro.md.CobD-lGB.js +1 -0
  131. package/export-template/docs/assets/tuning-reference_intro.md.CobD-lGB.lean.js +1 -0
  132. package/export-template/docs/assets/tuning-reference_joins.md.BtKs_CuW.js +1 -0
  133. package/export-template/docs/assets/tuning-reference_joins.md.BtKs_CuW.lean.js +1 -0
  134. package/export-template/docs/assets/tuning-reference_memory-model.md.DhT-n4y3.js +1 -0
  135. package/export-template/docs/assets/tuning-reference_memory-model.md.DhT-n4y3.lean.js +1 -0
  136. package/export-template/docs/assets/tuning-reference_metrics.md.mLOh7Apj.js +1 -0
  137. package/export-template/docs/assets/tuning-reference_metrics.md.mLOh7Apj.lean.js +1 -0
  138. package/export-template/docs/assets/tuning-reference_partitioning.md.q0zKF_8X.js +1 -0
  139. package/export-template/docs/assets/tuning-reference_partitioning.md.q0zKF_8X.lean.js +1 -0
  140. package/export-template/docs/assets/tuning-reference_pyspark.md.DDCfvN9t.js +6 -0
  141. package/export-template/docs/assets/tuning-reference_pyspark.md.DDCfvN9t.lean.js +1 -0
  142. package/export-template/docs/assets/tuning-reference_shuffle.md.BZZ7R4Ix.js +1 -0
  143. package/export-template/docs/assets/tuning-reference_shuffle.md.BZZ7R4Ix.lean.js +1 -0
  144. package/export-template/docs/assets/tuning-reference_spark-architecture.md.Dwzm5avO.js +1 -0
  145. package/export-template/docs/assets/tuning-reference_spark-architecture.md.Dwzm5avO.lean.js +1 -0
  146. package/export-template/docs/assets/tuning-reference_table-formats.md.D6wj-2dX.js +1 -0
  147. package/export-template/docs/assets/tuning-reference_table-formats.md.D6wj-2dX.lean.js +1 -0
  148. package/export-template/docs/assets/udf-execution-models.BUFDICuG.svg +1 -0
  149. package/export-template/docs/assets/udf-execution-models.dark.YTNS6GDq.svg +1 -0
  150. package/export-template/docs/assets/user-guide_alternative-log-retrieval.md.B4tPGIal.js +1 -0
  151. package/export-template/docs/assets/user-guide_alternative-log-retrieval.md.B4tPGIal.lean.js +1 -0
  152. package/export-template/docs/assets/user-guide_getting-started.md.BJvwLEIM.js +3 -0
  153. package/export-template/docs/assets/user-guide_getting-started.md.BJvwLEIM.lean.js +1 -0
  154. package/export-template/docs/assets/user-guide_mcp-tools.md.Vi3RoflJ.js +125 -0
  155. package/export-template/docs/assets/user-guide_mcp-tools.md.Vi3RoflJ.lean.js +1 -0
  156. package/export-template/docs/assets/user-guide_run-comparison.md.CQc1aoU8.js +1 -0
  157. package/export-template/docs/assets/user-guide_run-comparison.md.CQc1aoU8.lean.js +1 -0
  158. package/export-template/docs/assets/user-guide_understanding-findings.md.DL1UDhvR.js +1 -0
  159. package/export-template/docs/assets/user-guide_understanding-findings.md.DL1UDhvR.lean.js +1 -0
  160. package/export-template/docs/contributor-guide/architecture/board-widgets.html +25 -0
  161. package/export-template/docs/contributor-guide/architecture/detector-contract.html +25 -0
  162. package/export-template/docs/contributor-guide/architecture/drill-down.html +25 -0
  163. package/export-template/docs/contributor-guide/architecture/impact-estimation.html +25 -0
  164. package/export-template/docs/contributor-guide/architecture/index.html +25 -0
  165. package/export-template/docs/contributor-guide/architecture/overview.html +25 -0
  166. package/export-template/docs/contributor-guide/architecture/state-and-history.html +25 -0
  167. package/export-template/docs/contributor-guide/architecture/widget-rendering.html +25 -0
  168. package/export-template/docs/contributor-guide/architecture/worker-protocol.html +30 -0
  169. package/export-template/docs/contributor-guide/contributing.html +25 -0
  170. package/export-template/docs/contributor-guide/development-setup.html +36 -0
  171. package/export-template/docs/contributor-guide/testing.html +25 -0
  172. package/export-template/docs/favicon.svg +4 -0
  173. package/export-template/docs/hashmap.json +1 -0
  174. package/export-template/docs/index.html +25 -0
  175. package/export-template/docs/package.json +1 -0
  176. package/export-template/docs/tuning-reference/anti-patterns.html +25 -0
  177. package/export-template/docs/tuning-reference/aqe.html +25 -0
  178. package/export-template/docs/tuning-reference/bottleneck-broadcast-sizing.html +25 -0
  179. package/export-template/docs/tuning-reference/bottleneck-cold-start.html +31 -0
  180. package/export-template/docs/tuning-reference/bottleneck-duplicate-plan-subtree.html +25 -0
  181. package/export-template/docs/tuning-reference/bottleneck-failures.html +30 -0
  182. package/export-template/docs/tuning-reference/bottleneck-gc.html +30 -0
  183. package/export-template/docs/tuning-reference/bottleneck-job-failure-rate.html +32 -0
  184. package/export-template/docs/tuning-reference/bottleneck-memory-utilization.html +31 -0
  185. package/export-template/docs/tuning-reference/bottleneck-retry-waste.html +25 -0
  186. package/export-template/docs/tuning-reference/bottleneck-shuffle.html +36 -0
  187. package/export-template/docs/tuning-reference/bottleneck-skew.html +38 -0
  188. package/export-template/docs/tuning-reference/bottleneck-slow-host.html +31 -0
  189. package/export-template/docs/tuning-reference/bottleneck-small-files.html +29 -0
  190. package/export-template/docs/tuning-reference/bottleneck-spill.html +30 -0
  191. package/export-template/docs/tuning-reference/bottleneck-straggler.html +31 -0
  192. package/export-template/docs/tuning-reference/bottleneck-tiny-tasks.html +32 -0
  193. package/export-template/docs/tuning-reference/bottleneck-utilization.html +29 -0
  194. package/export-template/docs/tuning-reference/caching.html +25 -0
  195. package/export-template/docs/tuning-reference/cluster-config.html +25 -0
  196. package/export-template/docs/tuning-reference/config.html +25 -0
  197. package/export-template/docs/tuning-reference/data-formats.html +25 -0
  198. package/export-template/docs/tuning-reference/index.html +25 -0
  199. package/export-template/docs/tuning-reference/intro.html +25 -0
  200. package/export-template/docs/tuning-reference/joins.html +25 -0
  201. package/export-template/docs/tuning-reference/memory-model.html +25 -0
  202. package/export-template/docs/tuning-reference/metrics.html +25 -0
  203. package/export-template/docs/tuning-reference/partitioning.html +25 -0
  204. package/export-template/docs/tuning-reference/pyspark.html +30 -0
  205. package/export-template/docs/tuning-reference/shuffle.html +25 -0
  206. package/export-template/docs/tuning-reference/spark-architecture.html +25 -0
  207. package/export-template/docs/tuning-reference/table-formats.html +25 -0
  208. package/export-template/docs/user-guide/alternative-log-retrieval.html +25 -0
  209. package/export-template/docs/user-guide/getting-started.html +27 -0
  210. package/export-template/docs/user-guide/mcp-tools.html +149 -0
  211. package/export-template/docs/user-guide/run-comparison.html +25 -0
  212. package/export-template/docs/user-guide/understanding-findings.html +25 -0
  213. package/export-template/docs/vp-icons.css +0 -0
  214. package/export-template/favicon.svg +4 -0
  215. package/export-template/index.html +111 -0
  216. package/export-template/parser-worker-DyjiQvfP.js +112 -0
  217. package/export-template/sample-runs/sample-run.ndjson.gz +0 -0
  218. package/package.json +20 -6
  219. package/vendor-core/analyzer.js +74 -74
  220. package/vendor-core/cli/budgets.js +13 -27
  221. package/vendor-core/cli/collect-run.js +43 -19
  222. package/vendor-core/core-count.js +25 -27
  223. package/vendor-core/core-locality-ratio.js +4 -11
  224. package/vendor-core/core-time-series.js +6 -12
  225. package/vendor-core/core-usage-locality.js +3 -4
  226. package/vendor-core/detectors.js +395 -389
  227. package/vendor-core/docs-config.js +69 -21
  228. package/vendor-core/docs-content/chapters/01-intro.md +32 -0
  229. package/vendor-core/docs-content/chapters/02-spark-architecture.md +76 -0
  230. package/vendor-core/docs-content/chapters/03-memory-model.md +73 -0
  231. package/vendor-core/docs-content/chapters/04-partitioning.md +65 -0
  232. package/vendor-core/docs-content/chapters/05-joins.md +62 -0
  233. package/vendor-core/docs-content/chapters/06-shuffle.md +59 -0
  234. package/vendor-core/docs-content/chapters/07-data-formats.md +81 -0
  235. package/vendor-core/docs-content/chapters/07b-table-formats.md +56 -0
  236. package/vendor-core/docs-content/chapters/08-caching.md +58 -0
  237. package/vendor-core/docs-content/chapters/09-pyspark.md +78 -0
  238. package/vendor-core/docs-content/chapters/10-aqe.md +167 -0
  239. package/vendor-core/docs-content/chapters/11-cluster-config.md +170 -0
  240. package/vendor-core/docs-content/chapters/12-anti-patterns.md +171 -0
  241. package/vendor-core/docs-content/chapters/14-metrics.md +87 -0
  242. package/vendor-core/docs-content/chapters/15-config.md +93 -0
  243. package/vendor-core/docs-content/chapters/nav-index.json +370 -0
  244. package/vendor-core/docs-content/detection/cache.md +7 -0
  245. package/vendor-core/docs-content/detection/cfg.md +15 -0
  246. package/vendor-core/docs-content/detection/chrn.md +9 -0
  247. package/vendor-core/docs-content/detection/cold.md +4 -0
  248. package/vendor-core/docs-content/detection/cstor.md +4 -0
  249. package/vendor-core/docs-content/detection/fail.md +5 -0
  250. package/vendor-core/docs-content/detection/gc.md +4 -0
  251. package/vendor-core/docs-content/detection/host.md +5 -0
  252. package/vendor-core/docs-content/detection/incmp.md +6 -0
  253. package/vendor-core/docs-content/detection/jobs.md +4 -0
  254. package/vendor-core/docs-content/detection/local.md +6 -0
  255. package/vendor-core/docs-content/detection/mem.md +10 -0
  256. package/vendor-core/docs-content/detection/part.md +5 -0
  257. package/vendor-core/docs-content/detection/plan.md +14 -0
  258. package/vendor-core/docs-content/detection/retry.md +4 -0
  259. package/vendor-core/docs-content/detection/sfail.md +5 -0
  260. package/vendor-core/docs-content/detection/shape.md +5 -0
  261. package/vendor-core/docs-content/detection/shfl.md +4 -0
  262. package/vendor-core/docs-content/detection/skew.md +6 -0
  263. package/vendor-core/docs-content/detection/slow.md +6 -0
  264. package/vendor-core/docs-content/detection/spec.md +8 -0
  265. package/vendor-core/docs-content/detection/spill.md +7 -0
  266. package/vendor-core/docs-content/detection/strag.md +5 -0
  267. package/vendor-core/docs-content/detection/tiny.md +4 -0
  268. package/vendor-core/docs-content/detection/util.md +4 -0
  269. package/vendor-core/docs-content/diagrams/aqe-loop.dark.svg +1 -0
  270. package/vendor-core/docs-content/diagrams/aqe-loop.svg +1 -0
  271. package/vendor-core/docs-content/diagrams/broadcast-vs-shuffle.dark.svg +1 -0
  272. package/vendor-core/docs-content/diagrams/broadcast-vs-shuffle.svg +1 -0
  273. package/vendor-core/docs-content/diagrams/cache-lifecycle.dark.svg +1 -0
  274. package/vendor-core/docs-content/diagrams/cache-lifecycle.svg +1 -0
  275. package/vendor-core/docs-content/diagrams/cold-start-timeline.dark.svg +1 -0
  276. package/vendor-core/docs-content/diagrams/cold-start-timeline.svg +1 -0
  277. package/vendor-core/docs-content/diagrams/columnar-layout.dark.svg +1 -0
  278. package/vendor-core/docs-content/diagrams/columnar-layout.svg +1 -0
  279. package/vendor-core/docs-content/diagrams/container-memory.dark.svg +1 -0
  280. package/vendor-core/docs-content/diagrams/container-memory.svg +1 -0
  281. package/vendor-core/docs-content/diagrams/dag-stages.dark.svg +1 -0
  282. package/vendor-core/docs-content/diagrams/dag-stages.svg +1 -0
  283. package/vendor-core/docs-content/diagrams/driver-executor.dark.svg +1 -0
  284. package/vendor-core/docs-content/diagrams/driver-executor.svg +1 -0
  285. package/vendor-core/docs-content/diagrams/duplicate-plan-subtree.dark.svg +1 -0
  286. package/vendor-core/docs-content/diagrams/duplicate-plan-subtree.svg +1 -0
  287. package/vendor-core/docs-content/diagrams/join-strategy.dark.svg +1 -0
  288. package/vendor-core/docs-content/diagrams/join-strategy.svg +1 -0
  289. package/vendor-core/docs-content/diagrams/memory-borrowing.dark.svg +1 -0
  290. package/vendor-core/docs-content/diagrams/memory-borrowing.svg +1 -0
  291. package/vendor-core/docs-content/diagrams/memory-regions.dark.svg +1 -0
  292. package/vendor-core/docs-content/diagrams/memory-regions.svg +1 -0
  293. package/vendor-core/docs-content/diagrams/repartition-vs-coalesce.dark.svg +1 -0
  294. package/vendor-core/docs-content/diagrams/repartition-vs-coalesce.svg +1 -0
  295. package/vendor-core/docs-content/diagrams/retry-escalation-ladder.dark.svg +1 -0
  296. package/vendor-core/docs-content/diagrams/retry-escalation-ladder.svg +1 -0
  297. package/vendor-core/docs-content/diagrams/shuffle-map-reduce.dark.svg +1 -0
  298. package/vendor-core/docs-content/diagrams/shuffle-map-reduce.svg +1 -0
  299. package/vendor-core/docs-content/diagrams/spill-classification.dark.svg +1 -0
  300. package/vendor-core/docs-content/diagrams/spill-classification.svg +1 -0
  301. package/vendor-core/docs-content/diagrams/udf-execution-models.dark.svg +1 -0
  302. package/vendor-core/docs-content/diagrams/udf-execution-models.svg +1 -0
  303. package/vendor-core/docs-content/tuning/broadcast-sizing.md +78 -0
  304. package/vendor-core/docs-content/tuning/cold-start.md +81 -0
  305. package/vendor-core/docs-content/tuning/duplicate-plan-subtree.md +45 -0
  306. package/vendor-core/docs-content/tuning/failures.md +124 -0
  307. package/vendor-core/docs-content/tuning/gc.md +110 -0
  308. package/vendor-core/docs-content/tuning/job-failure-rate.md +101 -0
  309. package/vendor-core/docs-content/tuning/memory-utilization.md +58 -0
  310. package/vendor-core/docs-content/tuning/retry-waste.md +90 -0
  311. package/vendor-core/docs-content/tuning/shuffle.md +154 -0
  312. package/vendor-core/docs-content/tuning/skew.md +123 -0
  313. package/vendor-core/docs-content/tuning/slow-host.md +117 -0
  314. package/vendor-core/docs-content/tuning/small-files.md +99 -0
  315. package/vendor-core/docs-content/tuning/spill.md +114 -0
  316. package/vendor-core/docs-content/tuning/straggler.md +103 -0
  317. package/vendor-core/docs-content/tuning/tiny-tasks.md +94 -0
  318. package/vendor-core/docs-content/tuning/utilization.md +90 -0
  319. package/vendor-core/docs-site-config.js +10 -17
  320. package/vendor-core/efficiency-model.js +7 -13
  321. package/vendor-core/etl-phases.js +3 -5
  322. package/vendor-core/event-handlers.js +232 -134
  323. package/vendor-core/event-schemas.js +48 -114
  324. package/vendor-core/evidence-availability.js +5 -10
  325. package/vendor-core/evidence-report.js +73 -123
  326. package/vendor-core/export-data.js +48 -0
  327. package/vendor-core/finding-action-label.js +4 -10
  328. package/vendor-core/finding-filter-predicate.js +3 -7
  329. package/vendor-core/finding-generic-recommendation.js +112 -0
  330. package/vendor-core/finding-names.js +51 -0
  331. package/vendor-core/format-utils.js +112 -38
  332. package/vendor-core/impact-band.js +18 -24
  333. package/vendor-core/impact-estimator.js +38 -74
  334. package/vendor-core/ingest.js +7 -13
  335. package/vendor-core/job-groups.js +3 -6
  336. package/vendor-core/list-runs.js +278 -0
  337. package/vendor-core/load-vendored.js +6 -12
  338. package/vendor-core/log-header-peek.js +81 -0
  339. package/vendor-core/lz4-block.js +4 -6
  340. package/vendor-core/mcp-server-factory.js +38 -8
  341. package/vendor-core/mcp-tools.js +105 -76
  342. package/vendor-core/model-assembler.js +8 -16
  343. package/vendor-core/occupancy.js +5 -9
  344. package/vendor-core/parser-worker.js +20 -29
  345. package/vendor-core/plan-dot.js +2 -5
  346. package/vendor-core/plan-duration-attribution.js +78 -29
  347. package/vendor-core/plan-graph-model.js +126 -69
  348. package/vendor-core/plan-node-detail.js +31 -17
  349. package/vendor-core/plan-summary.js +19 -8
  350. package/vendor-core/recommendation-rollup.js +35 -39
  351. package/vendor-core/redact.js +72 -16
  352. package/vendor-core/rolling-log-reassembly.js +4 -6
  353. package/vendor-core/run-comparison.js +86 -72
  354. package/vendor-core/scaling-sim.js +5 -7
  355. package/vendor-core/session-snapshot.js +1 -1
  356. package/vendor-core/shs-fetch.js +4 -6
  357. package/vendor-core/shs-load.js +9 -13
  358. package/vendor-core/shs-request.js +1 -1
  359. package/vendor-core/stage-quantiles.js +14 -0
  360. package/vendor-core/types.js +78 -18
  361. package/vendor-core/wasted-core-hours.js +7 -12
@@ -0,0 +1,99 @@
1
+ # Small Files
2
+
3
+ <span class="tag">SMALLFILES</span>
4
+
5
+ ## What it is
6
+
7
+ The small-files problem is the cost of managing a large count of tiny files. Writing many
8
+ small files runs up significant metadata overhead, and Spark handles that pattern badly;
9
+ distributed filesystems such as HDFS handle it badly too, which is why it earned the name
10
+ "small file problem"[^1]. The same trap shows up on the read side. Push
11
+ `spark.sql.files.maxPartitionBytes` (default 128 MB, aligned to the [Parquet](#data-formats) block size)
12
+ too low and Spark carves the input into many small partition files, piling on disk I/O and
13
+ the filesystem overhead of opening, closing, and listing directories, all of which are slow
14
+ on a distributed store[^2][^3].
15
+
16
+ Spark writes one file per output partition, so a job left with 200 partitions writes 200
17
+ files and one left with 3000 writes 3000, even when many of those files hold only a handful
18
+ of rows. That count then punishes every downstream job forced to read thousands of tiny
19
+ files[^4]. The opposite extreme is not free either: files that are too large make it
20
+ inefficient to read a whole block when you only need a few rows[^5].
21
+
22
+ ## How it's detected
23
+
24
+ Reading the SQL plan surfaces write paths that will emit far more output files
25
+ than the data warrants, keying off the one-file-per-output-partition rule[^4]. The signal is
26
+ a high output-partition count against a modest data volume, which foreshadows a directory
27
+ full of tiny files.
28
+
29
+ | Signal | What it points to |
30
+ |---|---|
31
+ | Output partition count high vs. data volume | Many tiny files on write |
32
+ | Read partitioning below `spark.sql.files.maxPartitionBytes` (128 MB) | Over-split input, excess filesystem I/O[^2][^3] |
33
+
34
+ ## Why it matters
35
+
36
+ Every extra file is another open, close, and directory-list operation, and on a distributed
37
+ filesystem those are not cheap[^2]. The count compounds downstream: a stage that scatters
38
+ 3000 tiny files hands the next job 3000 files to read back, so the metadata tax is paid twice
39
+ over[^4]. None of it is a single dramatic failure. The waste is spread thin across the file
40
+ count, which is exactly why it goes unnoticed until listing and I/O start to dominate.
41
+
42
+ ## How to fix it
43
+
44
+ Repartition right before the write so the output lands in fewer, better-sized files.
45
+
46
+ - `coalesce()` is a shuffle-free merge: it fuses existing partitions with no data movement,
47
+ so `df.coalesce(100).write.parquet(...)` collapses 3000 tiny files into 100 reasonable
48
+ ones[^4]. It does not rebalance, though, so uneven inputs stay uneven once lumped together,
49
+ and pushing it to `coalesce(1)` kills parallelism by forcing one executor to do all the
50
+ work[^4].
51
+ - `repartition()` is a full reshuffle that buys even distribution at the cost of that
52
+ [shuffle](#shuffle). Reach for it when the distribution itself needs fixing, not just the file
53
+ count[^4].
54
+
55
+ > **PySpark:** both are one-line calls on a DataFrame: `df.coalesce(100)` for a shuffle-free
56
+ > merge, or `df.repartition(100)` when the data also needs rebalancing.
57
+
58
+ ```python
59
+ # Collapse many tiny output files without a shuffle (distribution already even)
60
+ df.coalesce(100).write.parquet(path)
61
+
62
+ # Full reshuffle to a target count when the distribution itself needs fixing
63
+ df.repartition(100).write.parquet(path)
64
+ ```
65
+
66
+ For runtime control, [Adaptive Query Execution](#aqe) can merge tiny shuffle partitions on its own.
67
+ With `spark.sql.adaptive.enabled` and `spark.sql.adaptive.coalescePartitions.enabled` (default
68
+ true) set, AQE coalesces contiguous shuffle partitions toward
69
+ `spark.sql.adaptive.advisoryPartitionSizeInBytes` (default 64 MB)[^4][^5]. Just mind the
70
+ scope: AQE only engages after the first shuffle, so it fixes shuffle-output partition counts
71
+ but will not repair input-side partitioning or a bad file layout on disk[^6].
72
+
73
+ ## Confidence
74
+
75
+ The core inference, that one output file per partition times a high
76
+ partition count yields many tiny files, is grounded directly in Spark's write behavior[^4],
77
+ and the mitigation levers (`coalesce()`, `repartition()`, and AQE coalescing) are documented
78
+ Spark features[^4][^5].
79
+
80
+ ## Limitations / false-positive risk
81
+
82
+ A high file count can be perfectly legitimate: partitioned output deliberately fans data
83
+ across many files by partition column, so a large count there is by design, not a defect.
84
+ And AQE coalescing is not a cure-all here, since it only affects post-shuffle partitions and
85
+ leaves the input file layout untouched[^6]. The signal reflects the shape, not intent, so
86
+ confirm the write is not an intended partitioned layout before acting.
87
+
88
+
89
+ ## Related
90
+
91
+ - **Partition sizing:** [Partitioning](#partitioning)
92
+ - **Too many small tasks:** [Tiny Tasks](#bottleneck-tiny-tasks)
93
+
94
+ [^1]: *Spark: The Definitive Guide*, Chambers & Zaharia, ch. 19
95
+ [^2]: *Learning Spark, 2nd Edition*, Damji, Wenig, Das, Lee, ch. 7
96
+ [^3]: [SQLConf.scala](https://raw.githubusercontent.com/apache/spark/v3.5.0/sql/catalyst/src/main/scala/org/apache/spark/sql/internal/SQLConf.scala)
97
+ [^4]: [Spark Partitions](https://luminousmen.com/post/spark-partitions)
98
+ [^5]: [Performance Tuning (Spark SQL, DataFrames and Datasets Guide)](https://spark.apache.org/docs/latest/sql-performance-tuning.html)
99
+ [^6]: [The Apache Spark Optimization Checklist](https://luminousmen.com/post/the-apache-spark-optimization-checklist)
@@ -0,0 +1,114 @@
1
+ # Memory / Disk Spill
2
+
3
+ <span class="tag">SPILL</span>
4
+
5
+ ## What it is
6
+
7
+ Spill happens when a task's share of execution memory runs out during a sort or hash
8
+ aggregation. Each task gets its own `TaskMemoryManager`, which arbitrates the shared execution
9
+ pool across every task running concurrently on an executor; with `n` tasks running
10
+ concurrently, each is allowed to allocate somewhere between `1/(2n)` and `1/n` of total
11
+ execution memory: a soft limit, not a hard stop.[^1] When a task's sort or hash-aggregation
12
+ operator keeps requesting execution-memory pages it can't get, Spark doesn't fail immediately:
13
+ it blocks the requesting task, triggers a spill of that task's in-memory data structure to
14
+ disk, or, in the worst case, throws an `OutOfMemoryError`.[^1]
15
+
16
+ The spill path runs through Spark's map/shuffle I/O machinery: shuffle partitions created by
17
+ wide transformations like `groupBy()` or `join()` spill to the executors' local disks, at the
18
+ location set by `spark.local.directory`.[^2] SQL physical operators apply the same idea with
19
+ their own guardrails: `SortMergeJoinExec`'s in-memory buffer and the cartesian-product
20
+ operator's buffer both spill once they cross a configured row-count threshold, which by
21
+ default is `spark.shuffle.spill.numElementsForceSpillThreshold`.[^3] Whatever the trigger, the
22
+ resulting spilled bytes are exposed on the task metrics as `memoryBytesSpilled`, visible in the
23
+ Spark UI and event log.[^4]
24
+
25
+ ## How it's detected
26
+
27
+ Detection: any `memoryBytesSpilled > 0` → Warning. The classification is the actionable
28
+ signal.
29
+
30
+ Spill classification:
31
+ - `skew`: ≥ 80% of tasks have zero spill; fix: address [task skew](#bottleneck-skew), not memory.
32
+ - `volume`: < 20% of tasks have zero spill; fix: more partitions or more memory.
33
+ - `unclassified`: neither condition; treat as volume.
34
+
35
+ <img class="light-only" src="../diagrams/spill-classification.svg" alt="How a nonzero memoryBytesSpilled is classified as skew, volume, or unclassified from the share of tasks with zero spill, and the fix each classification points to.">
36
+ <img class="dark-only" src="../diagrams/spill-classification.dark.svg" alt="How a nonzero memoryBytesSpilled is classified as skew, volume, or unclassified from the share of tasks with zero spill, and the fix each classification points to.">
37
+
38
+ ## Why it matters
39
+
40
+ Spill is the fallback Spark reaches for once a task can no longer get the execution memory
41
+ it's asking for: rather than failing outright, it blocks the task, writes its buffered data to
42
+ disk, or (if that's not enough) throws an `OutOfMemoryError`.[^1] A task that's spilling has
43
+ already given up pure in-memory speed to keep running at all, which is why the skew-vs-volume
44
+ split above is the actionable part of the signal: the two failure shapes call for opposite
45
+ fixes.
46
+
47
+ ## How to fix it
48
+
49
+ - **Skew**: When only a few tasks spill, the problem is the shape of the data, not the size
50
+ of the memory pool. `repartition()` is the tool for this: reach for it specifically to fix a
51
+ lopsided partition distribution or to increase partition count[^5], in contrast to
52
+ `coalesce()`, which never rebalances skewed data because it just stacks existing partitions
53
+ together: partitions that were uneven going in are still uneven coming out.[^6] For skewed
54
+ joins, [Adaptive Query Execution](#aqe) can detect and split[^9] oversized partitions automatically: a
55
+ partition counts as skewed once it's larger than
56
+ `spark.sql.adaptive.skewJoin.skewedPartitionFactor` (default `5.0`) times the median
57
+ partition size, and also larger than
58
+ `spark.sql.adaptive.skewJoin.skewedPartitionThresholdInBytes` (default `256 MB`).[^7][^8]
59
+ Before AQE existed, the manual equivalent was salting: adding a prefix to the skewed keys to
60
+ make the same key look different, then adjusting the data distribution accordingly.[^9]
61
+ Databricks' `/*+ SKEW(...) */` hint lets you name the skewed relation and specific key values
62
+ directly, so the planner targets just those keys instead of relying on automatic
63
+ detection.[^10]
64
+ - **Volume**: When most tasks spill, the fix is more room to work with: more partitions, more
65
+ memory, or both. `repartition(n)` guarantees exactly `n` output partitions via a hash
66
+ shuffle[^11], and since Spark's per-task startup overhead is low (unlike MapReduce's), the
67
+ general bias is to err toward more partitions rather than fewer.[^12]
68
+ `spark.sql.files.maxPartitionBytes` (default `128 MB`) is the equivalent lever on the input
69
+ side, governing how much file data gets packed into each input partition before a shuffle
70
+ even happens.[^6] On the memory side, execution memory is a fraction of the JVM heap.
71
+ `spark.memory.fraction` (default `0.6`) sizes the shared execution/storage region as
72
+ `(JVM heap − 300 MiB) × spark.memory.fraction`[^13]. Because each task's slice of that
73
+ region is capped by the `TaskMemoryManager` at between `1/(2n)` and `1/n` of the total,[^1]
74
+ giving executors more memory, or running fewer concurrent tasks on each one, directly raises
75
+ the ceiling before spill kicks in.
76
+
77
+ For the **volume** case, give tasks more room: defaults shown, comments say which way to move:
78
+
79
+ ```properties
80
+ # Create more, smaller input partitions before the shuffle (default 128m)
81
+ spark.sql.files.maxPartitionBytes=128m
82
+
83
+ # Execution/storage share of (JVM heap - 300MiB) (default 0.6);
84
+ # raising it grows execution memory but starves the untracked user-memory region
85
+ spark.memory.fraction=0.6
86
+ ```
87
+
88
+ For the **skew** case (only a few tasks spill), the fix is `df.repartition(n)`, not more memory: see [Partitioning](#partitioning).
89
+
90
+ ## Confidence
91
+
92
+ The core detection (any `memoryBytesSpilled > 0`) and the skew-vs-volume classification are validated: both the skew branch (most tasks spill nothing, so the shape of the data is the problem) and the volume branch (nearly every task spills, so the pool is too small) map to a well-understood fix, and neither needs further validation. <span class="tag">EXPERIMENTAL</span> When a spill matches neither shape, it falls back to a low-confidence unclassified finding that still requires validation, because there is no confirmed cause to act on; the page defaults it to the volume remedy, but that is a guess, not a diagnosis.
93
+
94
+ ## Limitations / false-positive risk
95
+
96
+ Some spill is normal. A large aggregation or join can spill by design once its working set outgrows execution memory, and a small spill on a healthy stage rarely repays the effort of chasing it. The raw `memoryBytesSpilled > 0` trigger fires on both the actionable and the routine cases, so treat a small spill as informational until the skew-vs-volume split says otherwise.
97
+
98
+ ## Related
99
+
100
+ - **Why it happens:** [Memory Management](#memory-model), [Partitioning](#partitioning)
101
+
102
+ [^1]: [Diving into Spark Memory Management](https://luminousmen.com/post/dive-into-spark-memory)
103
+ [^2]: *Learning Spark, 2nd Edition*, Damji, Wenig, Das, Lee, ch. 7
104
+ [^3]: [SQLConf.scala](https://raw.githubusercontent.com/apache/spark/v3.5.0/sql/catalyst/src/main/scala/org/apache/spark/sql/internal/SQLConf.scala)
105
+ [^4]: [Monitoring and Instrumentation](https://spark.apache.org/docs/latest/monitoring.html#spark-history-server)
106
+ [^5]: *Spark: The Definitive Guide*, Chambers & Zaharia, ch. 19
107
+ [^6]: [Spark Partitions](https://luminousmen.com/post/spark-partitions)
108
+ [^7]: [Configuration (Spark)](https://spark.apache.org/docs/latest/configuration.html)
109
+ [^8]: [Performance Tuning (Spark SQL, DataFrames and Datasets Guide)](https://spark.apache.org/docs/latest/sql-performance-tuning.html)
110
+ [^9]: [SPARK-29544: Optimize Skewed Join in SQL Adaptive Execution](https://issues.apache.org/jira/browse/SPARK-29544)
111
+ [^10]: [Skew Join Optimization](https://docs.databricks.com/aws/en/archive/legacy/skew-join)
112
+ [^11]: [RDD.scala](https://raw.githubusercontent.com/apache/spark/v3.5.0/core/src/main/scala/org/apache/spark/rdd/RDD.scala)
113
+ [^12]: [How to Tune Your Apache Spark Jobs (Part 2): Cloudera Engineering Blog](https://blog.cloudera.com/how-to-tune-your-apache-spark-jobs-part-2/)
114
+ [^13]: [Spark Tuning Guide](https://spark.apache.org/docs/latest/tuning.html)
@@ -0,0 +1,103 @@
1
+ # Stragglers
2
+
3
+ <span class="tag">STRAG</span>
4
+
5
+ ## What it is
6
+
7
+ A straggler is one task (or a small handful of tasks) that runs far longer than the rest of
8
+ the tasks in its stage, even when the stage is otherwise healthy. Because a stage only
9
+ completes once its last task finishes, a single straggler holds up the whole stage, and every
10
+ other executor sits idle waiting for it to catch up.
11
+
12
+ ## How it's detected
13
+
14
+ A task counts as a straggler when speculative tasks fired for it, or when more than 5% of the
15
+ tasks in a stage run at least 4× the stage's median task duration; this rule only applies once
16
+ a stage has at least 10 tasks.
17
+
18
+ "Duration" here is `executorRunTime`: elapsed wall-clock time, not CPU time, and it already
19
+ includes any time the task spent blocked fetching shuffle data[^1]. That matters when tracking
20
+ down why a task is slow: a long `executorRunTime` doesn't necessarily mean more compute
21
+ happened. A [garbage-collection pause](#bottleneck-gc) is counted inside it rather than added on top
22
+ (`jvmGCTime` is a subset of `executorRunTime`, not additive[^1]), and a task waiting on a
23
+ remote shuffle block it needs next shows up in `shuffleReadMetrics.fetchWaitTime`, which only
24
+ counts genuine blocking time, not blocks being prefetched in the background[^1].
25
+
26
+ The "speculative tasks fired" half of the rule is Spark's own detector: once
27
+ `spark.speculation.quantile` (default `0.9`) of a stage's tasks finish, Spark compares each
28
+ remaining task's duration against `spark.speculation.multiplier` (default `3`) times the
29
+ median of the tasks that already finished, subject to a `spark.speculation.minTaskRuntime`
30
+ floor (default `100ms`) so short tasks aren't flagged just for being slower than a tiny
31
+ median[^2]. Speculation itself is off by default (`spark.speculation` defaults to `false`), so
32
+ this half of the rule only fires on stages where it's been turned on[^2].
33
+
34
+ Watch for overlap with [task skew](#bottleneck-skew): an unevenly distributed key sends one
35
+ partition far more data than its peers, and that partition's task will trip this same duration
36
+ threshold even though the underlying cause is data volume, not a slow host or a GC pause.
37
+ Check the skew signals before assuming the latter.
38
+
39
+ ## Why it matters
40
+
41
+ A straggler wastes cluster capacity the same way a skewed stage does: every executor other
42
+ than the one running the slow task finishes early and idles, while total job time still
43
+ tracks the single slowest task. Because a straggler's duration can be inflated by a GC
44
+ pause[^1] or a slow [shuffle](#shuffle) fetch[^1] rather than genuinely more work, it's worth checking
45
+ those angles (and whether the real cause is skew rather than anything task-local) before
46
+ assuming a hardware explanation.
47
+
48
+ ## How to fix it
49
+
50
+ - Enable speculative execution: `spark.speculation` is `false` by default, so nothing reruns
51
+ a slow task automatically until it's turned on[^2].
52
+ - Tune the trigger: `spark.speculation.quantile` (default `0.9`) sets how much of the stage
53
+ must finish before speculation kicks in, and `spark.speculation.multiplier` (default `3`)
54
+ sets how many times slower than the median a task must be. For stages with very few tasks,
55
+ `spark.speculation.task.duration.threshold` (available since 3.0.0) gives an absolute-duration
56
+ trigger instead of relying on the median[^2].
57
+ - Since Spark 3.4, `spark.speculation.efficiency.enabled` (default `true`) adds a filter so a
58
+ task is only speculated if its data-process rate is below the stage average (times
59
+ `spark.speculation.efficiency.processRateMultiplier`, default `0.75`) or its duration exceeds
60
+ `spark.speculation.efficiency.longRunTaskFactor` (default `2`) times the same time threshold.
61
+ This avoids wasting a duplicate task slot on work that's simply doing more, not running
62
+ slower[^2].
63
+ - If the cause is really a skewed key, treat it as [task skew](#bottleneck-skew) instead: AQE's
64
+ skew-join optimization automatically splits a partition once it's larger than
65
+ `spark.sql.adaptive.skewJoin.skewedPartitionFactor` (default `5.0`) times the median
66
+ partition size and above `spark.sql.adaptive.skewJoin.skewedPartitionThresholdInBytes`
67
+ (default `256 MB`)[^2][^3]. Manual salting (appending a random prefix to the skewed key so
68
+ it spreads across more partitions) is the pre-AQE fallback, though the design doc behind
69
+ AQE's skew handling calls salting and other manual approaches limited compared to the
70
+ automatic option[^4].
71
+
72
+ > **PySpark:** speculation settings can be set on the session directly, no `spark-submit` flag
73
+ > needed: `spark.conf.set("spark.speculation", "true")`, then the quantile/multiplier
74
+ > equivalents the same way.
75
+
76
+ Speculation config: defaults shown, plus the absolute-duration trigger for small stages:
77
+
78
+ ```properties
79
+ # Speculatively relaunch straggler tasks (OFF by default)
80
+ spark.speculation=true
81
+ spark.speculation.quantile=0.9 # default; portion of tasks finished before speculation begins
82
+ spark.speculation.multiplier=3 # default; multiple of median duration that marks a task slow
83
+
84
+ # Absolute-duration trigger for stages with very few tasks (since Spark 3.0)
85
+ spark.speculation.task.duration.threshold=10s # 10s is an example
86
+ ```
87
+
88
+ ## Confidence
89
+
90
+ Validated.
91
+
92
+ ## Limitations / false-positive risk
93
+
94
+ The 4x-median rule flags a slow task, but slow is not the same as broken. The same threshold trips on a skewed key that simply has more data to process, or on a task that spent its time in a GC pause rather than doing extra work, so a flagged task is not automatically a slow host. Small stages make this worse: with only a handful of tasks the median is unstable, and one moderately slow task can look like a straggler against a median computed from too few peers.
95
+
96
+ ## Related
97
+
98
+ - **When the real cause is a skewed key:** [Partitioning](#partitioning), [Adaptive Query Execution](#aqe)
99
+
100
+ [^1]: [Monitoring and Instrumentation](https://spark.apache.org/docs/latest/monitoring.html#spark-history-server)
101
+ [^2]: [Configuration (Spark)](https://spark.apache.org/docs/latest/configuration.html)
102
+ [^3]: [Optimizing Skew Join (Spark SQL, DataFrames and Datasets Guide)](https://spark.apache.org/docs/latest/sql-performance-tuning.html#optimizing-skew-join)
103
+ [^4]: [SPARK-29544: Optimize Skewed Join at Runtime with New Adaptive Execution](https://issues.apache.org/jira/browse/SPARK-29544)
@@ -0,0 +1,94 @@
1
+ # Tiny Tasks
2
+
3
+ <span class="tag">TINY</span>
4
+
5
+ ## What it is
6
+
7
+ Every task pays a fixed placement and serialization cost before it does any real work. Spark
8
+ schedules around data locality rather than moving data to code: "Spark builds its scheduling
9
+ around this general principle" that shipping serialized code is cheaper than shipping data,
10
+ preferring `PROCESS_LOCAL` locality down to `ANY` and waiting a configurable timeout for a busy
11
+ CPU to free up before shipping data to a farther executor[^1]. On top of that placement cost,
12
+ every task also pays a serialization cost for its closure and, if not using Kryo, for its data:
13
+ Java's default `ObjectOutputStream`-based serialization "is flexible but often quite slow, and
14
+ leads to large serialized formats," while Kryo is "significantly faster and more compact than
15
+ Java serialization (often as much as 10x)"[^1]. None of that overhead is large per task: Spark's
16
+ own guidance leans toward more, smaller tasks rather than fewer, larger ones, unlike MapReduce:
17
+ "it's almost always better to err on the side of a larger number of tasks (and thus partitions)...
18
+ The difference stems from the fact that MapReduce has a high startup overhead for tasks, while
19
+ Spark does not"[^2]. But once partitions get small enough, that per-task overhead stops being
20
+ negligible and starts to dominate.
21
+
22
+ ## How it's detected
23
+
24
+ The failure mode is partitions so small that scheduling and serialization overhead outweighs the
25
+ actual work. With Spark's default of 200 shuffle partitions applied to a dataset of only a few
26
+ megabytes, "those 200 partitions will each get like ten rows. Tasks become microscopic. Most of
27
+ your CPUs will just be sitting there doing nothing, wasting cluster hours"[^3]. The same
28
+ small-partition penalty shows up on the [shuffle-read side](#bottleneck-shuffle): production measurements found "the
29
+ average shuffle block size is only 10s of KBs, which leads to delayed shuffle data fetch," and
30
+ the shuffle reduce stages with the largest fetch delays (over 30 seconds per task) were
31
+ consistently the ones with small block sizes[^4]. A high task count paired with very short median
32
+ task duration, and/or very small shuffle block sizes on the read side, are the signals to look
33
+ for.
34
+
35
+ ## Why it matters
36
+
37
+ Tiny tasks waste cluster capacity even though no single task is a straggler the way
38
+ [skewed](#bottleneck-skew) or [straggler](#bottleneck-straggler) tasks are: the cost here is
39
+ spread evenly across thousands of tasks, each individually cheap but collectively adding up in
40
+ scheduling and serialization overhead, while cores that could be doing other useful work sit
41
+ mostly idle between task launches[^3].
42
+
43
+ ## How to fix it
44
+
45
+ - `coalesce()` merges existing partitions without a shuffle ("no data movement, no shuffle")
46
+ and is typically used right before a write, e.g. `df.coalesce(100).write.parquet(...)`, to
47
+ collapse many tiny output files into a manageable number cheaply[^3]. Because it only stacks
48
+ partitions together rather than redistributing data, it doesn't fix an uneven underlying
49
+ distribution (uneven input partitions stay uneven, just grouped) and pushing it too far
50
+ (`coalesce(1)`) kills parallelism by funneling all work onto one executor while the rest sit
51
+ idle[^3].
52
+ - `repartition()` performs a full reshuffle, giving control and rebalancing that Spark won't do
53
+ on its own: with `df.repartition(200)`, "you're paying for predictability"[^3]. Use it when
54
+ the partitioning itself needs to be fixed or rebalanced, not merely reduced in count, accepting
55
+ the shuffle cost to get there.
56
+ - There's a middle option, `coalesce(N, shuffle=True)`, which "acts more like a repartition, but
57
+ leaning toward reduction... not free (you pay for the shuffle cost) but you get better
58
+ distribution and fewer partitions"[^3].
59
+ - Rule of thumb: reach for `coalesce()` when only the partition count needs to shrink and the
60
+ existing distribution is already reasonably even (e.g. collapsing output files); reach for
61
+ `repartition()` when the distribution itself is the problem.
62
+
63
+ > **PySpark:** both are one-line calls on a DataFrame: `df.coalesce(100)` for a shuffle-free
64
+ > merge, `df.repartition(200)` for a full reshuffle, or `df.coalesce(100, shuffle=True)` for the
65
+ > shuffled middle option.
66
+
67
+ The three sizing calls, side by side:
68
+
69
+ ```python
70
+ # Collapse many tiny output files without a shuffle (use when distribution is already even)
71
+ df.coalesce(100).write.parquet(path)
72
+
73
+ # Full reshuffle to a target count, use when the distribution itself needs rebalancing
74
+ df = df.repartition(200)
75
+
76
+ # Middle ground: reduce partition count but still rebalance (pays a shuffle)
77
+ df = df.coalesce(100, shuffle=True)
78
+ ```
79
+
80
+ ## Limitations / false-positive risk
81
+
82
+ Very short tasks are not always a problem. A tiny final stage can be perfectly fine, and one
83
+ that just writes a small result set doesn't need to be resized. The scheduling-overhead penalty
84
+ only bites when tiny tasks dominate the stage, so treat a handful of short tasks as noise rather
85
+ than a finding.
86
+
87
+ ## Related
88
+
89
+ - **Partition sizing:** [Partitioning](#partitioning)
90
+
91
+ [^1]: [Tuning (Spark)](https://spark.apache.org/docs/latest/tuning.html)
92
+ [^2]: [How to Tune Your Apache Spark Jobs (Part 2): Cloudera](https://blog.cloudera.com/how-to-tune-your-apache-spark-jobs-part-2/)
93
+ [^3]: [Spark Partitions](https://luminousmen.com/post/spark-partitions)
94
+ [^4]: [SPARK-30602](https://issues.apache.org/jira/browse/SPARK-30602)
@@ -0,0 +1,90 @@
1
+ # Executor Utilization
2
+
3
+ <span class="tag">UTIL</span>
4
+
5
+ ## What it is
6
+
7
+ Executor utilization measures how much of the cluster's allocated executor capacity a job actually keeps busy. When the average number of active executors trails the peak number allocated, the cluster is holding compute (cores and memory) that isn't running any tasks.
8
+
9
+ ## How it's detected
10
+
11
+ The signal is the ratio of average active executors to the peak active executor count observed over the job's lifetime:
12
+
13
+ | avg active executors / peak | Level |
14
+ |---|---|
15
+ | < 60% | Info |
16
+ | < 40% | Warning |
17
+ | < 20% | Critical |
18
+
19
+ ## Why it matters
20
+
21
+ Idle executors are allocation you're paying for without getting work done. One common cause is too little task parallelism to occupy the allocated cores: the guidance is to keep at least as many partitions as there are cores across the executors, so no core sits idle[^1]. Parallelism can also be lost by accident rather than by under-partitioning upfront: because `coalesce` is a narrow transformation, reducing partition count with it forces the *entire* upstream stage down to the reduced parallelism, not just the coalesce step, trading a shuffle for lost concurrency[^2]. Separately, under dynamic allocation, a workload with many small tasks can end up over-provisioned: by default it requests enough executors to maximize parallelism for the task count, and with small tasks that can mean some executors "might not even do any work," wasting resources on allocation overhead[^3].
22
+
23
+ ## How to fix it
24
+
25
+ - Reach for `repartition` (not `coalesce`) when you need to raise partition count or rebalance data: `coalesce` only avoids a shuffle when shrinking partition count, and doing so also drags the entire upstream stage down to the reduced parallelism[^4][^2].
26
+ - Size partitions so the task count is at least the core count available across executors, to avoid leaving cores idle[^1].
27
+ - Under dynamic allocation, lower `spark.dynamicAllocation.executorAllocationRatio` below its default of `1.0` to scale back the number of executors requested for workloads made up of many small tasks[^3].
28
+ - Dynamic allocation requires shuffle tracking, the external shuffle service, or shuffle-block decommissioning to be enabled as a precondition[^3]; without one of them, an executor removed mid-shuffle takes its unfetched shuffle files with it, forcing a recompute[^5], which undercuts using dynamic allocation to shed idle executors in the first place. `spark.dynamicAllocation.shuffleTracking.enabled` defaults to `true` since Spark 3.0, satisfying that precondition without needing a separate external shuffle service[^3].
29
+
30
+ Restore parallelism after a heavy filter, and rein in over-provisioning for many-small-task jobs:
31
+
32
+ ```python
33
+ # repartition (not coalesce) rebalances and can raise partition count; aim >= total executor cores
34
+ df = df.filter(heavy_predicate).repartition(spark.sparkContext.defaultParallelism)
35
+
36
+ # Scale back executors requested for many-small-task workloads (default ratio 1.0)
37
+ spark.conf.set("spark.dynamicAllocation.executorAllocationRatio", "0.5") # 0.5 is an example
38
+ ```
39
+
40
+ ## Limitations / false-positive risk
41
+
42
+ A low average-to-peak ratio isn't always waste. A bursty or I/O-bound job legitimately holds executors while tasks wait on external systems rather than burning cores, and the ratio is sensitive to short stages, where a brief spike in allocation skews the average without meaning the cluster was genuinely idle.
43
+
44
+ ## Caching opportunity
45
+
46
+ <span class="tag">CACHE</span>
47
+
48
+ When the same DataFrame, RDD, or input is scanned more than once, low utilization can trace back to repeated recomputation rather than idle cores. Spark keeps nothing between actions: transformations only build a DAG, and once an action finishes its intermediate results are discarded[^6]. Call a second action on the same logic and Spark re-runs the whole DAG from the source, which can mean re-reading a terabyte from S3, re-reading Kafka, or repeating expensive decompression[^6]. Fork that logic into two pipeline branches and you sign up to recompute everything twice[^6].
49
+
50
+ <img class="light-only" src="../diagrams/duplicate-plan-subtree.svg" alt="How two branches that repeat the same scan and operators each recompute it, until a shared cached or reused node lets both read one materialized result.">
51
+ <img class="dark-only" src="../diagrams/duplicate-plan-subtree.dark.svg" alt="How two branches that repeat the same scan and operators each recompute it, until a shared cached or reused node lets both read one materialized result.">
52
+
53
+ [`cache()` and `persist()`](#caching) are how you stop that. Persisting materializes the RDD (usually in memory on the executors) so it can be reused within the job, while Spark keeps its lineage to recompute any lost partition[^7]. For a dataset you read repeatedly, this is one of the most useful optimizations available: it parks a DataFrame, table, or RDD in temporary storage across the executors and makes later reads fast[^8]. Materialization is lazy and per-block, so only an action that touches every partition (`count()` or a full write) caches all of it; a subset scan like `take(10)` leaves a partial cache[^6].
54
+
55
+ `cache()` is shorthand; `persist()` takes a `StorageLevel` and lets you pick the tradeoff:
56
+
57
+ | Level | Tradeoff |
58
+ |---|---|
59
+ | `MEMORY_ONLY` | Fastest reads (deserialized JVM objects), but partitions that don't fit are recomputed on the fly each time[^8]. |
60
+ | `MEMORY_AND_DISK` | Spills the overflow to disk instead of recomputing; Spark's default caching strategy and fine for most pipelines[^9]. For DataFrames, `.cache()` maps to this[^6]. |
61
+ | Serialized (`_SER`) | Byte arrays instead of raw objects: smaller footprint, slower reads. Reach for it when memory is tight[^9]. |
62
+ | Replicated (`_2`) | A second copy for fast fault recovery instead of waiting on recomputation, at twice the space[^10][^9]. |
63
+
64
+ Whether to spill to disk hinges on recomputation cost: reading a block back from disk only beats recomputing it when the work that produced the data is expensive or filters out a large fraction, so the RDD guide says don't enable disk otherwise[^10].
65
+
66
+ Caching is not free, and several cases don't warrant it:
67
+
68
+ - **Single use.** Caching adds serialization, deserialization, and storage cost, so caching data you read once only slows you down[^8].
69
+ - **Data larger than storage memory,** or a cheap transformation that isn't reused often regardless of size[^11].
70
+ - **Memory pressure.** Cache memory is memory taken from processing, and under the default spill strategy cached data can land on slower disk, so caching can cost more than just re-reading the source[^9].
71
+ - **Lost optimizer freedom.** Once a dataset is cached, Catalyst works on the in-memory copy and can no longer push filters down to the source[^9].
72
+
73
+ A persisted RDD you've stopped using still occupies memory until the app ends or eviction forces it out, so call `unpersist()` to reclaim it deliberately, which matters most on shared clusters[^9].
74
+
75
+ ## Related
76
+
77
+ - **Task parallelism:** [Partitioning](#partitioning)
78
+ - **Executor sizing:** [Cluster Tuning](#cluster-config)
79
+
80
+ [^1]: *Learning Spark, 2nd Edition*, Damji, Wenig, Das & Lee, ch. 7
81
+ [^2]: *High Performance Spark, 2nd Edition*, Karau, Polak & Warren, ch. 8
82
+ [^3]: [Configuration — Spark](https://spark.apache.org/docs/latest/configuration.html)
83
+ [^4]: *Spark: The Definitive Guide*, Chambers & Zaharia, ch. 19
84
+ [^5]: [Job Scheduling — Dynamic Resource Allocation](https://spark.apache.org/docs/latest/job-scheduling.html)
85
+ [^6]: [Explaining the Mechanics of Spark Caching](https://luminousmen.com/post/explaining-the-mechanics-of-spark-caching)
86
+ [^7]: *High Performance Spark, 2nd Edition*, Karau, Polak & Warren, ch. 2
87
+ [^8]: *Spark: The Definitive Guide*, Chambers & Zaharia, ch. 19
88
+ [^9]: [Spark Tips: Caching](https://luminousmen.com/post/spark-tips-caching)
89
+ [^10]: [RDD Programming Guide — Spark](https://spark.apache.org/docs/latest/rdd-programming-guide.html)
90
+ [^11]: *Learning Spark, 2nd Edition*, Damji, Wenig, Das & Lee, ch. 7
@@ -1,23 +1,16 @@
1
1
  import { typeTag } from './format-utils.js';
2
2
 
3
- // docs-config.ts's own header comment scopes it to "the single source of the
4
- // [vendor] docs-panel URL surface"; this is the equivalent single source for
5
- // docs-site (VitePress) links, kept in its own file so the two unrelated doc
6
- // systems don't blur into one.
3
+ // Single source for docs-site (VitePress) links, kept separate from docs-config.ts's vendor
4
+ // docs-panel surface so the two doc systems don't blur.
7
5
  //
8
- // The `.html` extension matters: it's what makes this link resolve in the
9
- // packaged `server/` deploy mode. That static file server
10
- // (packages/server/lib/static-files.js) matches a request path to a file exactly, with
11
- // no directory-index or extension-guessing fallback (unlike the Vite
12
- // dev-server's docs-site middleware in vite.config.ts, which tries
13
- // bare/`.html`/`index.html` candidates). VitePress's default build
14
- // (`cleanUrls` unset, i.e. false) names each page's output file `<slug>.html`,
15
- // so the extension-less form would 404 there.
6
+ // The `.html` extension matters: it's what makes this link resolve in the packaged server/ deploy
7
+ // mode, whose static file server matches a request path to a file exactly (no extension guessing).
8
+ // VitePress names each page <slug>.html, so the extension-less form would 404 there.
16
9
  //
17
- // No allowlist/gating function is needed here, unlike `isKnownDocAnchor` for
18
- // the vendor link: `tests/docs-site-tag-coverage.test.js` already fails CI if
19
- // any `TYPE_TAG_MAP` value lacks a documented `{#tag}` heading, and both the
20
- // map and the page live in this same repo, so a mismatch can't ship past CI.
10
+ // No allowlist needed (unlike isKnownDocAnchor): docs-site-tag-coverage.test.js fails CI if any
11
+ // TYPE_TAG_MAP value lacks a documented {#tag} heading.
12
+ //
13
+ // Relative, not /docs/: keeps working under any subpath the app is published at.
21
14
  export function findingGuideUrl(type ) {
22
- return `/docs/user-guide/understanding-findings.html#${typeTag(type).toLowerCase()}`;
15
+ return `docs/user-guide/understanding-findings.html#${typeTag(type).toLowerCase()}`;
23
16
  }
@@ -1,6 +1,5 @@
1
- // §5 Efficiency/wastage model (SparkLens EfficiencyStatisticsAnalyzer).
2
- // DESIGN SPIKE: decomposes available compute-hours into driver-bound vs.
3
- // executor-bound waste, plus two theoretical floors. Mapped onto computeWallClock.
1
+ // §5 Efficiency/wastage model. DESIGN SPIKE:
2
+ // decomposes available compute-hours into driver-bound vs executor-bound waste, plus two floors.
4
3
  import { computeWallClock } from './wall-clock.js';
5
4
  import { computeTotalCores } from './core-count.js';
6
5
 
@@ -18,13 +17,9 @@ export function computeEfficiencyModel({ app, stages, executorsAdded, runAggrega
18
17
 
19
18
 
20
19
  {
21
- // `app ?? {}`: computeTotalCores just falls back to the executor-derived core sum when
22
- // `resources` is absent, and every caller here already tolerates a null `app` (malformed
23
- // logs with no ApplicationStart); `app!` would crash on `app.resources` in that case.
24
- // `executorsAdded` cast: computeTotalCores only reads `totalCores`, present on
25
- // ExecutorAddedEvent (the only kind real callers pass here) but not ExecutorRemovedEvent,
26
- // so the union as a whole is a structural mismatch against computeTotalCores's
27
- // `{totalCores?}` shape.
20
+ // `app ?? {}`: computeTotalCores falls back to the executor core sum when resources is absent,
21
+ // and callers tolerate a null app (malformed logs); `app!` would crash on app.resources.
22
+ // Cast: computeTotalCores reads only totalCores, absent on ExecutorRemovedEvent, so the union mismatches.
28
23
  const totalCores = computeTotalCores(app ?? {}, executorsAdded );
29
24
  const appDurationMs = (app?.endTime ?? 0) - (app?.startTime ?? 0);
30
25
  const wc = computeWallClock(app, stages );
@@ -35,9 +30,8 @@ export function computeEfficiencyModel({ app, stages, executorsAdded, runAggrega
35
30
  const driverIdleMs = wc.startup + wc.gaps + wc.idle;
36
31
  const driverWasteHours = totalCores * (driverIdleMs / 3600000);
37
32
 
38
- // Executor-bound: during stagesActive, allocated capacity beyond busy cores.
39
- // Tasks only run during active windows, so whole-run busyCoreMs already lives
40
- // inside stagesActive; no separate per-window sweep needed.
33
+ // Executor-bound: allocated capacity beyond busy cores during stagesActive. Tasks only run in
34
+ // active windows, so whole-run busyCoreMs already lives inside stagesActive.
41
35
  const allocatedActiveCoreMs = totalCores * wc.stagesActive;
42
36
  const busyCoreMs = runAggregates?.busyCoreMs ?? 0;
43
37
  const executorWasteHours = Math.max(0, allocatedActiveCoreMs - busyCoreMs) / 3600000;
@@ -1,8 +1,6 @@
1
- // §7 ETL-phase time attribution (Onehouse Spark Analyzer). Heuristic and
2
- // approximate: classify each stage by byte-flow shape. A stage can be both
3
- // Transform and Load (shuffle + write), so buckets overlap and need not sum to
4
- // wall-clock. Storage-format-aware attribution (Hudi/Delta/Iceberg) is NOT
5
- // portable (no table-format metadata in event logs) and is out of scope.
1
+ // §7 ETL-phase time attribution (Onehouse Spark Analyzer). Heuristic: classify each stage by
2
+ // byte-flow shape. A stage can be both Transform and Load, so buckets overlap and need not sum to
3
+ // wall-clock. Storage-format-aware attribution (Hudi/Delta/Iceberg) isn't portable, out of scope.
6
4
 
7
5
 
8
6