sparkforensics-cli 0.1.0 → 0.2.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +6 -0
- package/bin/sparkforensics-analyze.mjs +113 -48
- package/export-template/docs/404.html +25 -0
- package/export-template/docs/assets/app.DQTZyGL1.js +1 -0
- package/export-template/docs/assets/aqe-loop.IwQSATHw.svg +1 -0
- package/export-template/docs/assets/aqe-loop.dark.DGbaxqJE.svg +1 -0
- package/export-template/docs/assets/broadcast-vs-shuffle.Db4WY1XK.svg +1 -0
- package/export-template/docs/assets/broadcast-vs-shuffle.dark.C7Bxs0mG.svg +1 -0
- package/export-template/docs/assets/cache-lifecycle.dark.B-hS7AgU.svg +1 -0
- package/export-template/docs/assets/cache-lifecycle.rEOVYQNU.svg +1 -0
- package/export-template/docs/assets/chunks/@localSearchIndexroot.DppnXnDE.js +1 -0
- package/export-template/docs/assets/chunks/VPLocalSearchBox.BkBIPFs6.js +9 -0
- package/export-template/docs/assets/chunks/duplicate-plan-subtree.dark.Cdp70QhV.js +1 -0
- package/export-template/docs/assets/chunks/framework.DSg0KOwT.js +20 -0
- package/export-template/docs/assets/chunks/retry-escalation-ladder.dark.DHipdJgZ.js +1 -0
- package/export-template/docs/assets/chunks/theme.DP0u1AUq.js +2 -0
- package/export-template/docs/assets/cold-start-timeline.DxC_Sc7w.svg +1 -0
- package/export-template/docs/assets/cold-start-timeline.dark.CZ17YcAG.svg +1 -0
- package/export-template/docs/assets/columnar-layout.PghGeOEA.svg +1 -0
- package/export-template/docs/assets/columnar-layout.dark.BVNlz0ff.svg +1 -0
- package/export-template/docs/assets/container-memory.DIO0AnIm.svg +1 -0
- package/export-template/docs/assets/container-memory.dark.CP-5zuCl.svg +1 -0
- package/export-template/docs/assets/contributor-guide_architecture_board-widgets.md.CWpj01WU.js +1 -0
- package/export-template/docs/assets/contributor-guide_architecture_board-widgets.md.CWpj01WU.lean.js +1 -0
- package/export-template/docs/assets/contributor-guide_architecture_detector-contract.md.CgzUsQ6W.js +1 -0
- package/export-template/docs/assets/contributor-guide_architecture_detector-contract.md.CgzUsQ6W.lean.js +1 -0
- package/export-template/docs/assets/contributor-guide_architecture_drill-down.md.BtPdlM7r.js +1 -0
- package/export-template/docs/assets/contributor-guide_architecture_drill-down.md.BtPdlM7r.lean.js +1 -0
- package/export-template/docs/assets/contributor-guide_architecture_impact-estimation.md.CooslVJt.js +1 -0
- package/export-template/docs/assets/contributor-guide_architecture_impact-estimation.md.CooslVJt.lean.js +1 -0
- package/export-template/docs/assets/contributor-guide_architecture_index.md.3TO9ic6w.js +1 -0
- package/export-template/docs/assets/contributor-guide_architecture_index.md.3TO9ic6w.lean.js +1 -0
- package/export-template/docs/assets/contributor-guide_architecture_overview.md.CehiRmGn.js +1 -0
- package/export-template/docs/assets/contributor-guide_architecture_overview.md.CehiRmGn.lean.js +1 -0
- package/export-template/docs/assets/contributor-guide_architecture_state-and-history.md.C-xxn0q7.js +1 -0
- package/export-template/docs/assets/contributor-guide_architecture_state-and-history.md.C-xxn0q7.lean.js +1 -0
- package/export-template/docs/assets/contributor-guide_architecture_widget-rendering.md.R27gQrgY.js +1 -0
- package/export-template/docs/assets/contributor-guide_architecture_widget-rendering.md.R27gQrgY.lean.js +1 -0
- package/export-template/docs/assets/contributor-guide_architecture_worker-protocol.md.IbnfNrV3.js +6 -0
- package/export-template/docs/assets/contributor-guide_architecture_worker-protocol.md.IbnfNrV3.lean.js +1 -0
- package/export-template/docs/assets/contributor-guide_contributing.md.CvRsdr6J.js +1 -0
- package/export-template/docs/assets/contributor-guide_contributing.md.CvRsdr6J.lean.js +1 -0
- package/export-template/docs/assets/contributor-guide_development-setup.md.DvAN_9mK.js +12 -0
- package/export-template/docs/assets/contributor-guide_development-setup.md.DvAN_9mK.lean.js +1 -0
- package/export-template/docs/assets/contributor-guide_testing.md.6rIKqSyY.js +1 -0
- package/export-template/docs/assets/contributor-guide_testing.md.6rIKqSyY.lean.js +1 -0
- package/export-template/docs/assets/dag-stages.DSz_S937.svg +1 -0
- package/export-template/docs/assets/dag-stages.dark.F72UzxH4.svg +1 -0
- package/export-template/docs/assets/driver-executor.D5pQ7YN1.svg +1 -0
- package/export-template/docs/assets/driver-executor.dark.BmX9cPvh.svg +1 -0
- package/export-template/docs/assets/duplicate-plan-subtree.B4cvN6fj.svg +1 -0
- package/export-template/docs/assets/duplicate-plan-subtree.dark.Dw8wS0Ag.svg +1 -0
- package/export-template/docs/assets/index.md.CHJVslga.js +1 -0
- package/export-template/docs/assets/index.md.CHJVslga.lean.js +1 -0
- package/export-template/docs/assets/inter-italic-cyrillic-ext.r48I6akx.woff2 +0 -0
- package/export-template/docs/assets/inter-italic-cyrillic.By2_1cv3.woff2 +0 -0
- package/export-template/docs/assets/inter-italic-greek-ext.1u6EdAuj.woff2 +0 -0
- package/export-template/docs/assets/inter-italic-greek.DJ8dCoTZ.woff2 +0 -0
- package/export-template/docs/assets/inter-italic-latin-ext.CN1xVJS-.woff2 +0 -0
- package/export-template/docs/assets/inter-italic-latin.C2AdPX0b.woff2 +0 -0
- package/export-template/docs/assets/inter-italic-vietnamese.BSbpV94h.woff2 +0 -0
- package/export-template/docs/assets/inter-roman-cyrillic-ext.BBPuwvHQ.woff2 +0 -0
- package/export-template/docs/assets/inter-roman-cyrillic.C5lxZ8CY.woff2 +0 -0
- package/export-template/docs/assets/inter-roman-greek-ext.CqjqNYQ-.woff2 +0 -0
- package/export-template/docs/assets/inter-roman-greek.BBVDIX6e.woff2 +0 -0
- package/export-template/docs/assets/inter-roman-latin-ext.4ZJIpNVo.woff2 +0 -0
- package/export-template/docs/assets/inter-roman-latin.Di8DUHzh.woff2 +0 -0
- package/export-template/docs/assets/inter-roman-vietnamese.BjW4sHH5.woff2 +0 -0
- package/export-template/docs/assets/join-strategy.C_FvrCEo.svg +1 -0
- package/export-template/docs/assets/join-strategy.dark.ChMLnNII.svg +1 -0
- package/export-template/docs/assets/memory-borrowing.BqQRJg0u.svg +1 -0
- package/export-template/docs/assets/memory-borrowing.dark.Yhh20O9C.svg +1 -0
- package/export-template/docs/assets/memory-regions.XHvO7jHG.svg +1 -0
- package/export-template/docs/assets/memory-regions.dark.D4TP9_08.svg +1 -0
- package/export-template/docs/assets/repartition-vs-coalesce.BovLRrpj.svg +1 -0
- package/export-template/docs/assets/repartition-vs-coalesce.dark.BhAczKZQ.svg +1 -0
- package/export-template/docs/assets/retry-escalation-ladder.DyTKJJmZ.svg +1 -0
- package/export-template/docs/assets/retry-escalation-ladder.dark.BdsabtU3.svg +1 -0
- package/export-template/docs/assets/shuffle-map-reduce.KuOEZVmg.svg +1 -0
- package/export-template/docs/assets/shuffle-map-reduce.dark.BgQZnFSb.svg +1 -0
- package/export-template/docs/assets/spill-classification.BU2euYDO.svg +1 -0
- package/export-template/docs/assets/spill-classification.dark.D7i1M40d.svg +1 -0
- package/export-template/docs/assets/style.DSixAiZE.css +1 -0
- package/export-template/docs/assets/tuning-reference_anti-patterns.md.Df1YMIHu.js +1 -0
- package/export-template/docs/assets/tuning-reference_anti-patterns.md.Df1YMIHu.lean.js +1 -0
- package/export-template/docs/assets/tuning-reference_aqe.md.BIsCtLzm.js +1 -0
- package/export-template/docs/assets/tuning-reference_aqe.md.BIsCtLzm.lean.js +1 -0
- package/export-template/docs/assets/tuning-reference_bottleneck-broadcast-sizing.md.CEstB3Ia.js +1 -0
- package/export-template/docs/assets/tuning-reference_bottleneck-broadcast-sizing.md.CEstB3Ia.lean.js +1 -0
- package/export-template/docs/assets/tuning-reference_bottleneck-cold-start.md.CEuy-72y.js +7 -0
- package/export-template/docs/assets/tuning-reference_bottleneck-cold-start.md.CEuy-72y.lean.js +1 -0
- package/export-template/docs/assets/tuning-reference_bottleneck-duplicate-plan-subtree.md.CIohQDfn.js +1 -0
- package/export-template/docs/assets/tuning-reference_bottleneck-duplicate-plan-subtree.md.CIohQDfn.lean.js +1 -0
- package/export-template/docs/assets/tuning-reference_bottleneck-failures.md.4z5BXGJ2.js +6 -0
- package/export-template/docs/assets/tuning-reference_bottleneck-failures.md.4z5BXGJ2.lean.js +1 -0
- package/export-template/docs/assets/tuning-reference_bottleneck-gc.md.DSxzZRK7.js +6 -0
- package/export-template/docs/assets/tuning-reference_bottleneck-gc.md.DSxzZRK7.lean.js +1 -0
- package/export-template/docs/assets/tuning-reference_bottleneck-job-failure-rate.md.BaJl__1W.js +8 -0
- package/export-template/docs/assets/tuning-reference_bottleneck-job-failure-rate.md.BaJl__1W.lean.js +1 -0
- package/export-template/docs/assets/tuning-reference_bottleneck-memory-utilization.md.DbP-SJZc.js +7 -0
- package/export-template/docs/assets/tuning-reference_bottleneck-memory-utilization.md.DbP-SJZc.lean.js +1 -0
- package/export-template/docs/assets/tuning-reference_bottleneck-retry-waste.md.D5JMjVOt.js +1 -0
- package/export-template/docs/assets/tuning-reference_bottleneck-retry-waste.md.D5JMjVOt.lean.js +1 -0
- package/export-template/docs/assets/tuning-reference_bottleneck-shuffle.md.CM-nTmIH.js +12 -0
- package/export-template/docs/assets/tuning-reference_bottleneck-shuffle.md.CM-nTmIH.lean.js +1 -0
- package/export-template/docs/assets/tuning-reference_bottleneck-skew.md.BdUwiDhn.js +14 -0
- package/export-template/docs/assets/tuning-reference_bottleneck-skew.md.BdUwiDhn.lean.js +1 -0
- package/export-template/docs/assets/tuning-reference_bottleneck-slow-host.md.BlIo6UDW.js +7 -0
- package/export-template/docs/assets/tuning-reference_bottleneck-slow-host.md.BlIo6UDW.lean.js +1 -0
- package/export-template/docs/assets/tuning-reference_bottleneck-small-files.md.B8kloyx8.js +5 -0
- package/export-template/docs/assets/tuning-reference_bottleneck-small-files.md.B8kloyx8.lean.js +1 -0
- package/export-template/docs/assets/tuning-reference_bottleneck-spill.md.PNH7mITt.js +6 -0
- package/export-template/docs/assets/tuning-reference_bottleneck-spill.md.PNH7mITt.lean.js +1 -0
- package/export-template/docs/assets/tuning-reference_bottleneck-straggler.md.DY36fHN5.js +7 -0
- package/export-template/docs/assets/tuning-reference_bottleneck-straggler.md.DY36fHN5.lean.js +1 -0
- package/export-template/docs/assets/tuning-reference_bottleneck-tiny-tasks.md.QTV7O8kU.js +8 -0
- package/export-template/docs/assets/tuning-reference_bottleneck-tiny-tasks.md.QTV7O8kU.lean.js +1 -0
- package/export-template/docs/assets/tuning-reference_bottleneck-utilization.md.DTiueZC3.js +5 -0
- package/export-template/docs/assets/tuning-reference_bottleneck-utilization.md.DTiueZC3.lean.js +1 -0
- package/export-template/docs/assets/tuning-reference_caching.md.B7aQ8asB.js +1 -0
- package/export-template/docs/assets/tuning-reference_caching.md.B7aQ8asB.lean.js +1 -0
- package/export-template/docs/assets/tuning-reference_cluster-config.md.ZVmDGsQ3.js +1 -0
- package/export-template/docs/assets/tuning-reference_cluster-config.md.ZVmDGsQ3.lean.js +1 -0
- package/export-template/docs/assets/tuning-reference_config.md.UvveiWG3.js +1 -0
- package/export-template/docs/assets/tuning-reference_config.md.UvveiWG3.lean.js +1 -0
- package/export-template/docs/assets/tuning-reference_data-formats.md.bjCAWH3N.js +1 -0
- package/export-template/docs/assets/tuning-reference_data-formats.md.bjCAWH3N.lean.js +1 -0
- package/export-template/docs/assets/tuning-reference_index.md.BQ_NooMV.js +1 -0
- package/export-template/docs/assets/tuning-reference_index.md.BQ_NooMV.lean.js +1 -0
- package/export-template/docs/assets/tuning-reference_intro.md.CobD-lGB.js +1 -0
- package/export-template/docs/assets/tuning-reference_intro.md.CobD-lGB.lean.js +1 -0
- package/export-template/docs/assets/tuning-reference_joins.md.BtKs_CuW.js +1 -0
- package/export-template/docs/assets/tuning-reference_joins.md.BtKs_CuW.lean.js +1 -0
- package/export-template/docs/assets/tuning-reference_memory-model.md.DhT-n4y3.js +1 -0
- package/export-template/docs/assets/tuning-reference_memory-model.md.DhT-n4y3.lean.js +1 -0
- package/export-template/docs/assets/tuning-reference_metrics.md.mLOh7Apj.js +1 -0
- package/export-template/docs/assets/tuning-reference_metrics.md.mLOh7Apj.lean.js +1 -0
- package/export-template/docs/assets/tuning-reference_partitioning.md.q0zKF_8X.js +1 -0
- package/export-template/docs/assets/tuning-reference_partitioning.md.q0zKF_8X.lean.js +1 -0
- package/export-template/docs/assets/tuning-reference_pyspark.md.DDCfvN9t.js +6 -0
- package/export-template/docs/assets/tuning-reference_pyspark.md.DDCfvN9t.lean.js +1 -0
- package/export-template/docs/assets/tuning-reference_shuffle.md.BZZ7R4Ix.js +1 -0
- package/export-template/docs/assets/tuning-reference_shuffle.md.BZZ7R4Ix.lean.js +1 -0
- package/export-template/docs/assets/tuning-reference_spark-architecture.md.Dwzm5avO.js +1 -0
- package/export-template/docs/assets/tuning-reference_spark-architecture.md.Dwzm5avO.lean.js +1 -0
- package/export-template/docs/assets/tuning-reference_table-formats.md.D6wj-2dX.js +1 -0
- package/export-template/docs/assets/tuning-reference_table-formats.md.D6wj-2dX.lean.js +1 -0
- package/export-template/docs/assets/udf-execution-models.BUFDICuG.svg +1 -0
- package/export-template/docs/assets/udf-execution-models.dark.YTNS6GDq.svg +1 -0
- package/export-template/docs/assets/user-guide_alternative-log-retrieval.md.B4tPGIal.js +1 -0
- package/export-template/docs/assets/user-guide_alternative-log-retrieval.md.B4tPGIal.lean.js +1 -0
- package/export-template/docs/assets/user-guide_getting-started.md.BJvwLEIM.js +3 -0
- package/export-template/docs/assets/user-guide_getting-started.md.BJvwLEIM.lean.js +1 -0
- package/export-template/docs/assets/user-guide_mcp-tools.md.Vi3RoflJ.js +125 -0
- package/export-template/docs/assets/user-guide_mcp-tools.md.Vi3RoflJ.lean.js +1 -0
- package/export-template/docs/assets/user-guide_run-comparison.md.CQc1aoU8.js +1 -0
- package/export-template/docs/assets/user-guide_run-comparison.md.CQc1aoU8.lean.js +1 -0
- package/export-template/docs/assets/user-guide_understanding-findings.md.DL1UDhvR.js +1 -0
- package/export-template/docs/assets/user-guide_understanding-findings.md.DL1UDhvR.lean.js +1 -0
- package/export-template/docs/contributor-guide/architecture/board-widgets.html +25 -0
- package/export-template/docs/contributor-guide/architecture/detector-contract.html +25 -0
- package/export-template/docs/contributor-guide/architecture/drill-down.html +25 -0
- package/export-template/docs/contributor-guide/architecture/impact-estimation.html +25 -0
- package/export-template/docs/contributor-guide/architecture/index.html +25 -0
- package/export-template/docs/contributor-guide/architecture/overview.html +25 -0
- package/export-template/docs/contributor-guide/architecture/state-and-history.html +25 -0
- package/export-template/docs/contributor-guide/architecture/widget-rendering.html +25 -0
- package/export-template/docs/contributor-guide/architecture/worker-protocol.html +30 -0
- package/export-template/docs/contributor-guide/contributing.html +25 -0
- package/export-template/docs/contributor-guide/development-setup.html +36 -0
- package/export-template/docs/contributor-guide/testing.html +25 -0
- package/export-template/docs/favicon.svg +4 -0
- package/export-template/docs/hashmap.json +1 -0
- package/export-template/docs/index.html +25 -0
- package/export-template/docs/package.json +1 -0
- package/export-template/docs/tuning-reference/anti-patterns.html +25 -0
- package/export-template/docs/tuning-reference/aqe.html +25 -0
- package/export-template/docs/tuning-reference/bottleneck-broadcast-sizing.html +25 -0
- package/export-template/docs/tuning-reference/bottleneck-cold-start.html +31 -0
- package/export-template/docs/tuning-reference/bottleneck-duplicate-plan-subtree.html +25 -0
- package/export-template/docs/tuning-reference/bottleneck-failures.html +30 -0
- package/export-template/docs/tuning-reference/bottleneck-gc.html +30 -0
- package/export-template/docs/tuning-reference/bottleneck-job-failure-rate.html +32 -0
- package/export-template/docs/tuning-reference/bottleneck-memory-utilization.html +31 -0
- package/export-template/docs/tuning-reference/bottleneck-retry-waste.html +25 -0
- package/export-template/docs/tuning-reference/bottleneck-shuffle.html +36 -0
- package/export-template/docs/tuning-reference/bottleneck-skew.html +38 -0
- package/export-template/docs/tuning-reference/bottleneck-slow-host.html +31 -0
- package/export-template/docs/tuning-reference/bottleneck-small-files.html +29 -0
- package/export-template/docs/tuning-reference/bottleneck-spill.html +30 -0
- package/export-template/docs/tuning-reference/bottleneck-straggler.html +31 -0
- package/export-template/docs/tuning-reference/bottleneck-tiny-tasks.html +32 -0
- package/export-template/docs/tuning-reference/bottleneck-utilization.html +29 -0
- package/export-template/docs/tuning-reference/caching.html +25 -0
- package/export-template/docs/tuning-reference/cluster-config.html +25 -0
- package/export-template/docs/tuning-reference/config.html +25 -0
- package/export-template/docs/tuning-reference/data-formats.html +25 -0
- package/export-template/docs/tuning-reference/index.html +25 -0
- package/export-template/docs/tuning-reference/intro.html +25 -0
- package/export-template/docs/tuning-reference/joins.html +25 -0
- package/export-template/docs/tuning-reference/memory-model.html +25 -0
- package/export-template/docs/tuning-reference/metrics.html +25 -0
- package/export-template/docs/tuning-reference/partitioning.html +25 -0
- package/export-template/docs/tuning-reference/pyspark.html +30 -0
- package/export-template/docs/tuning-reference/shuffle.html +25 -0
- package/export-template/docs/tuning-reference/spark-architecture.html +25 -0
- package/export-template/docs/tuning-reference/table-formats.html +25 -0
- package/export-template/docs/user-guide/alternative-log-retrieval.html +25 -0
- package/export-template/docs/user-guide/getting-started.html +27 -0
- package/export-template/docs/user-guide/mcp-tools.html +149 -0
- package/export-template/docs/user-guide/run-comparison.html +25 -0
- package/export-template/docs/user-guide/understanding-findings.html +25 -0
- package/export-template/docs/vp-icons.css +0 -0
- package/export-template/favicon.svg +4 -0
- package/export-template/index.html +111 -0
- package/export-template/parser-worker-DyjiQvfP.js +112 -0
- package/export-template/sample-runs/sample-run.ndjson.gz +0 -0
- package/package.json +20 -6
- package/vendor-core/analyzer.js +74 -74
- package/vendor-core/cli/budgets.js +13 -27
- package/vendor-core/cli/collect-run.js +43 -19
- package/vendor-core/core-count.js +25 -27
- package/vendor-core/core-locality-ratio.js +4 -11
- package/vendor-core/core-time-series.js +6 -12
- package/vendor-core/core-usage-locality.js +3 -4
- package/vendor-core/detectors.js +395 -389
- package/vendor-core/docs-config.js +69 -21
- package/vendor-core/docs-content/chapters/01-intro.md +32 -0
- package/vendor-core/docs-content/chapters/02-spark-architecture.md +76 -0
- package/vendor-core/docs-content/chapters/03-memory-model.md +73 -0
- package/vendor-core/docs-content/chapters/04-partitioning.md +65 -0
- package/vendor-core/docs-content/chapters/05-joins.md +62 -0
- package/vendor-core/docs-content/chapters/06-shuffle.md +59 -0
- package/vendor-core/docs-content/chapters/07-data-formats.md +81 -0
- package/vendor-core/docs-content/chapters/07b-table-formats.md +56 -0
- package/vendor-core/docs-content/chapters/08-caching.md +58 -0
- package/vendor-core/docs-content/chapters/09-pyspark.md +78 -0
- package/vendor-core/docs-content/chapters/10-aqe.md +167 -0
- package/vendor-core/docs-content/chapters/11-cluster-config.md +170 -0
- package/vendor-core/docs-content/chapters/12-anti-patterns.md +171 -0
- package/vendor-core/docs-content/chapters/14-metrics.md +87 -0
- package/vendor-core/docs-content/chapters/15-config.md +93 -0
- package/vendor-core/docs-content/chapters/nav-index.json +370 -0
- package/vendor-core/docs-content/detection/cache.md +7 -0
- package/vendor-core/docs-content/detection/cfg.md +15 -0
- package/vendor-core/docs-content/detection/chrn.md +9 -0
- package/vendor-core/docs-content/detection/cold.md +4 -0
- package/vendor-core/docs-content/detection/cstor.md +4 -0
- package/vendor-core/docs-content/detection/fail.md +5 -0
- package/vendor-core/docs-content/detection/gc.md +4 -0
- package/vendor-core/docs-content/detection/host.md +5 -0
- package/vendor-core/docs-content/detection/incmp.md +6 -0
- package/vendor-core/docs-content/detection/jobs.md +4 -0
- package/vendor-core/docs-content/detection/local.md +6 -0
- package/vendor-core/docs-content/detection/mem.md +10 -0
- package/vendor-core/docs-content/detection/part.md +5 -0
- package/vendor-core/docs-content/detection/plan.md +14 -0
- package/vendor-core/docs-content/detection/retry.md +4 -0
- package/vendor-core/docs-content/detection/sfail.md +5 -0
- package/vendor-core/docs-content/detection/shape.md +5 -0
- package/vendor-core/docs-content/detection/shfl.md +4 -0
- package/vendor-core/docs-content/detection/skew.md +6 -0
- package/vendor-core/docs-content/detection/slow.md +6 -0
- package/vendor-core/docs-content/detection/spec.md +8 -0
- package/vendor-core/docs-content/detection/spill.md +7 -0
- package/vendor-core/docs-content/detection/strag.md +5 -0
- package/vendor-core/docs-content/detection/tiny.md +4 -0
- package/vendor-core/docs-content/detection/util.md +4 -0
- package/vendor-core/docs-content/diagrams/aqe-loop.dark.svg +1 -0
- package/vendor-core/docs-content/diagrams/aqe-loop.svg +1 -0
- package/vendor-core/docs-content/diagrams/broadcast-vs-shuffle.dark.svg +1 -0
- package/vendor-core/docs-content/diagrams/broadcast-vs-shuffle.svg +1 -0
- package/vendor-core/docs-content/diagrams/cache-lifecycle.dark.svg +1 -0
- package/vendor-core/docs-content/diagrams/cache-lifecycle.svg +1 -0
- package/vendor-core/docs-content/diagrams/cold-start-timeline.dark.svg +1 -0
- package/vendor-core/docs-content/diagrams/cold-start-timeline.svg +1 -0
- package/vendor-core/docs-content/diagrams/columnar-layout.dark.svg +1 -0
- package/vendor-core/docs-content/diagrams/columnar-layout.svg +1 -0
- package/vendor-core/docs-content/diagrams/container-memory.dark.svg +1 -0
- package/vendor-core/docs-content/diagrams/container-memory.svg +1 -0
- package/vendor-core/docs-content/diagrams/dag-stages.dark.svg +1 -0
- package/vendor-core/docs-content/diagrams/dag-stages.svg +1 -0
- package/vendor-core/docs-content/diagrams/driver-executor.dark.svg +1 -0
- package/vendor-core/docs-content/diagrams/driver-executor.svg +1 -0
- package/vendor-core/docs-content/diagrams/duplicate-plan-subtree.dark.svg +1 -0
- package/vendor-core/docs-content/diagrams/duplicate-plan-subtree.svg +1 -0
- package/vendor-core/docs-content/diagrams/join-strategy.dark.svg +1 -0
- package/vendor-core/docs-content/diagrams/join-strategy.svg +1 -0
- package/vendor-core/docs-content/diagrams/memory-borrowing.dark.svg +1 -0
- package/vendor-core/docs-content/diagrams/memory-borrowing.svg +1 -0
- package/vendor-core/docs-content/diagrams/memory-regions.dark.svg +1 -0
- package/vendor-core/docs-content/diagrams/memory-regions.svg +1 -0
- package/vendor-core/docs-content/diagrams/repartition-vs-coalesce.dark.svg +1 -0
- package/vendor-core/docs-content/diagrams/repartition-vs-coalesce.svg +1 -0
- package/vendor-core/docs-content/diagrams/retry-escalation-ladder.dark.svg +1 -0
- package/vendor-core/docs-content/diagrams/retry-escalation-ladder.svg +1 -0
- package/vendor-core/docs-content/diagrams/shuffle-map-reduce.dark.svg +1 -0
- package/vendor-core/docs-content/diagrams/shuffle-map-reduce.svg +1 -0
- package/vendor-core/docs-content/diagrams/spill-classification.dark.svg +1 -0
- package/vendor-core/docs-content/diagrams/spill-classification.svg +1 -0
- package/vendor-core/docs-content/diagrams/udf-execution-models.dark.svg +1 -0
- package/vendor-core/docs-content/diagrams/udf-execution-models.svg +1 -0
- package/vendor-core/docs-content/tuning/broadcast-sizing.md +78 -0
- package/vendor-core/docs-content/tuning/cold-start.md +81 -0
- package/vendor-core/docs-content/tuning/duplicate-plan-subtree.md +45 -0
- package/vendor-core/docs-content/tuning/failures.md +124 -0
- package/vendor-core/docs-content/tuning/gc.md +110 -0
- package/vendor-core/docs-content/tuning/job-failure-rate.md +101 -0
- package/vendor-core/docs-content/tuning/memory-utilization.md +58 -0
- package/vendor-core/docs-content/tuning/retry-waste.md +90 -0
- package/vendor-core/docs-content/tuning/shuffle.md +154 -0
- package/vendor-core/docs-content/tuning/skew.md +123 -0
- package/vendor-core/docs-content/tuning/slow-host.md +117 -0
- package/vendor-core/docs-content/tuning/small-files.md +99 -0
- package/vendor-core/docs-content/tuning/spill.md +114 -0
- package/vendor-core/docs-content/tuning/straggler.md +103 -0
- package/vendor-core/docs-content/tuning/tiny-tasks.md +94 -0
- package/vendor-core/docs-content/tuning/utilization.md +90 -0
- package/vendor-core/docs-site-config.js +10 -17
- package/vendor-core/efficiency-model.js +7 -13
- package/vendor-core/etl-phases.js +3 -5
- package/vendor-core/event-handlers.js +232 -134
- package/vendor-core/event-schemas.js +48 -114
- package/vendor-core/evidence-availability.js +5 -10
- package/vendor-core/evidence-report.js +73 -123
- package/vendor-core/export-data.js +48 -0
- package/vendor-core/finding-action-label.js +4 -10
- package/vendor-core/finding-filter-predicate.js +3 -7
- package/vendor-core/finding-generic-recommendation.js +112 -0
- package/vendor-core/finding-names.js +51 -0
- package/vendor-core/format-utils.js +112 -38
- package/vendor-core/impact-band.js +18 -24
- package/vendor-core/impact-estimator.js +38 -74
- package/vendor-core/ingest.js +7 -13
- package/vendor-core/job-groups.js +3 -6
- package/vendor-core/list-runs.js +278 -0
- package/vendor-core/load-vendored.js +6 -12
- package/vendor-core/log-header-peek.js +81 -0
- package/vendor-core/lz4-block.js +4 -6
- package/vendor-core/mcp-server-factory.js +38 -8
- package/vendor-core/mcp-tools.js +105 -76
- package/vendor-core/model-assembler.js +8 -16
- package/vendor-core/occupancy.js +5 -9
- package/vendor-core/parser-worker.js +20 -29
- package/vendor-core/plan-dot.js +2 -5
- package/vendor-core/plan-duration-attribution.js +78 -29
- package/vendor-core/plan-graph-model.js +126 -69
- package/vendor-core/plan-node-detail.js +31 -17
- package/vendor-core/plan-summary.js +19 -8
- package/vendor-core/recommendation-rollup.js +35 -39
- package/vendor-core/redact.js +72 -16
- package/vendor-core/rolling-log-reassembly.js +4 -6
- package/vendor-core/run-comparison.js +86 -72
- package/vendor-core/scaling-sim.js +5 -7
- package/vendor-core/session-snapshot.js +1 -1
- package/vendor-core/shs-fetch.js +4 -6
- package/vendor-core/shs-load.js +9 -13
- package/vendor-core/shs-request.js +1 -1
- package/vendor-core/stage-quantiles.js +14 -0
- package/vendor-core/types.js +78 -18
- package/vendor-core/wasted-core-hours.js +7 -12
|
@@ -0,0 +1,99 @@
|
|
|
1
|
+
# Small Files
|
|
2
|
+
|
|
3
|
+
<span class="tag">SMALLFILES</span>
|
|
4
|
+
|
|
5
|
+
## What it is
|
|
6
|
+
|
|
7
|
+
The small-files problem is the cost of managing a large count of tiny files. Writing many
|
|
8
|
+
small files runs up significant metadata overhead, and Spark handles that pattern badly;
|
|
9
|
+
distributed filesystems such as HDFS handle it badly too, which is why it earned the name
|
|
10
|
+
"small file problem"[^1]. The same trap shows up on the read side. Push
|
|
11
|
+
`spark.sql.files.maxPartitionBytes` (default 128 MB, aligned to the [Parquet](#data-formats) block size)
|
|
12
|
+
too low and Spark carves the input into many small partition files, piling on disk I/O and
|
|
13
|
+
the filesystem overhead of opening, closing, and listing directories, all of which are slow
|
|
14
|
+
on a distributed store[^2][^3].
|
|
15
|
+
|
|
16
|
+
Spark writes one file per output partition, so a job left with 200 partitions writes 200
|
|
17
|
+
files and one left with 3000 writes 3000, even when many of those files hold only a handful
|
|
18
|
+
of rows. That count then punishes every downstream job forced to read thousands of tiny
|
|
19
|
+
files[^4]. The opposite extreme is not free either: files that are too large make it
|
|
20
|
+
inefficient to read a whole block when you only need a few rows[^5].
|
|
21
|
+
|
|
22
|
+
## How it's detected
|
|
23
|
+
|
|
24
|
+
Reading the SQL plan surfaces write paths that will emit far more output files
|
|
25
|
+
than the data warrants, keying off the one-file-per-output-partition rule[^4]. The signal is
|
|
26
|
+
a high output-partition count against a modest data volume, which foreshadows a directory
|
|
27
|
+
full of tiny files.
|
|
28
|
+
|
|
29
|
+
| Signal | What it points to |
|
|
30
|
+
|---|---|
|
|
31
|
+
| Output partition count high vs. data volume | Many tiny files on write |
|
|
32
|
+
| Read partitioning below `spark.sql.files.maxPartitionBytes` (128 MB) | Over-split input, excess filesystem I/O[^2][^3] |
|
|
33
|
+
|
|
34
|
+
## Why it matters
|
|
35
|
+
|
|
36
|
+
Every extra file is another open, close, and directory-list operation, and on a distributed
|
|
37
|
+
filesystem those are not cheap[^2]. The count compounds downstream: a stage that scatters
|
|
38
|
+
3000 tiny files hands the next job 3000 files to read back, so the metadata tax is paid twice
|
|
39
|
+
over[^4]. None of it is a single dramatic failure. The waste is spread thin across the file
|
|
40
|
+
count, which is exactly why it goes unnoticed until listing and I/O start to dominate.
|
|
41
|
+
|
|
42
|
+
## How to fix it
|
|
43
|
+
|
|
44
|
+
Repartition right before the write so the output lands in fewer, better-sized files.
|
|
45
|
+
|
|
46
|
+
- `coalesce()` is a shuffle-free merge: it fuses existing partitions with no data movement,
|
|
47
|
+
so `df.coalesce(100).write.parquet(...)` collapses 3000 tiny files into 100 reasonable
|
|
48
|
+
ones[^4]. It does not rebalance, though, so uneven inputs stay uneven once lumped together,
|
|
49
|
+
and pushing it to `coalesce(1)` kills parallelism by forcing one executor to do all the
|
|
50
|
+
work[^4].
|
|
51
|
+
- `repartition()` is a full reshuffle that buys even distribution at the cost of that
|
|
52
|
+
[shuffle](#shuffle). Reach for it when the distribution itself needs fixing, not just the file
|
|
53
|
+
count[^4].
|
|
54
|
+
|
|
55
|
+
> **PySpark:** both are one-line calls on a DataFrame: `df.coalesce(100)` for a shuffle-free
|
|
56
|
+
> merge, or `df.repartition(100)` when the data also needs rebalancing.
|
|
57
|
+
|
|
58
|
+
```python
|
|
59
|
+
# Collapse many tiny output files without a shuffle (distribution already even)
|
|
60
|
+
df.coalesce(100).write.parquet(path)
|
|
61
|
+
|
|
62
|
+
# Full reshuffle to a target count when the distribution itself needs fixing
|
|
63
|
+
df.repartition(100).write.parquet(path)
|
|
64
|
+
```
|
|
65
|
+
|
|
66
|
+
For runtime control, [Adaptive Query Execution](#aqe) can merge tiny shuffle partitions on its own.
|
|
67
|
+
With `spark.sql.adaptive.enabled` and `spark.sql.adaptive.coalescePartitions.enabled` (default
|
|
68
|
+
true) set, AQE coalesces contiguous shuffle partitions toward
|
|
69
|
+
`spark.sql.adaptive.advisoryPartitionSizeInBytes` (default 64 MB)[^4][^5]. Just mind the
|
|
70
|
+
scope: AQE only engages after the first shuffle, so it fixes shuffle-output partition counts
|
|
71
|
+
but will not repair input-side partitioning or a bad file layout on disk[^6].
|
|
72
|
+
|
|
73
|
+
## Confidence
|
|
74
|
+
|
|
75
|
+
The core inference, that one output file per partition times a high
|
|
76
|
+
partition count yields many tiny files, is grounded directly in Spark's write behavior[^4],
|
|
77
|
+
and the mitigation levers (`coalesce()`, `repartition()`, and AQE coalescing) are documented
|
|
78
|
+
Spark features[^4][^5].
|
|
79
|
+
|
|
80
|
+
## Limitations / false-positive risk
|
|
81
|
+
|
|
82
|
+
A high file count can be perfectly legitimate: partitioned output deliberately fans data
|
|
83
|
+
across many files by partition column, so a large count there is by design, not a defect.
|
|
84
|
+
And AQE coalescing is not a cure-all here, since it only affects post-shuffle partitions and
|
|
85
|
+
leaves the input file layout untouched[^6]. The signal reflects the shape, not intent, so
|
|
86
|
+
confirm the write is not an intended partitioned layout before acting.
|
|
87
|
+
|
|
88
|
+
|
|
89
|
+
## Related
|
|
90
|
+
|
|
91
|
+
- **Partition sizing:** [Partitioning](#partitioning)
|
|
92
|
+
- **Too many small tasks:** [Tiny Tasks](#bottleneck-tiny-tasks)
|
|
93
|
+
|
|
94
|
+
[^1]: *Spark: The Definitive Guide*, Chambers & Zaharia, ch. 19
|
|
95
|
+
[^2]: *Learning Spark, 2nd Edition*, Damji, Wenig, Das, Lee, ch. 7
|
|
96
|
+
[^3]: [SQLConf.scala](https://raw.githubusercontent.com/apache/spark/v3.5.0/sql/catalyst/src/main/scala/org/apache/spark/sql/internal/SQLConf.scala)
|
|
97
|
+
[^4]: [Spark Partitions](https://luminousmen.com/post/spark-partitions)
|
|
98
|
+
[^5]: [Performance Tuning (Spark SQL, DataFrames and Datasets Guide)](https://spark.apache.org/docs/latest/sql-performance-tuning.html)
|
|
99
|
+
[^6]: [The Apache Spark Optimization Checklist](https://luminousmen.com/post/the-apache-spark-optimization-checklist)
|
|
@@ -0,0 +1,114 @@
|
|
|
1
|
+
# Memory / Disk Spill
|
|
2
|
+
|
|
3
|
+
<span class="tag">SPILL</span>
|
|
4
|
+
|
|
5
|
+
## What it is
|
|
6
|
+
|
|
7
|
+
Spill happens when a task's share of execution memory runs out during a sort or hash
|
|
8
|
+
aggregation. Each task gets its own `TaskMemoryManager`, which arbitrates the shared execution
|
|
9
|
+
pool across every task running concurrently on an executor; with `n` tasks running
|
|
10
|
+
concurrently, each is allowed to allocate somewhere between `1/(2n)` and `1/n` of total
|
|
11
|
+
execution memory: a soft limit, not a hard stop.[^1] When a task's sort or hash-aggregation
|
|
12
|
+
operator keeps requesting execution-memory pages it can't get, Spark doesn't fail immediately:
|
|
13
|
+
it blocks the requesting task, triggers a spill of that task's in-memory data structure to
|
|
14
|
+
disk, or, in the worst case, throws an `OutOfMemoryError`.[^1]
|
|
15
|
+
|
|
16
|
+
The spill path runs through Spark's map/shuffle I/O machinery: shuffle partitions created by
|
|
17
|
+
wide transformations like `groupBy()` or `join()` spill to the executors' local disks, at the
|
|
18
|
+
location set by `spark.local.directory`.[^2] SQL physical operators apply the same idea with
|
|
19
|
+
their own guardrails: `SortMergeJoinExec`'s in-memory buffer and the cartesian-product
|
|
20
|
+
operator's buffer both spill once they cross a configured row-count threshold, which by
|
|
21
|
+
default is `spark.shuffle.spill.numElementsForceSpillThreshold`.[^3] Whatever the trigger, the
|
|
22
|
+
resulting spilled bytes are exposed on the task metrics as `memoryBytesSpilled`, visible in the
|
|
23
|
+
Spark UI and event log.[^4]
|
|
24
|
+
|
|
25
|
+
## How it's detected
|
|
26
|
+
|
|
27
|
+
Detection: any `memoryBytesSpilled > 0` → Warning. The classification is the actionable
|
|
28
|
+
signal.
|
|
29
|
+
|
|
30
|
+
Spill classification:
|
|
31
|
+
- `skew`: ≥ 80% of tasks have zero spill; fix: address [task skew](#bottleneck-skew), not memory.
|
|
32
|
+
- `volume`: < 20% of tasks have zero spill; fix: more partitions or more memory.
|
|
33
|
+
- `unclassified`: neither condition; treat as volume.
|
|
34
|
+
|
|
35
|
+
<img class="light-only" src="../diagrams/spill-classification.svg" alt="How a nonzero memoryBytesSpilled is classified as skew, volume, or unclassified from the share of tasks with zero spill, and the fix each classification points to.">
|
|
36
|
+
<img class="dark-only" src="../diagrams/spill-classification.dark.svg" alt="How a nonzero memoryBytesSpilled is classified as skew, volume, or unclassified from the share of tasks with zero spill, and the fix each classification points to.">
|
|
37
|
+
|
|
38
|
+
## Why it matters
|
|
39
|
+
|
|
40
|
+
Spill is the fallback Spark reaches for once a task can no longer get the execution memory
|
|
41
|
+
it's asking for: rather than failing outright, it blocks the task, writes its buffered data to
|
|
42
|
+
disk, or (if that's not enough) throws an `OutOfMemoryError`.[^1] A task that's spilling has
|
|
43
|
+
already given up pure in-memory speed to keep running at all, which is why the skew-vs-volume
|
|
44
|
+
split above is the actionable part of the signal: the two failure shapes call for opposite
|
|
45
|
+
fixes.
|
|
46
|
+
|
|
47
|
+
## How to fix it
|
|
48
|
+
|
|
49
|
+
- **Skew**: When only a few tasks spill, the problem is the shape of the data, not the size
|
|
50
|
+
of the memory pool. `repartition()` is the tool for this: reach for it specifically to fix a
|
|
51
|
+
lopsided partition distribution or to increase partition count[^5], in contrast to
|
|
52
|
+
`coalesce()`, which never rebalances skewed data because it just stacks existing partitions
|
|
53
|
+
together: partitions that were uneven going in are still uneven coming out.[^6] For skewed
|
|
54
|
+
joins, [Adaptive Query Execution](#aqe) can detect and split[^9] oversized partitions automatically: a
|
|
55
|
+
partition counts as skewed once it's larger than
|
|
56
|
+
`spark.sql.adaptive.skewJoin.skewedPartitionFactor` (default `5.0`) times the median
|
|
57
|
+
partition size, and also larger than
|
|
58
|
+
`spark.sql.adaptive.skewJoin.skewedPartitionThresholdInBytes` (default `256 MB`).[^7][^8]
|
|
59
|
+
Before AQE existed, the manual equivalent was salting: adding a prefix to the skewed keys to
|
|
60
|
+
make the same key look different, then adjusting the data distribution accordingly.[^9]
|
|
61
|
+
Databricks' `/*+ SKEW(...) */` hint lets you name the skewed relation and specific key values
|
|
62
|
+
directly, so the planner targets just those keys instead of relying on automatic
|
|
63
|
+
detection.[^10]
|
|
64
|
+
- **Volume**: When most tasks spill, the fix is more room to work with: more partitions, more
|
|
65
|
+
memory, or both. `repartition(n)` guarantees exactly `n` output partitions via a hash
|
|
66
|
+
shuffle[^11], and since Spark's per-task startup overhead is low (unlike MapReduce's), the
|
|
67
|
+
general bias is to err toward more partitions rather than fewer.[^12]
|
|
68
|
+
`spark.sql.files.maxPartitionBytes` (default `128 MB`) is the equivalent lever on the input
|
|
69
|
+
side, governing how much file data gets packed into each input partition before a shuffle
|
|
70
|
+
even happens.[^6] On the memory side, execution memory is a fraction of the JVM heap.
|
|
71
|
+
`spark.memory.fraction` (default `0.6`) sizes the shared execution/storage region as
|
|
72
|
+
`(JVM heap − 300 MiB) × spark.memory.fraction`[^13]. Because each task's slice of that
|
|
73
|
+
region is capped by the `TaskMemoryManager` at between `1/(2n)` and `1/n` of the total,[^1]
|
|
74
|
+
giving executors more memory, or running fewer concurrent tasks on each one, directly raises
|
|
75
|
+
the ceiling before spill kicks in.
|
|
76
|
+
|
|
77
|
+
For the **volume** case, give tasks more room: defaults shown, comments say which way to move:
|
|
78
|
+
|
|
79
|
+
```properties
|
|
80
|
+
# Create more, smaller input partitions before the shuffle (default 128m)
|
|
81
|
+
spark.sql.files.maxPartitionBytes=128m
|
|
82
|
+
|
|
83
|
+
# Execution/storage share of (JVM heap - 300MiB) (default 0.6);
|
|
84
|
+
# raising it grows execution memory but starves the untracked user-memory region
|
|
85
|
+
spark.memory.fraction=0.6
|
|
86
|
+
```
|
|
87
|
+
|
|
88
|
+
For the **skew** case (only a few tasks spill), the fix is `df.repartition(n)`, not more memory: see [Partitioning](#partitioning).
|
|
89
|
+
|
|
90
|
+
## Confidence
|
|
91
|
+
|
|
92
|
+
The core detection (any `memoryBytesSpilled > 0`) and the skew-vs-volume classification are validated: both the skew branch (most tasks spill nothing, so the shape of the data is the problem) and the volume branch (nearly every task spills, so the pool is too small) map to a well-understood fix, and neither needs further validation. <span class="tag">EXPERIMENTAL</span> When a spill matches neither shape, it falls back to a low-confidence unclassified finding that still requires validation, because there is no confirmed cause to act on; the page defaults it to the volume remedy, but that is a guess, not a diagnosis.
|
|
93
|
+
|
|
94
|
+
## Limitations / false-positive risk
|
|
95
|
+
|
|
96
|
+
Some spill is normal. A large aggregation or join can spill by design once its working set outgrows execution memory, and a small spill on a healthy stage rarely repays the effort of chasing it. The raw `memoryBytesSpilled > 0` trigger fires on both the actionable and the routine cases, so treat a small spill as informational until the skew-vs-volume split says otherwise.
|
|
97
|
+
|
|
98
|
+
## Related
|
|
99
|
+
|
|
100
|
+
- **Why it happens:** [Memory Management](#memory-model), [Partitioning](#partitioning)
|
|
101
|
+
|
|
102
|
+
[^1]: [Diving into Spark Memory Management](https://luminousmen.com/post/dive-into-spark-memory)
|
|
103
|
+
[^2]: *Learning Spark, 2nd Edition*, Damji, Wenig, Das, Lee, ch. 7
|
|
104
|
+
[^3]: [SQLConf.scala](https://raw.githubusercontent.com/apache/spark/v3.5.0/sql/catalyst/src/main/scala/org/apache/spark/sql/internal/SQLConf.scala)
|
|
105
|
+
[^4]: [Monitoring and Instrumentation](https://spark.apache.org/docs/latest/monitoring.html#spark-history-server)
|
|
106
|
+
[^5]: *Spark: The Definitive Guide*, Chambers & Zaharia, ch. 19
|
|
107
|
+
[^6]: [Spark Partitions](https://luminousmen.com/post/spark-partitions)
|
|
108
|
+
[^7]: [Configuration (Spark)](https://spark.apache.org/docs/latest/configuration.html)
|
|
109
|
+
[^8]: [Performance Tuning (Spark SQL, DataFrames and Datasets Guide)](https://spark.apache.org/docs/latest/sql-performance-tuning.html)
|
|
110
|
+
[^9]: [SPARK-29544: Optimize Skewed Join in SQL Adaptive Execution](https://issues.apache.org/jira/browse/SPARK-29544)
|
|
111
|
+
[^10]: [Skew Join Optimization](https://docs.databricks.com/aws/en/archive/legacy/skew-join)
|
|
112
|
+
[^11]: [RDD.scala](https://raw.githubusercontent.com/apache/spark/v3.5.0/core/src/main/scala/org/apache/spark/rdd/RDD.scala)
|
|
113
|
+
[^12]: [How to Tune Your Apache Spark Jobs (Part 2): Cloudera Engineering Blog](https://blog.cloudera.com/how-to-tune-your-apache-spark-jobs-part-2/)
|
|
114
|
+
[^13]: [Spark Tuning Guide](https://spark.apache.org/docs/latest/tuning.html)
|
|
@@ -0,0 +1,103 @@
|
|
|
1
|
+
# Stragglers
|
|
2
|
+
|
|
3
|
+
<span class="tag">STRAG</span>
|
|
4
|
+
|
|
5
|
+
## What it is
|
|
6
|
+
|
|
7
|
+
A straggler is one task (or a small handful of tasks) that runs far longer than the rest of
|
|
8
|
+
the tasks in its stage, even when the stage is otherwise healthy. Because a stage only
|
|
9
|
+
completes once its last task finishes, a single straggler holds up the whole stage, and every
|
|
10
|
+
other executor sits idle waiting for it to catch up.
|
|
11
|
+
|
|
12
|
+
## How it's detected
|
|
13
|
+
|
|
14
|
+
A task counts as a straggler when speculative tasks fired for it, or when more than 5% of the
|
|
15
|
+
tasks in a stage run at least 4× the stage's median task duration; this rule only applies once
|
|
16
|
+
a stage has at least 10 tasks.
|
|
17
|
+
|
|
18
|
+
"Duration" here is `executorRunTime`: elapsed wall-clock time, not CPU time, and it already
|
|
19
|
+
includes any time the task spent blocked fetching shuffle data[^1]. That matters when tracking
|
|
20
|
+
down why a task is slow: a long `executorRunTime` doesn't necessarily mean more compute
|
|
21
|
+
happened. A [garbage-collection pause](#bottleneck-gc) is counted inside it rather than added on top
|
|
22
|
+
(`jvmGCTime` is a subset of `executorRunTime`, not additive[^1]), and a task waiting on a
|
|
23
|
+
remote shuffle block it needs next shows up in `shuffleReadMetrics.fetchWaitTime`, which only
|
|
24
|
+
counts genuine blocking time, not blocks being prefetched in the background[^1].
|
|
25
|
+
|
|
26
|
+
The "speculative tasks fired" half of the rule is Spark's own detector: once
|
|
27
|
+
`spark.speculation.quantile` (default `0.9`) of a stage's tasks finish, Spark compares each
|
|
28
|
+
remaining task's duration against `spark.speculation.multiplier` (default `3`) times the
|
|
29
|
+
median of the tasks that already finished, subject to a `spark.speculation.minTaskRuntime`
|
|
30
|
+
floor (default `100ms`) so short tasks aren't flagged just for being slower than a tiny
|
|
31
|
+
median[^2]. Speculation itself is off by default (`spark.speculation` defaults to `false`), so
|
|
32
|
+
this half of the rule only fires on stages where it's been turned on[^2].
|
|
33
|
+
|
|
34
|
+
Watch for overlap with [task skew](#bottleneck-skew): an unevenly distributed key sends one
|
|
35
|
+
partition far more data than its peers, and that partition's task will trip this same duration
|
|
36
|
+
threshold even though the underlying cause is data volume, not a slow host or a GC pause.
|
|
37
|
+
Check the skew signals before assuming the latter.
|
|
38
|
+
|
|
39
|
+
## Why it matters
|
|
40
|
+
|
|
41
|
+
A straggler wastes cluster capacity the same way a skewed stage does: every executor other
|
|
42
|
+
than the one running the slow task finishes early and idles, while total job time still
|
|
43
|
+
tracks the single slowest task. Because a straggler's duration can be inflated by a GC
|
|
44
|
+
pause[^1] or a slow [shuffle](#shuffle) fetch[^1] rather than genuinely more work, it's worth checking
|
|
45
|
+
those angles (and whether the real cause is skew rather than anything task-local) before
|
|
46
|
+
assuming a hardware explanation.
|
|
47
|
+
|
|
48
|
+
## How to fix it
|
|
49
|
+
|
|
50
|
+
- Enable speculative execution: `spark.speculation` is `false` by default, so nothing reruns
|
|
51
|
+
a slow task automatically until it's turned on[^2].
|
|
52
|
+
- Tune the trigger: `spark.speculation.quantile` (default `0.9`) sets how much of the stage
|
|
53
|
+
must finish before speculation kicks in, and `spark.speculation.multiplier` (default `3`)
|
|
54
|
+
sets how many times slower than the median a task must be. For stages with very few tasks,
|
|
55
|
+
`spark.speculation.task.duration.threshold` (available since 3.0.0) gives an absolute-duration
|
|
56
|
+
trigger instead of relying on the median[^2].
|
|
57
|
+
- Since Spark 3.4, `spark.speculation.efficiency.enabled` (default `true`) adds a filter so a
|
|
58
|
+
task is only speculated if its data-process rate is below the stage average (times
|
|
59
|
+
`spark.speculation.efficiency.processRateMultiplier`, default `0.75`) or its duration exceeds
|
|
60
|
+
`spark.speculation.efficiency.longRunTaskFactor` (default `2`) times the same time threshold.
|
|
61
|
+
This avoids wasting a duplicate task slot on work that's simply doing more, not running
|
|
62
|
+
slower[^2].
|
|
63
|
+
- If the cause is really a skewed key, treat it as [task skew](#bottleneck-skew) instead: AQE's
|
|
64
|
+
skew-join optimization automatically splits a partition once it's larger than
|
|
65
|
+
`spark.sql.adaptive.skewJoin.skewedPartitionFactor` (default `5.0`) times the median
|
|
66
|
+
partition size and above `spark.sql.adaptive.skewJoin.skewedPartitionThresholdInBytes`
|
|
67
|
+
(default `256 MB`)[^2][^3]. Manual salting (appending a random prefix to the skewed key so
|
|
68
|
+
it spreads across more partitions) is the pre-AQE fallback, though the design doc behind
|
|
69
|
+
AQE's skew handling calls salting and other manual approaches limited compared to the
|
|
70
|
+
automatic option[^4].
|
|
71
|
+
|
|
72
|
+
> **PySpark:** speculation settings can be set on the session directly, no `spark-submit` flag
|
|
73
|
+
> needed: `spark.conf.set("spark.speculation", "true")`, then the quantile/multiplier
|
|
74
|
+
> equivalents the same way.
|
|
75
|
+
|
|
76
|
+
Speculation config: defaults shown, plus the absolute-duration trigger for small stages:
|
|
77
|
+
|
|
78
|
+
```properties
|
|
79
|
+
# Speculatively relaunch straggler tasks (OFF by default)
|
|
80
|
+
spark.speculation=true
|
|
81
|
+
spark.speculation.quantile=0.9 # default; portion of tasks finished before speculation begins
|
|
82
|
+
spark.speculation.multiplier=3 # default; multiple of median duration that marks a task slow
|
|
83
|
+
|
|
84
|
+
# Absolute-duration trigger for stages with very few tasks (since Spark 3.0)
|
|
85
|
+
spark.speculation.task.duration.threshold=10s # 10s is an example
|
|
86
|
+
```
|
|
87
|
+
|
|
88
|
+
## Confidence
|
|
89
|
+
|
|
90
|
+
Validated.
|
|
91
|
+
|
|
92
|
+
## Limitations / false-positive risk
|
|
93
|
+
|
|
94
|
+
The 4x-median rule flags a slow task, but slow is not the same as broken. The same threshold trips on a skewed key that simply has more data to process, or on a task that spent its time in a GC pause rather than doing extra work, so a flagged task is not automatically a slow host. Small stages make this worse: with only a handful of tasks the median is unstable, and one moderately slow task can look like a straggler against a median computed from too few peers.
|
|
95
|
+
|
|
96
|
+
## Related
|
|
97
|
+
|
|
98
|
+
- **When the real cause is a skewed key:** [Partitioning](#partitioning), [Adaptive Query Execution](#aqe)
|
|
99
|
+
|
|
100
|
+
[^1]: [Monitoring and Instrumentation](https://spark.apache.org/docs/latest/monitoring.html#spark-history-server)
|
|
101
|
+
[^2]: [Configuration (Spark)](https://spark.apache.org/docs/latest/configuration.html)
|
|
102
|
+
[^3]: [Optimizing Skew Join (Spark SQL, DataFrames and Datasets Guide)](https://spark.apache.org/docs/latest/sql-performance-tuning.html#optimizing-skew-join)
|
|
103
|
+
[^4]: [SPARK-29544: Optimize Skewed Join at Runtime with New Adaptive Execution](https://issues.apache.org/jira/browse/SPARK-29544)
|
|
@@ -0,0 +1,94 @@
|
|
|
1
|
+
# Tiny Tasks
|
|
2
|
+
|
|
3
|
+
<span class="tag">TINY</span>
|
|
4
|
+
|
|
5
|
+
## What it is
|
|
6
|
+
|
|
7
|
+
Every task pays a fixed placement and serialization cost before it does any real work. Spark
|
|
8
|
+
schedules around data locality rather than moving data to code: "Spark builds its scheduling
|
|
9
|
+
around this general principle" that shipping serialized code is cheaper than shipping data,
|
|
10
|
+
preferring `PROCESS_LOCAL` locality down to `ANY` and waiting a configurable timeout for a busy
|
|
11
|
+
CPU to free up before shipping data to a farther executor[^1]. On top of that placement cost,
|
|
12
|
+
every task also pays a serialization cost for its closure and, if not using Kryo, for its data:
|
|
13
|
+
Java's default `ObjectOutputStream`-based serialization "is flexible but often quite slow, and
|
|
14
|
+
leads to large serialized formats," while Kryo is "significantly faster and more compact than
|
|
15
|
+
Java serialization (often as much as 10x)"[^1]. None of that overhead is large per task: Spark's
|
|
16
|
+
own guidance leans toward more, smaller tasks rather than fewer, larger ones, unlike MapReduce:
|
|
17
|
+
"it's almost always better to err on the side of a larger number of tasks (and thus partitions)...
|
|
18
|
+
The difference stems from the fact that MapReduce has a high startup overhead for tasks, while
|
|
19
|
+
Spark does not"[^2]. But once partitions get small enough, that per-task overhead stops being
|
|
20
|
+
negligible and starts to dominate.
|
|
21
|
+
|
|
22
|
+
## How it's detected
|
|
23
|
+
|
|
24
|
+
The failure mode is partitions so small that scheduling and serialization overhead outweighs the
|
|
25
|
+
actual work. With Spark's default of 200 shuffle partitions applied to a dataset of only a few
|
|
26
|
+
megabytes, "those 200 partitions will each get like ten rows. Tasks become microscopic. Most of
|
|
27
|
+
your CPUs will just be sitting there doing nothing, wasting cluster hours"[^3]. The same
|
|
28
|
+
small-partition penalty shows up on the [shuffle-read side](#bottleneck-shuffle): production measurements found "the
|
|
29
|
+
average shuffle block size is only 10s of KBs, which leads to delayed shuffle data fetch," and
|
|
30
|
+
the shuffle reduce stages with the largest fetch delays (over 30 seconds per task) were
|
|
31
|
+
consistently the ones with small block sizes[^4]. A high task count paired with very short median
|
|
32
|
+
task duration, and/or very small shuffle block sizes on the read side, are the signals to look
|
|
33
|
+
for.
|
|
34
|
+
|
|
35
|
+
## Why it matters
|
|
36
|
+
|
|
37
|
+
Tiny tasks waste cluster capacity even though no single task is a straggler the way
|
|
38
|
+
[skewed](#bottleneck-skew) or [straggler](#bottleneck-straggler) tasks are: the cost here is
|
|
39
|
+
spread evenly across thousands of tasks, each individually cheap but collectively adding up in
|
|
40
|
+
scheduling and serialization overhead, while cores that could be doing other useful work sit
|
|
41
|
+
mostly idle between task launches[^3].
|
|
42
|
+
|
|
43
|
+
## How to fix it
|
|
44
|
+
|
|
45
|
+
- `coalesce()` merges existing partitions without a shuffle ("no data movement, no shuffle")
|
|
46
|
+
and is typically used right before a write, e.g. `df.coalesce(100).write.parquet(...)`, to
|
|
47
|
+
collapse many tiny output files into a manageable number cheaply[^3]. Because it only stacks
|
|
48
|
+
partitions together rather than redistributing data, it doesn't fix an uneven underlying
|
|
49
|
+
distribution (uneven input partitions stay uneven, just grouped) and pushing it too far
|
|
50
|
+
(`coalesce(1)`) kills parallelism by funneling all work onto one executor while the rest sit
|
|
51
|
+
idle[^3].
|
|
52
|
+
- `repartition()` performs a full reshuffle, giving control and rebalancing that Spark won't do
|
|
53
|
+
on its own: with `df.repartition(200)`, "you're paying for predictability"[^3]. Use it when
|
|
54
|
+
the partitioning itself needs to be fixed or rebalanced, not merely reduced in count, accepting
|
|
55
|
+
the shuffle cost to get there.
|
|
56
|
+
- There's a middle option, `coalesce(N, shuffle=True)`, which "acts more like a repartition, but
|
|
57
|
+
leaning toward reduction... not free (you pay for the shuffle cost) but you get better
|
|
58
|
+
distribution and fewer partitions"[^3].
|
|
59
|
+
- Rule of thumb: reach for `coalesce()` when only the partition count needs to shrink and the
|
|
60
|
+
existing distribution is already reasonably even (e.g. collapsing output files); reach for
|
|
61
|
+
`repartition()` when the distribution itself is the problem.
|
|
62
|
+
|
|
63
|
+
> **PySpark:** both are one-line calls on a DataFrame: `df.coalesce(100)` for a shuffle-free
|
|
64
|
+
> merge, `df.repartition(200)` for a full reshuffle, or `df.coalesce(100, shuffle=True)` for the
|
|
65
|
+
> shuffled middle option.
|
|
66
|
+
|
|
67
|
+
The three sizing calls, side by side:
|
|
68
|
+
|
|
69
|
+
```python
|
|
70
|
+
# Collapse many tiny output files without a shuffle (use when distribution is already even)
|
|
71
|
+
df.coalesce(100).write.parquet(path)
|
|
72
|
+
|
|
73
|
+
# Full reshuffle to a target count, use when the distribution itself needs rebalancing
|
|
74
|
+
df = df.repartition(200)
|
|
75
|
+
|
|
76
|
+
# Middle ground: reduce partition count but still rebalance (pays a shuffle)
|
|
77
|
+
df = df.coalesce(100, shuffle=True)
|
|
78
|
+
```
|
|
79
|
+
|
|
80
|
+
## Limitations / false-positive risk
|
|
81
|
+
|
|
82
|
+
Very short tasks are not always a problem. A tiny final stage can be perfectly fine, and one
|
|
83
|
+
that just writes a small result set doesn't need to be resized. The scheduling-overhead penalty
|
|
84
|
+
only bites when tiny tasks dominate the stage, so treat a handful of short tasks as noise rather
|
|
85
|
+
than a finding.
|
|
86
|
+
|
|
87
|
+
## Related
|
|
88
|
+
|
|
89
|
+
- **Partition sizing:** [Partitioning](#partitioning)
|
|
90
|
+
|
|
91
|
+
[^1]: [Tuning (Spark)](https://spark.apache.org/docs/latest/tuning.html)
|
|
92
|
+
[^2]: [How to Tune Your Apache Spark Jobs (Part 2): Cloudera](https://blog.cloudera.com/how-to-tune-your-apache-spark-jobs-part-2/)
|
|
93
|
+
[^3]: [Spark Partitions](https://luminousmen.com/post/spark-partitions)
|
|
94
|
+
[^4]: [SPARK-30602](https://issues.apache.org/jira/browse/SPARK-30602)
|
|
@@ -0,0 +1,90 @@
|
|
|
1
|
+
# Executor Utilization
|
|
2
|
+
|
|
3
|
+
<span class="tag">UTIL</span>
|
|
4
|
+
|
|
5
|
+
## What it is
|
|
6
|
+
|
|
7
|
+
Executor utilization measures how much of the cluster's allocated executor capacity a job actually keeps busy. When the average number of active executors trails the peak number allocated, the cluster is holding compute (cores and memory) that isn't running any tasks.
|
|
8
|
+
|
|
9
|
+
## How it's detected
|
|
10
|
+
|
|
11
|
+
The signal is the ratio of average active executors to the peak active executor count observed over the job's lifetime:
|
|
12
|
+
|
|
13
|
+
| avg active executors / peak | Level |
|
|
14
|
+
|---|---|
|
|
15
|
+
| < 60% | Info |
|
|
16
|
+
| < 40% | Warning |
|
|
17
|
+
| < 20% | Critical |
|
|
18
|
+
|
|
19
|
+
## Why it matters
|
|
20
|
+
|
|
21
|
+
Idle executors are allocation you're paying for without getting work done. One common cause is too little task parallelism to occupy the allocated cores: the guidance is to keep at least as many partitions as there are cores across the executors, so no core sits idle[^1]. Parallelism can also be lost by accident rather than by under-partitioning upfront: because `coalesce` is a narrow transformation, reducing partition count with it forces the *entire* upstream stage down to the reduced parallelism, not just the coalesce step, trading a shuffle for lost concurrency[^2]. Separately, under dynamic allocation, a workload with many small tasks can end up over-provisioned: by default it requests enough executors to maximize parallelism for the task count, and with small tasks that can mean some executors "might not even do any work," wasting resources on allocation overhead[^3].
|
|
22
|
+
|
|
23
|
+
## How to fix it
|
|
24
|
+
|
|
25
|
+
- Reach for `repartition` (not `coalesce`) when you need to raise partition count or rebalance data: `coalesce` only avoids a shuffle when shrinking partition count, and doing so also drags the entire upstream stage down to the reduced parallelism[^4][^2].
|
|
26
|
+
- Size partitions so the task count is at least the core count available across executors, to avoid leaving cores idle[^1].
|
|
27
|
+
- Under dynamic allocation, lower `spark.dynamicAllocation.executorAllocationRatio` below its default of `1.0` to scale back the number of executors requested for workloads made up of many small tasks[^3].
|
|
28
|
+
- Dynamic allocation requires shuffle tracking, the external shuffle service, or shuffle-block decommissioning to be enabled as a precondition[^3]; without one of them, an executor removed mid-shuffle takes its unfetched shuffle files with it, forcing a recompute[^5], which undercuts using dynamic allocation to shed idle executors in the first place. `spark.dynamicAllocation.shuffleTracking.enabled` defaults to `true` since Spark 3.0, satisfying that precondition without needing a separate external shuffle service[^3].
|
|
29
|
+
|
|
30
|
+
Restore parallelism after a heavy filter, and rein in over-provisioning for many-small-task jobs:
|
|
31
|
+
|
|
32
|
+
```python
|
|
33
|
+
# repartition (not coalesce) rebalances and can raise partition count; aim >= total executor cores
|
|
34
|
+
df = df.filter(heavy_predicate).repartition(spark.sparkContext.defaultParallelism)
|
|
35
|
+
|
|
36
|
+
# Scale back executors requested for many-small-task workloads (default ratio 1.0)
|
|
37
|
+
spark.conf.set("spark.dynamicAllocation.executorAllocationRatio", "0.5") # 0.5 is an example
|
|
38
|
+
```
|
|
39
|
+
|
|
40
|
+
## Limitations / false-positive risk
|
|
41
|
+
|
|
42
|
+
A low average-to-peak ratio isn't always waste. A bursty or I/O-bound job legitimately holds executors while tasks wait on external systems rather than burning cores, and the ratio is sensitive to short stages, where a brief spike in allocation skews the average without meaning the cluster was genuinely idle.
|
|
43
|
+
|
|
44
|
+
## Caching opportunity
|
|
45
|
+
|
|
46
|
+
<span class="tag">CACHE</span>
|
|
47
|
+
|
|
48
|
+
When the same DataFrame, RDD, or input is scanned more than once, low utilization can trace back to repeated recomputation rather than idle cores. Spark keeps nothing between actions: transformations only build a DAG, and once an action finishes its intermediate results are discarded[^6]. Call a second action on the same logic and Spark re-runs the whole DAG from the source, which can mean re-reading a terabyte from S3, re-reading Kafka, or repeating expensive decompression[^6]. Fork that logic into two pipeline branches and you sign up to recompute everything twice[^6].
|
|
49
|
+
|
|
50
|
+
<img class="light-only" src="../diagrams/duplicate-plan-subtree.svg" alt="How two branches that repeat the same scan and operators each recompute it, until a shared cached or reused node lets both read one materialized result.">
|
|
51
|
+
<img class="dark-only" src="../diagrams/duplicate-plan-subtree.dark.svg" alt="How two branches that repeat the same scan and operators each recompute it, until a shared cached or reused node lets both read one materialized result.">
|
|
52
|
+
|
|
53
|
+
[`cache()` and `persist()`](#caching) are how you stop that. Persisting materializes the RDD (usually in memory on the executors) so it can be reused within the job, while Spark keeps its lineage to recompute any lost partition[^7]. For a dataset you read repeatedly, this is one of the most useful optimizations available: it parks a DataFrame, table, or RDD in temporary storage across the executors and makes later reads fast[^8]. Materialization is lazy and per-block, so only an action that touches every partition (`count()` or a full write) caches all of it; a subset scan like `take(10)` leaves a partial cache[^6].
|
|
54
|
+
|
|
55
|
+
`cache()` is shorthand; `persist()` takes a `StorageLevel` and lets you pick the tradeoff:
|
|
56
|
+
|
|
57
|
+
| Level | Tradeoff |
|
|
58
|
+
|---|---|
|
|
59
|
+
| `MEMORY_ONLY` | Fastest reads (deserialized JVM objects), but partitions that don't fit are recomputed on the fly each time[^8]. |
|
|
60
|
+
| `MEMORY_AND_DISK` | Spills the overflow to disk instead of recomputing; Spark's default caching strategy and fine for most pipelines[^9]. For DataFrames, `.cache()` maps to this[^6]. |
|
|
61
|
+
| Serialized (`_SER`) | Byte arrays instead of raw objects: smaller footprint, slower reads. Reach for it when memory is tight[^9]. |
|
|
62
|
+
| Replicated (`_2`) | A second copy for fast fault recovery instead of waiting on recomputation, at twice the space[^10][^9]. |
|
|
63
|
+
|
|
64
|
+
Whether to spill to disk hinges on recomputation cost: reading a block back from disk only beats recomputing it when the work that produced the data is expensive or filters out a large fraction, so the RDD guide says don't enable disk otherwise[^10].
|
|
65
|
+
|
|
66
|
+
Caching is not free, and several cases don't warrant it:
|
|
67
|
+
|
|
68
|
+
- **Single use.** Caching adds serialization, deserialization, and storage cost, so caching data you read once only slows you down[^8].
|
|
69
|
+
- **Data larger than storage memory,** or a cheap transformation that isn't reused often regardless of size[^11].
|
|
70
|
+
- **Memory pressure.** Cache memory is memory taken from processing, and under the default spill strategy cached data can land on slower disk, so caching can cost more than just re-reading the source[^9].
|
|
71
|
+
- **Lost optimizer freedom.** Once a dataset is cached, Catalyst works on the in-memory copy and can no longer push filters down to the source[^9].
|
|
72
|
+
|
|
73
|
+
A persisted RDD you've stopped using still occupies memory until the app ends or eviction forces it out, so call `unpersist()` to reclaim it deliberately, which matters most on shared clusters[^9].
|
|
74
|
+
|
|
75
|
+
## Related
|
|
76
|
+
|
|
77
|
+
- **Task parallelism:** [Partitioning](#partitioning)
|
|
78
|
+
- **Executor sizing:** [Cluster Tuning](#cluster-config)
|
|
79
|
+
|
|
80
|
+
[^1]: *Learning Spark, 2nd Edition*, Damji, Wenig, Das & Lee, ch. 7
|
|
81
|
+
[^2]: *High Performance Spark, 2nd Edition*, Karau, Polak & Warren, ch. 8
|
|
82
|
+
[^3]: [Configuration — Spark](https://spark.apache.org/docs/latest/configuration.html)
|
|
83
|
+
[^4]: *Spark: The Definitive Guide*, Chambers & Zaharia, ch. 19
|
|
84
|
+
[^5]: [Job Scheduling — Dynamic Resource Allocation](https://spark.apache.org/docs/latest/job-scheduling.html)
|
|
85
|
+
[^6]: [Explaining the Mechanics of Spark Caching](https://luminousmen.com/post/explaining-the-mechanics-of-spark-caching)
|
|
86
|
+
[^7]: *High Performance Spark, 2nd Edition*, Karau, Polak & Warren, ch. 2
|
|
87
|
+
[^8]: *Spark: The Definitive Guide*, Chambers & Zaharia, ch. 19
|
|
88
|
+
[^9]: [Spark Tips: Caching](https://luminousmen.com/post/spark-tips-caching)
|
|
89
|
+
[^10]: [RDD Programming Guide — Spark](https://spark.apache.org/docs/latest/rdd-programming-guide.html)
|
|
90
|
+
[^11]: *Learning Spark, 2nd Edition*, Damji, Wenig, Das & Lee, ch. 7
|
|
@@ -1,23 +1,16 @@
|
|
|
1
1
|
import { typeTag } from './format-utils.js';
|
|
2
2
|
|
|
3
|
-
//
|
|
4
|
-
//
|
|
5
|
-
// docs-site (VitePress) links, kept in its own file so the two unrelated doc
|
|
6
|
-
// systems don't blur into one.
|
|
3
|
+
// Single source for docs-site (VitePress) links, kept separate from docs-config.ts's vendor
|
|
4
|
+
// docs-panel surface so the two doc systems don't blur.
|
|
7
5
|
//
|
|
8
|
-
// The `.html` extension matters: it's what makes this link resolve in the
|
|
9
|
-
//
|
|
10
|
-
//
|
|
11
|
-
// no directory-index or extension-guessing fallback (unlike the Vite
|
|
12
|
-
// dev-server's docs-site middleware in vite.config.ts, which tries
|
|
13
|
-
// bare/`.html`/`index.html` candidates). VitePress's default build
|
|
14
|
-
// (`cleanUrls` unset, i.e. false) names each page's output file `<slug>.html`,
|
|
15
|
-
// so the extension-less form would 404 there.
|
|
6
|
+
// The `.html` extension matters: it's what makes this link resolve in the packaged server/ deploy
|
|
7
|
+
// mode, whose static file server matches a request path to a file exactly (no extension guessing).
|
|
8
|
+
// VitePress names each page <slug>.html, so the extension-less form would 404 there.
|
|
16
9
|
//
|
|
17
|
-
// No allowlist
|
|
18
|
-
//
|
|
19
|
-
//
|
|
20
|
-
//
|
|
10
|
+
// No allowlist needed (unlike isKnownDocAnchor): docs-site-tag-coverage.test.js fails CI if any
|
|
11
|
+
// TYPE_TAG_MAP value lacks a documented {#tag} heading.
|
|
12
|
+
//
|
|
13
|
+
// Relative, not /docs/: keeps working under any subpath the app is published at.
|
|
21
14
|
export function findingGuideUrl(type ) {
|
|
22
|
-
return
|
|
15
|
+
return `docs/user-guide/understanding-findings.html#${typeTag(type).toLowerCase()}`;
|
|
23
16
|
}
|
|
@@ -1,6 +1,5 @@
|
|
|
1
|
-
// §5 Efficiency/wastage model
|
|
2
|
-
//
|
|
3
|
-
// executor-bound waste, plus two theoretical floors. Mapped onto computeWallClock.
|
|
1
|
+
// §5 Efficiency/wastage model. DESIGN SPIKE:
|
|
2
|
+
// decomposes available compute-hours into driver-bound vs executor-bound waste, plus two floors.
|
|
4
3
|
import { computeWallClock } from './wall-clock.js';
|
|
5
4
|
import { computeTotalCores } from './core-count.js';
|
|
6
5
|
|
|
@@ -18,13 +17,9 @@ export function computeEfficiencyModel({ app, stages, executorsAdded, runAggrega
|
|
|
18
17
|
|
|
19
18
|
|
|
20
19
|
{
|
|
21
|
-
// `app ?? {}`: computeTotalCores
|
|
22
|
-
//
|
|
23
|
-
//
|
|
24
|
-
// `executorsAdded` cast: computeTotalCores only reads `totalCores`, present on
|
|
25
|
-
// ExecutorAddedEvent (the only kind real callers pass here) but not ExecutorRemovedEvent,
|
|
26
|
-
// so the union as a whole is a structural mismatch against computeTotalCores's
|
|
27
|
-
// `{totalCores?}` shape.
|
|
20
|
+
// `app ?? {}`: computeTotalCores falls back to the executor core sum when resources is absent,
|
|
21
|
+
// and callers tolerate a null app (malformed logs); `app!` would crash on app.resources.
|
|
22
|
+
// Cast: computeTotalCores reads only totalCores, absent on ExecutorRemovedEvent, so the union mismatches.
|
|
28
23
|
const totalCores = computeTotalCores(app ?? {}, executorsAdded );
|
|
29
24
|
const appDurationMs = (app?.endTime ?? 0) - (app?.startTime ?? 0);
|
|
30
25
|
const wc = computeWallClock(app, stages );
|
|
@@ -35,9 +30,8 @@ export function computeEfficiencyModel({ app, stages, executorsAdded, runAggrega
|
|
|
35
30
|
const driverIdleMs = wc.startup + wc.gaps + wc.idle;
|
|
36
31
|
const driverWasteHours = totalCores * (driverIdleMs / 3600000);
|
|
37
32
|
|
|
38
|
-
// Executor-bound:
|
|
39
|
-
//
|
|
40
|
-
// inside stagesActive; no separate per-window sweep needed.
|
|
33
|
+
// Executor-bound: allocated capacity beyond busy cores during stagesActive. Tasks only run in
|
|
34
|
+
// active windows, so whole-run busyCoreMs already lives inside stagesActive.
|
|
41
35
|
const allocatedActiveCoreMs = totalCores * wc.stagesActive;
|
|
42
36
|
const busyCoreMs = runAggregates?.busyCoreMs ?? 0;
|
|
43
37
|
const executorWasteHours = Math.max(0, allocatedActiveCoreMs - busyCoreMs) / 3600000;
|
|
@@ -1,8 +1,6 @@
|
|
|
1
|
-
// §7 ETL-phase time attribution (Onehouse Spark Analyzer). Heuristic
|
|
2
|
-
//
|
|
3
|
-
//
|
|
4
|
-
// wall-clock. Storage-format-aware attribution (Hudi/Delta/Iceberg) is NOT
|
|
5
|
-
// portable (no table-format metadata in event logs) and is out of scope.
|
|
1
|
+
// §7 ETL-phase time attribution (Onehouse Spark Analyzer). Heuristic: classify each stage by
|
|
2
|
+
// byte-flow shape. A stage can be both Transform and Load, so buckets overlap and need not sum to
|
|
3
|
+
// wall-clock. Storage-format-aware attribution (Hudi/Delta/Iceberg) isn't portable, out of scope.
|
|
6
4
|
|
|
7
5
|
|
|
8
6
|
|