sparkforensics-cli 0.1.0 → 0.2.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +6 -0
- package/bin/sparkforensics-analyze.mjs +113 -48
- package/export-template/docs/404.html +25 -0
- package/export-template/docs/assets/app.DQTZyGL1.js +1 -0
- package/export-template/docs/assets/aqe-loop.IwQSATHw.svg +1 -0
- package/export-template/docs/assets/aqe-loop.dark.DGbaxqJE.svg +1 -0
- package/export-template/docs/assets/broadcast-vs-shuffle.Db4WY1XK.svg +1 -0
- package/export-template/docs/assets/broadcast-vs-shuffle.dark.C7Bxs0mG.svg +1 -0
- package/export-template/docs/assets/cache-lifecycle.dark.B-hS7AgU.svg +1 -0
- package/export-template/docs/assets/cache-lifecycle.rEOVYQNU.svg +1 -0
- package/export-template/docs/assets/chunks/@localSearchIndexroot.DppnXnDE.js +1 -0
- package/export-template/docs/assets/chunks/VPLocalSearchBox.BkBIPFs6.js +9 -0
- package/export-template/docs/assets/chunks/duplicate-plan-subtree.dark.Cdp70QhV.js +1 -0
- package/export-template/docs/assets/chunks/framework.DSg0KOwT.js +20 -0
- package/export-template/docs/assets/chunks/retry-escalation-ladder.dark.DHipdJgZ.js +1 -0
- package/export-template/docs/assets/chunks/theme.DP0u1AUq.js +2 -0
- package/export-template/docs/assets/cold-start-timeline.DxC_Sc7w.svg +1 -0
- package/export-template/docs/assets/cold-start-timeline.dark.CZ17YcAG.svg +1 -0
- package/export-template/docs/assets/columnar-layout.PghGeOEA.svg +1 -0
- package/export-template/docs/assets/columnar-layout.dark.BVNlz0ff.svg +1 -0
- package/export-template/docs/assets/container-memory.DIO0AnIm.svg +1 -0
- package/export-template/docs/assets/container-memory.dark.CP-5zuCl.svg +1 -0
- package/export-template/docs/assets/contributor-guide_architecture_board-widgets.md.CWpj01WU.js +1 -0
- package/export-template/docs/assets/contributor-guide_architecture_board-widgets.md.CWpj01WU.lean.js +1 -0
- package/export-template/docs/assets/contributor-guide_architecture_detector-contract.md.CgzUsQ6W.js +1 -0
- package/export-template/docs/assets/contributor-guide_architecture_detector-contract.md.CgzUsQ6W.lean.js +1 -0
- package/export-template/docs/assets/contributor-guide_architecture_drill-down.md.BtPdlM7r.js +1 -0
- package/export-template/docs/assets/contributor-guide_architecture_drill-down.md.BtPdlM7r.lean.js +1 -0
- package/export-template/docs/assets/contributor-guide_architecture_impact-estimation.md.CooslVJt.js +1 -0
- package/export-template/docs/assets/contributor-guide_architecture_impact-estimation.md.CooslVJt.lean.js +1 -0
- package/export-template/docs/assets/contributor-guide_architecture_index.md.3TO9ic6w.js +1 -0
- package/export-template/docs/assets/contributor-guide_architecture_index.md.3TO9ic6w.lean.js +1 -0
- package/export-template/docs/assets/contributor-guide_architecture_overview.md.CehiRmGn.js +1 -0
- package/export-template/docs/assets/contributor-guide_architecture_overview.md.CehiRmGn.lean.js +1 -0
- package/export-template/docs/assets/contributor-guide_architecture_state-and-history.md.C-xxn0q7.js +1 -0
- package/export-template/docs/assets/contributor-guide_architecture_state-and-history.md.C-xxn0q7.lean.js +1 -0
- package/export-template/docs/assets/contributor-guide_architecture_widget-rendering.md.R27gQrgY.js +1 -0
- package/export-template/docs/assets/contributor-guide_architecture_widget-rendering.md.R27gQrgY.lean.js +1 -0
- package/export-template/docs/assets/contributor-guide_architecture_worker-protocol.md.IbnfNrV3.js +6 -0
- package/export-template/docs/assets/contributor-guide_architecture_worker-protocol.md.IbnfNrV3.lean.js +1 -0
- package/export-template/docs/assets/contributor-guide_contributing.md.CvRsdr6J.js +1 -0
- package/export-template/docs/assets/contributor-guide_contributing.md.CvRsdr6J.lean.js +1 -0
- package/export-template/docs/assets/contributor-guide_development-setup.md.DvAN_9mK.js +12 -0
- package/export-template/docs/assets/contributor-guide_development-setup.md.DvAN_9mK.lean.js +1 -0
- package/export-template/docs/assets/contributor-guide_testing.md.6rIKqSyY.js +1 -0
- package/export-template/docs/assets/contributor-guide_testing.md.6rIKqSyY.lean.js +1 -0
- package/export-template/docs/assets/dag-stages.DSz_S937.svg +1 -0
- package/export-template/docs/assets/dag-stages.dark.F72UzxH4.svg +1 -0
- package/export-template/docs/assets/driver-executor.D5pQ7YN1.svg +1 -0
- package/export-template/docs/assets/driver-executor.dark.BmX9cPvh.svg +1 -0
- package/export-template/docs/assets/duplicate-plan-subtree.B4cvN6fj.svg +1 -0
- package/export-template/docs/assets/duplicate-plan-subtree.dark.Dw8wS0Ag.svg +1 -0
- package/export-template/docs/assets/index.md.CHJVslga.js +1 -0
- package/export-template/docs/assets/index.md.CHJVslga.lean.js +1 -0
- package/export-template/docs/assets/inter-italic-cyrillic-ext.r48I6akx.woff2 +0 -0
- package/export-template/docs/assets/inter-italic-cyrillic.By2_1cv3.woff2 +0 -0
- package/export-template/docs/assets/inter-italic-greek-ext.1u6EdAuj.woff2 +0 -0
- package/export-template/docs/assets/inter-italic-greek.DJ8dCoTZ.woff2 +0 -0
- package/export-template/docs/assets/inter-italic-latin-ext.CN1xVJS-.woff2 +0 -0
- package/export-template/docs/assets/inter-italic-latin.C2AdPX0b.woff2 +0 -0
- package/export-template/docs/assets/inter-italic-vietnamese.BSbpV94h.woff2 +0 -0
- package/export-template/docs/assets/inter-roman-cyrillic-ext.BBPuwvHQ.woff2 +0 -0
- package/export-template/docs/assets/inter-roman-cyrillic.C5lxZ8CY.woff2 +0 -0
- package/export-template/docs/assets/inter-roman-greek-ext.CqjqNYQ-.woff2 +0 -0
- package/export-template/docs/assets/inter-roman-greek.BBVDIX6e.woff2 +0 -0
- package/export-template/docs/assets/inter-roman-latin-ext.4ZJIpNVo.woff2 +0 -0
- package/export-template/docs/assets/inter-roman-latin.Di8DUHzh.woff2 +0 -0
- package/export-template/docs/assets/inter-roman-vietnamese.BjW4sHH5.woff2 +0 -0
- package/export-template/docs/assets/join-strategy.C_FvrCEo.svg +1 -0
- package/export-template/docs/assets/join-strategy.dark.ChMLnNII.svg +1 -0
- package/export-template/docs/assets/memory-borrowing.BqQRJg0u.svg +1 -0
- package/export-template/docs/assets/memory-borrowing.dark.Yhh20O9C.svg +1 -0
- package/export-template/docs/assets/memory-regions.XHvO7jHG.svg +1 -0
- package/export-template/docs/assets/memory-regions.dark.D4TP9_08.svg +1 -0
- package/export-template/docs/assets/repartition-vs-coalesce.BovLRrpj.svg +1 -0
- package/export-template/docs/assets/repartition-vs-coalesce.dark.BhAczKZQ.svg +1 -0
- package/export-template/docs/assets/retry-escalation-ladder.DyTKJJmZ.svg +1 -0
- package/export-template/docs/assets/retry-escalation-ladder.dark.BdsabtU3.svg +1 -0
- package/export-template/docs/assets/shuffle-map-reduce.KuOEZVmg.svg +1 -0
- package/export-template/docs/assets/shuffle-map-reduce.dark.BgQZnFSb.svg +1 -0
- package/export-template/docs/assets/spill-classification.BU2euYDO.svg +1 -0
- package/export-template/docs/assets/spill-classification.dark.D7i1M40d.svg +1 -0
- package/export-template/docs/assets/style.DSixAiZE.css +1 -0
- package/export-template/docs/assets/tuning-reference_anti-patterns.md.Df1YMIHu.js +1 -0
- package/export-template/docs/assets/tuning-reference_anti-patterns.md.Df1YMIHu.lean.js +1 -0
- package/export-template/docs/assets/tuning-reference_aqe.md.BIsCtLzm.js +1 -0
- package/export-template/docs/assets/tuning-reference_aqe.md.BIsCtLzm.lean.js +1 -0
- package/export-template/docs/assets/tuning-reference_bottleneck-broadcast-sizing.md.CEstB3Ia.js +1 -0
- package/export-template/docs/assets/tuning-reference_bottleneck-broadcast-sizing.md.CEstB3Ia.lean.js +1 -0
- package/export-template/docs/assets/tuning-reference_bottleneck-cold-start.md.CEuy-72y.js +7 -0
- package/export-template/docs/assets/tuning-reference_bottleneck-cold-start.md.CEuy-72y.lean.js +1 -0
- package/export-template/docs/assets/tuning-reference_bottleneck-duplicate-plan-subtree.md.CIohQDfn.js +1 -0
- package/export-template/docs/assets/tuning-reference_bottleneck-duplicate-plan-subtree.md.CIohQDfn.lean.js +1 -0
- package/export-template/docs/assets/tuning-reference_bottleneck-failures.md.4z5BXGJ2.js +6 -0
- package/export-template/docs/assets/tuning-reference_bottleneck-failures.md.4z5BXGJ2.lean.js +1 -0
- package/export-template/docs/assets/tuning-reference_bottleneck-gc.md.DSxzZRK7.js +6 -0
- package/export-template/docs/assets/tuning-reference_bottleneck-gc.md.DSxzZRK7.lean.js +1 -0
- package/export-template/docs/assets/tuning-reference_bottleneck-job-failure-rate.md.BaJl__1W.js +8 -0
- package/export-template/docs/assets/tuning-reference_bottleneck-job-failure-rate.md.BaJl__1W.lean.js +1 -0
- package/export-template/docs/assets/tuning-reference_bottleneck-memory-utilization.md.DbP-SJZc.js +7 -0
- package/export-template/docs/assets/tuning-reference_bottleneck-memory-utilization.md.DbP-SJZc.lean.js +1 -0
- package/export-template/docs/assets/tuning-reference_bottleneck-retry-waste.md.D5JMjVOt.js +1 -0
- package/export-template/docs/assets/tuning-reference_bottleneck-retry-waste.md.D5JMjVOt.lean.js +1 -0
- package/export-template/docs/assets/tuning-reference_bottleneck-shuffle.md.CM-nTmIH.js +12 -0
- package/export-template/docs/assets/tuning-reference_bottleneck-shuffle.md.CM-nTmIH.lean.js +1 -0
- package/export-template/docs/assets/tuning-reference_bottleneck-skew.md.BdUwiDhn.js +14 -0
- package/export-template/docs/assets/tuning-reference_bottleneck-skew.md.BdUwiDhn.lean.js +1 -0
- package/export-template/docs/assets/tuning-reference_bottleneck-slow-host.md.BlIo6UDW.js +7 -0
- package/export-template/docs/assets/tuning-reference_bottleneck-slow-host.md.BlIo6UDW.lean.js +1 -0
- package/export-template/docs/assets/tuning-reference_bottleneck-small-files.md.B8kloyx8.js +5 -0
- package/export-template/docs/assets/tuning-reference_bottleneck-small-files.md.B8kloyx8.lean.js +1 -0
- package/export-template/docs/assets/tuning-reference_bottleneck-spill.md.PNH7mITt.js +6 -0
- package/export-template/docs/assets/tuning-reference_bottleneck-spill.md.PNH7mITt.lean.js +1 -0
- package/export-template/docs/assets/tuning-reference_bottleneck-straggler.md.DY36fHN5.js +7 -0
- package/export-template/docs/assets/tuning-reference_bottleneck-straggler.md.DY36fHN5.lean.js +1 -0
- package/export-template/docs/assets/tuning-reference_bottleneck-tiny-tasks.md.QTV7O8kU.js +8 -0
- package/export-template/docs/assets/tuning-reference_bottleneck-tiny-tasks.md.QTV7O8kU.lean.js +1 -0
- package/export-template/docs/assets/tuning-reference_bottleneck-utilization.md.DTiueZC3.js +5 -0
- package/export-template/docs/assets/tuning-reference_bottleneck-utilization.md.DTiueZC3.lean.js +1 -0
- package/export-template/docs/assets/tuning-reference_caching.md.B7aQ8asB.js +1 -0
- package/export-template/docs/assets/tuning-reference_caching.md.B7aQ8asB.lean.js +1 -0
- package/export-template/docs/assets/tuning-reference_cluster-config.md.ZVmDGsQ3.js +1 -0
- package/export-template/docs/assets/tuning-reference_cluster-config.md.ZVmDGsQ3.lean.js +1 -0
- package/export-template/docs/assets/tuning-reference_config.md.UvveiWG3.js +1 -0
- package/export-template/docs/assets/tuning-reference_config.md.UvveiWG3.lean.js +1 -0
- package/export-template/docs/assets/tuning-reference_data-formats.md.bjCAWH3N.js +1 -0
- package/export-template/docs/assets/tuning-reference_data-formats.md.bjCAWH3N.lean.js +1 -0
- package/export-template/docs/assets/tuning-reference_index.md.BQ_NooMV.js +1 -0
- package/export-template/docs/assets/tuning-reference_index.md.BQ_NooMV.lean.js +1 -0
- package/export-template/docs/assets/tuning-reference_intro.md.CobD-lGB.js +1 -0
- package/export-template/docs/assets/tuning-reference_intro.md.CobD-lGB.lean.js +1 -0
- package/export-template/docs/assets/tuning-reference_joins.md.BtKs_CuW.js +1 -0
- package/export-template/docs/assets/tuning-reference_joins.md.BtKs_CuW.lean.js +1 -0
- package/export-template/docs/assets/tuning-reference_memory-model.md.DhT-n4y3.js +1 -0
- package/export-template/docs/assets/tuning-reference_memory-model.md.DhT-n4y3.lean.js +1 -0
- package/export-template/docs/assets/tuning-reference_metrics.md.mLOh7Apj.js +1 -0
- package/export-template/docs/assets/tuning-reference_metrics.md.mLOh7Apj.lean.js +1 -0
- package/export-template/docs/assets/tuning-reference_partitioning.md.q0zKF_8X.js +1 -0
- package/export-template/docs/assets/tuning-reference_partitioning.md.q0zKF_8X.lean.js +1 -0
- package/export-template/docs/assets/tuning-reference_pyspark.md.DDCfvN9t.js +6 -0
- package/export-template/docs/assets/tuning-reference_pyspark.md.DDCfvN9t.lean.js +1 -0
- package/export-template/docs/assets/tuning-reference_shuffle.md.BZZ7R4Ix.js +1 -0
- package/export-template/docs/assets/tuning-reference_shuffle.md.BZZ7R4Ix.lean.js +1 -0
- package/export-template/docs/assets/tuning-reference_spark-architecture.md.Dwzm5avO.js +1 -0
- package/export-template/docs/assets/tuning-reference_spark-architecture.md.Dwzm5avO.lean.js +1 -0
- package/export-template/docs/assets/tuning-reference_table-formats.md.D6wj-2dX.js +1 -0
- package/export-template/docs/assets/tuning-reference_table-formats.md.D6wj-2dX.lean.js +1 -0
- package/export-template/docs/assets/udf-execution-models.BUFDICuG.svg +1 -0
- package/export-template/docs/assets/udf-execution-models.dark.YTNS6GDq.svg +1 -0
- package/export-template/docs/assets/user-guide_alternative-log-retrieval.md.B4tPGIal.js +1 -0
- package/export-template/docs/assets/user-guide_alternative-log-retrieval.md.B4tPGIal.lean.js +1 -0
- package/export-template/docs/assets/user-guide_getting-started.md.BJvwLEIM.js +3 -0
- package/export-template/docs/assets/user-guide_getting-started.md.BJvwLEIM.lean.js +1 -0
- package/export-template/docs/assets/user-guide_mcp-tools.md.Vi3RoflJ.js +125 -0
- package/export-template/docs/assets/user-guide_mcp-tools.md.Vi3RoflJ.lean.js +1 -0
- package/export-template/docs/assets/user-guide_run-comparison.md.CQc1aoU8.js +1 -0
- package/export-template/docs/assets/user-guide_run-comparison.md.CQc1aoU8.lean.js +1 -0
- package/export-template/docs/assets/user-guide_understanding-findings.md.DL1UDhvR.js +1 -0
- package/export-template/docs/assets/user-guide_understanding-findings.md.DL1UDhvR.lean.js +1 -0
- package/export-template/docs/contributor-guide/architecture/board-widgets.html +25 -0
- package/export-template/docs/contributor-guide/architecture/detector-contract.html +25 -0
- package/export-template/docs/contributor-guide/architecture/drill-down.html +25 -0
- package/export-template/docs/contributor-guide/architecture/impact-estimation.html +25 -0
- package/export-template/docs/contributor-guide/architecture/index.html +25 -0
- package/export-template/docs/contributor-guide/architecture/overview.html +25 -0
- package/export-template/docs/contributor-guide/architecture/state-and-history.html +25 -0
- package/export-template/docs/contributor-guide/architecture/widget-rendering.html +25 -0
- package/export-template/docs/contributor-guide/architecture/worker-protocol.html +30 -0
- package/export-template/docs/contributor-guide/contributing.html +25 -0
- package/export-template/docs/contributor-guide/development-setup.html +36 -0
- package/export-template/docs/contributor-guide/testing.html +25 -0
- package/export-template/docs/favicon.svg +4 -0
- package/export-template/docs/hashmap.json +1 -0
- package/export-template/docs/index.html +25 -0
- package/export-template/docs/package.json +1 -0
- package/export-template/docs/tuning-reference/anti-patterns.html +25 -0
- package/export-template/docs/tuning-reference/aqe.html +25 -0
- package/export-template/docs/tuning-reference/bottleneck-broadcast-sizing.html +25 -0
- package/export-template/docs/tuning-reference/bottleneck-cold-start.html +31 -0
- package/export-template/docs/tuning-reference/bottleneck-duplicate-plan-subtree.html +25 -0
- package/export-template/docs/tuning-reference/bottleneck-failures.html +30 -0
- package/export-template/docs/tuning-reference/bottleneck-gc.html +30 -0
- package/export-template/docs/tuning-reference/bottleneck-job-failure-rate.html +32 -0
- package/export-template/docs/tuning-reference/bottleneck-memory-utilization.html +31 -0
- package/export-template/docs/tuning-reference/bottleneck-retry-waste.html +25 -0
- package/export-template/docs/tuning-reference/bottleneck-shuffle.html +36 -0
- package/export-template/docs/tuning-reference/bottleneck-skew.html +38 -0
- package/export-template/docs/tuning-reference/bottleneck-slow-host.html +31 -0
- package/export-template/docs/tuning-reference/bottleneck-small-files.html +29 -0
- package/export-template/docs/tuning-reference/bottleneck-spill.html +30 -0
- package/export-template/docs/tuning-reference/bottleneck-straggler.html +31 -0
- package/export-template/docs/tuning-reference/bottleneck-tiny-tasks.html +32 -0
- package/export-template/docs/tuning-reference/bottleneck-utilization.html +29 -0
- package/export-template/docs/tuning-reference/caching.html +25 -0
- package/export-template/docs/tuning-reference/cluster-config.html +25 -0
- package/export-template/docs/tuning-reference/config.html +25 -0
- package/export-template/docs/tuning-reference/data-formats.html +25 -0
- package/export-template/docs/tuning-reference/index.html +25 -0
- package/export-template/docs/tuning-reference/intro.html +25 -0
- package/export-template/docs/tuning-reference/joins.html +25 -0
- package/export-template/docs/tuning-reference/memory-model.html +25 -0
- package/export-template/docs/tuning-reference/metrics.html +25 -0
- package/export-template/docs/tuning-reference/partitioning.html +25 -0
- package/export-template/docs/tuning-reference/pyspark.html +30 -0
- package/export-template/docs/tuning-reference/shuffle.html +25 -0
- package/export-template/docs/tuning-reference/spark-architecture.html +25 -0
- package/export-template/docs/tuning-reference/table-formats.html +25 -0
- package/export-template/docs/user-guide/alternative-log-retrieval.html +25 -0
- package/export-template/docs/user-guide/getting-started.html +27 -0
- package/export-template/docs/user-guide/mcp-tools.html +149 -0
- package/export-template/docs/user-guide/run-comparison.html +25 -0
- package/export-template/docs/user-guide/understanding-findings.html +25 -0
- package/export-template/docs/vp-icons.css +0 -0
- package/export-template/favicon.svg +4 -0
- package/export-template/index.html +111 -0
- package/export-template/parser-worker-DyjiQvfP.js +112 -0
- package/export-template/sample-runs/sample-run.ndjson.gz +0 -0
- package/package.json +20 -6
- package/vendor-core/analyzer.js +74 -74
- package/vendor-core/cli/budgets.js +13 -27
- package/vendor-core/cli/collect-run.js +43 -19
- package/vendor-core/core-count.js +25 -27
- package/vendor-core/core-locality-ratio.js +4 -11
- package/vendor-core/core-time-series.js +6 -12
- package/vendor-core/core-usage-locality.js +3 -4
- package/vendor-core/detectors.js +395 -389
- package/vendor-core/docs-config.js +69 -21
- package/vendor-core/docs-content/chapters/01-intro.md +32 -0
- package/vendor-core/docs-content/chapters/02-spark-architecture.md +76 -0
- package/vendor-core/docs-content/chapters/03-memory-model.md +73 -0
- package/vendor-core/docs-content/chapters/04-partitioning.md +65 -0
- package/vendor-core/docs-content/chapters/05-joins.md +62 -0
- package/vendor-core/docs-content/chapters/06-shuffle.md +59 -0
- package/vendor-core/docs-content/chapters/07-data-formats.md +81 -0
- package/vendor-core/docs-content/chapters/07b-table-formats.md +56 -0
- package/vendor-core/docs-content/chapters/08-caching.md +58 -0
- package/vendor-core/docs-content/chapters/09-pyspark.md +78 -0
- package/vendor-core/docs-content/chapters/10-aqe.md +167 -0
- package/vendor-core/docs-content/chapters/11-cluster-config.md +170 -0
- package/vendor-core/docs-content/chapters/12-anti-patterns.md +171 -0
- package/vendor-core/docs-content/chapters/14-metrics.md +87 -0
- package/vendor-core/docs-content/chapters/15-config.md +93 -0
- package/vendor-core/docs-content/chapters/nav-index.json +370 -0
- package/vendor-core/docs-content/detection/cache.md +7 -0
- package/vendor-core/docs-content/detection/cfg.md +15 -0
- package/vendor-core/docs-content/detection/chrn.md +9 -0
- package/vendor-core/docs-content/detection/cold.md +4 -0
- package/vendor-core/docs-content/detection/cstor.md +4 -0
- package/vendor-core/docs-content/detection/fail.md +5 -0
- package/vendor-core/docs-content/detection/gc.md +4 -0
- package/vendor-core/docs-content/detection/host.md +5 -0
- package/vendor-core/docs-content/detection/incmp.md +6 -0
- package/vendor-core/docs-content/detection/jobs.md +4 -0
- package/vendor-core/docs-content/detection/local.md +6 -0
- package/vendor-core/docs-content/detection/mem.md +10 -0
- package/vendor-core/docs-content/detection/part.md +5 -0
- package/vendor-core/docs-content/detection/plan.md +14 -0
- package/vendor-core/docs-content/detection/retry.md +4 -0
- package/vendor-core/docs-content/detection/sfail.md +5 -0
- package/vendor-core/docs-content/detection/shape.md +5 -0
- package/vendor-core/docs-content/detection/shfl.md +4 -0
- package/vendor-core/docs-content/detection/skew.md +6 -0
- package/vendor-core/docs-content/detection/slow.md +6 -0
- package/vendor-core/docs-content/detection/spec.md +8 -0
- package/vendor-core/docs-content/detection/spill.md +7 -0
- package/vendor-core/docs-content/detection/strag.md +5 -0
- package/vendor-core/docs-content/detection/tiny.md +4 -0
- package/vendor-core/docs-content/detection/util.md +4 -0
- package/vendor-core/docs-content/diagrams/aqe-loop.dark.svg +1 -0
- package/vendor-core/docs-content/diagrams/aqe-loop.svg +1 -0
- package/vendor-core/docs-content/diagrams/broadcast-vs-shuffle.dark.svg +1 -0
- package/vendor-core/docs-content/diagrams/broadcast-vs-shuffle.svg +1 -0
- package/vendor-core/docs-content/diagrams/cache-lifecycle.dark.svg +1 -0
- package/vendor-core/docs-content/diagrams/cache-lifecycle.svg +1 -0
- package/vendor-core/docs-content/diagrams/cold-start-timeline.dark.svg +1 -0
- package/vendor-core/docs-content/diagrams/cold-start-timeline.svg +1 -0
- package/vendor-core/docs-content/diagrams/columnar-layout.dark.svg +1 -0
- package/vendor-core/docs-content/diagrams/columnar-layout.svg +1 -0
- package/vendor-core/docs-content/diagrams/container-memory.dark.svg +1 -0
- package/vendor-core/docs-content/diagrams/container-memory.svg +1 -0
- package/vendor-core/docs-content/diagrams/dag-stages.dark.svg +1 -0
- package/vendor-core/docs-content/diagrams/dag-stages.svg +1 -0
- package/vendor-core/docs-content/diagrams/driver-executor.dark.svg +1 -0
- package/vendor-core/docs-content/diagrams/driver-executor.svg +1 -0
- package/vendor-core/docs-content/diagrams/duplicate-plan-subtree.dark.svg +1 -0
- package/vendor-core/docs-content/diagrams/duplicate-plan-subtree.svg +1 -0
- package/vendor-core/docs-content/diagrams/join-strategy.dark.svg +1 -0
- package/vendor-core/docs-content/diagrams/join-strategy.svg +1 -0
- package/vendor-core/docs-content/diagrams/memory-borrowing.dark.svg +1 -0
- package/vendor-core/docs-content/diagrams/memory-borrowing.svg +1 -0
- package/vendor-core/docs-content/diagrams/memory-regions.dark.svg +1 -0
- package/vendor-core/docs-content/diagrams/memory-regions.svg +1 -0
- package/vendor-core/docs-content/diagrams/repartition-vs-coalesce.dark.svg +1 -0
- package/vendor-core/docs-content/diagrams/repartition-vs-coalesce.svg +1 -0
- package/vendor-core/docs-content/diagrams/retry-escalation-ladder.dark.svg +1 -0
- package/vendor-core/docs-content/diagrams/retry-escalation-ladder.svg +1 -0
- package/vendor-core/docs-content/diagrams/shuffle-map-reduce.dark.svg +1 -0
- package/vendor-core/docs-content/diagrams/shuffle-map-reduce.svg +1 -0
- package/vendor-core/docs-content/diagrams/spill-classification.dark.svg +1 -0
- package/vendor-core/docs-content/diagrams/spill-classification.svg +1 -0
- package/vendor-core/docs-content/diagrams/udf-execution-models.dark.svg +1 -0
- package/vendor-core/docs-content/diagrams/udf-execution-models.svg +1 -0
- package/vendor-core/docs-content/tuning/broadcast-sizing.md +78 -0
- package/vendor-core/docs-content/tuning/cold-start.md +81 -0
- package/vendor-core/docs-content/tuning/duplicate-plan-subtree.md +45 -0
- package/vendor-core/docs-content/tuning/failures.md +124 -0
- package/vendor-core/docs-content/tuning/gc.md +110 -0
- package/vendor-core/docs-content/tuning/job-failure-rate.md +101 -0
- package/vendor-core/docs-content/tuning/memory-utilization.md +58 -0
- package/vendor-core/docs-content/tuning/retry-waste.md +90 -0
- package/vendor-core/docs-content/tuning/shuffle.md +154 -0
- package/vendor-core/docs-content/tuning/skew.md +123 -0
- package/vendor-core/docs-content/tuning/slow-host.md +117 -0
- package/vendor-core/docs-content/tuning/small-files.md +99 -0
- package/vendor-core/docs-content/tuning/spill.md +114 -0
- package/vendor-core/docs-content/tuning/straggler.md +103 -0
- package/vendor-core/docs-content/tuning/tiny-tasks.md +94 -0
- package/vendor-core/docs-content/tuning/utilization.md +90 -0
- package/vendor-core/docs-site-config.js +10 -17
- package/vendor-core/efficiency-model.js +7 -13
- package/vendor-core/etl-phases.js +3 -5
- package/vendor-core/event-handlers.js +232 -134
- package/vendor-core/event-schemas.js +48 -114
- package/vendor-core/evidence-availability.js +5 -10
- package/vendor-core/evidence-report.js +73 -123
- package/vendor-core/export-data.js +48 -0
- package/vendor-core/finding-action-label.js +4 -10
- package/vendor-core/finding-filter-predicate.js +3 -7
- package/vendor-core/finding-generic-recommendation.js +112 -0
- package/vendor-core/finding-names.js +51 -0
- package/vendor-core/format-utils.js +112 -38
- package/vendor-core/impact-band.js +18 -24
- package/vendor-core/impact-estimator.js +38 -74
- package/vendor-core/ingest.js +7 -13
- package/vendor-core/job-groups.js +3 -6
- package/vendor-core/list-runs.js +278 -0
- package/vendor-core/load-vendored.js +6 -12
- package/vendor-core/log-header-peek.js +81 -0
- package/vendor-core/lz4-block.js +4 -6
- package/vendor-core/mcp-server-factory.js +38 -8
- package/vendor-core/mcp-tools.js +105 -76
- package/vendor-core/model-assembler.js +8 -16
- package/vendor-core/occupancy.js +5 -9
- package/vendor-core/parser-worker.js +20 -29
- package/vendor-core/plan-dot.js +2 -5
- package/vendor-core/plan-duration-attribution.js +78 -29
- package/vendor-core/plan-graph-model.js +126 -69
- package/vendor-core/plan-node-detail.js +31 -17
- package/vendor-core/plan-summary.js +19 -8
- package/vendor-core/recommendation-rollup.js +35 -39
- package/vendor-core/redact.js +72 -16
- package/vendor-core/rolling-log-reassembly.js +4 -6
- package/vendor-core/run-comparison.js +86 -72
- package/vendor-core/scaling-sim.js +5 -7
- package/vendor-core/session-snapshot.js +1 -1
- package/vendor-core/shs-fetch.js +4 -6
- package/vendor-core/shs-load.js +9 -13
- package/vendor-core/shs-request.js +1 -1
- package/vendor-core/stage-quantiles.js +14 -0
- package/vendor-core/types.js +78 -18
- package/vendor-core/wasted-core-hours.js +7 -12
package/vendor-core/detectors.js
CHANGED
|
@@ -1,33 +1,23 @@
|
|
|
1
1
|
import { pathBasename, formatBytes, nsToMs, IMPACT_BAND_ORDER } from './format-utils.js';
|
|
2
2
|
import { scanRelationId } from './plan-summary.js';
|
|
3
|
-
import {
|
|
3
|
+
import { computePeakConcurrentCores, computePeakConcurrentExecutorCount } from './core-count.js';
|
|
4
4
|
import { walkPlanTree } from './plan-tree-walk.js';
|
|
5
5
|
import { computeCoreLocalityRatio } from './core-locality-ratio.js';
|
|
6
6
|
import { estimateSingleStage, } from './occupancy.js';
|
|
7
|
+
import { isExchangeNode, isBroadcastExchangeNode } from './plan-node-detail.js';
|
|
7
8
|
|
|
8
9
|
|
|
9
10
|
const MB = 1024 * 1024;
|
|
10
11
|
const GB = 1024 * MB;
|
|
11
12
|
const TB = 1024 * GB;
|
|
12
13
|
|
|
13
|
-
// ---------------------------------------------------------------------------
|
|
14
14
|
// Local runtime shapes.
|
|
15
15
|
//
|
|
16
|
-
//
|
|
17
|
-
//
|
|
18
|
-
//
|
|
19
|
-
//
|
|
20
|
-
//
|
|
21
|
-
// `event-handlers.ts`'s stage/sql/app records actually carry, accessed
|
|
22
|
-
// directly (arithmetic, comparisons) rather than read-and-display, so an
|
|
23
|
-
// index-signature-shaped type would force an `unknown` cast at nearly every
|
|
24
|
-
// field access. Every field below is verified against a real access in this
|
|
25
|
-
// file (or a helper it calls); none are speculative.
|
|
26
|
-
//
|
|
27
|
-
// `analyzer.js` (the only real caller of `detect()`, still plain JS) passes
|
|
28
|
-
// whatever the parser actually produced, so these types describe reality,
|
|
29
|
-
// not a narrowing of some existing stricter type; there is nothing unsound
|
|
30
|
-
// about them being independent of `types.ts`'s `Stage`/`SqlExecution`.
|
|
16
|
+
// types.ts's Stage/SqlExecution/SparkAppInfo describe the posted AppModel surface for the view
|
|
17
|
+
// layer (with a catch-all index signature). This file needs the FULL set of fields finalizeStage
|
|
18
|
+
// computes and event-handlers.ts records carry, accessed directly (arithmetic, comparisons), so
|
|
19
|
+
// an index-signature type would force an `unknown` cast at nearly every access. Every field below
|
|
20
|
+
// is verified against a real access here.
|
|
31
21
|
|
|
32
22
|
|
|
33
23
|
|
|
@@ -35,9 +25,18 @@ const TB = 1024 * GB;
|
|
|
35
25
|
|
|
36
26
|
|
|
37
27
|
|
|
38
|
-
|
|
39
|
-
|
|
40
|
-
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
// Per-executor snapshot from a StageExecutorMetrics event: a loose bag of Spark's
|
|
39
|
+
// ExecutorMetrics field names, only a few of which any detector reads.
|
|
41
40
|
|
|
42
41
|
|
|
43
42
|
|
|
@@ -81,6 +80,8 @@ const TB = 1024 * GB;
|
|
|
81
80
|
|
|
82
81
|
|
|
83
82
|
|
|
83
|
+
|
|
84
|
+
|
|
84
85
|
|
|
85
86
|
|
|
86
87
|
|
|
@@ -116,27 +117,19 @@ const TB = 1024 * GB;
|
|
|
116
117
|
|
|
117
118
|
|
|
118
119
|
|
|
119
|
-
|
|
120
|
-
|
|
120
|
+
|
|
121
121
|
|
|
122
122
|
|
|
123
123
|
|
|
124
124
|
|
|
125
|
-
// The full context
|
|
126
|
-
//
|
|
127
|
-
// 'app' scope (`d.detect(ctx)`). 'config' scope gets a narrower `{ app }`
|
|
128
|
-
// (see auditConfig in analyzer.js), typed per-entry below instead of here.
|
|
125
|
+
// The full context analyze() passes as every stage/sql detect()'s second arg, and as the sole
|
|
126
|
+
// arg for 'app' scope. 'config' scope gets a narrower `{ app }` (auditConfig), typed per-entry.
|
|
129
127
|
//
|
|
130
|
-
// `app` is non-nullable here (unlike
|
|
131
|
-
//
|
|
132
|
-
//
|
|
133
|
-
//
|
|
134
|
-
//
|
|
135
|
-
// (harmless on a non-nullable value), and `coldStart` reads `app.startTime`
|
|
136
|
-
// with no guard at all, which only type-checks if `app` is non-nullable.
|
|
137
|
-
// `auditConfig`'s separate config-scope target type keeps `app` nullable
|
|
138
|
-
// instead, since `auditConfig(appModel.app)` (src/analyzer.js) really can be
|
|
139
|
-
// called with a `null` app and every config-scope entry optional-chains it.
|
|
128
|
+
// `app` is typed as non-nullable here (unlike AppModel.app), but incompleteRun, coldStart,
|
|
129
|
+
// utilization, and autoscalingChurn defensively guard against null at runtime to tolerate
|
|
130
|
+
// malformed/incomplete logs. Despite the type annotation, app may be null in edge cases, and
|
|
131
|
+
// these detectors handle it gracefully. auditConfig's config-scope target keeps `app` nullable
|
|
132
|
+
// instead.
|
|
140
133
|
|
|
141
134
|
|
|
142
135
|
|
|
@@ -145,17 +138,13 @@ const TB = 1024 * GB;
|
|
|
145
138
|
|
|
146
139
|
|
|
147
140
|
|
|
148
|
-
|
|
149
|
-
|
|
150
|
-
|
|
151
|
-
|
|
152
|
-
|
|
141
|
+
|
|
142
|
+
|
|
153
143
|
|
|
154
144
|
|
|
155
145
|
|
|
156
|
-
//
|
|
157
|
-
//
|
|
158
|
-
// (`SparkAppInfo | null`), independent of the full `DetectorCtx` above.
|
|
146
|
+
// auditConfig(app) calls every config-scope detect({ app }) with whatever appModel.app is
|
|
147
|
+
// (SparkAppInfo | null), independent of DetectorCtx.
|
|
159
148
|
|
|
160
149
|
|
|
161
150
|
|
|
@@ -167,8 +156,7 @@ const TB = 1024 * GB;
|
|
|
167
156
|
|
|
168
157
|
|
|
169
158
|
|
|
170
|
-
//
|
|
171
|
-
// Returns { magnitude } | null. Magnitude ∈ 'severe'|'high'|'medium'.
|
|
159
|
+
// SpillPressureDetector (5a) + SpillSkewDetector (5b).
|
|
172
160
|
function computeSpillMagnitude(
|
|
173
161
|
stage ,
|
|
174
162
|
t ,
|
|
@@ -195,15 +183,9 @@ function pickDominantReason(reasons )
|
|
|
195
183
|
return [...reasons].sort((a, b) => b.count - a.count)[0].reason;
|
|
196
184
|
}
|
|
197
185
|
|
|
198
|
-
// Shared by every scope:'sql' detector
|
|
199
|
-
//
|
|
200
|
-
//
|
|
201
|
-
// stage's own (correctly populated) `sqlExecutionId`. Parameter type is
|
|
202
|
-
// intentionally the minimal shape needed (not `DetectorStage`): callers
|
|
203
|
-
// outside this file (Topbar.tsx, plan-node-detail.ts, plan-graph-model.ts)
|
|
204
|
-
// pass `appModel.stages`, typed `Map<StageId, Stage>` per types.ts, which
|
|
205
|
-
// carries `id`/`sqlExecutionId` but not this file's fuller `DetectorStage`
|
|
206
|
-
// shape.
|
|
186
|
+
// Shared by every scope:'sql' detector. sql.get(id).stageIds is always empty (parser-worker
|
|
187
|
+
// never populates it), so stage linkage is derived from each stage's own sqlExecutionId.
|
|
188
|
+
// Minimal param shape (not DetectorStage): external callers pass Map<StageId, Stage>.
|
|
207
189
|
export function stageIdsForSqlExec(
|
|
208
190
|
executionId ,
|
|
209
191
|
stages ,
|
|
@@ -213,38 +195,23 @@ export function stageIdsForSqlExec(
|
|
|
213
195
|
return out;
|
|
214
196
|
}
|
|
215
197
|
|
|
216
|
-
// Shared by the three Plan Advisor detectors
|
|
217
|
-
//
|
|
218
|
-
//
|
|
219
|
-
// coverage. A finding never partially blends a narrowed set with the
|
|
220
|
-
// execution-wide one (see docs/architecture.md's plan-node-to-stage mapping
|
|
221
|
-
// section). `fallback` is a param, not computed here, so callers can compute
|
|
222
|
-
// `stageIdsForSqlExec` once per `detect()` call and reuse it across every
|
|
223
|
-
// finding in that call instead of re-walking `stages` per finding.
|
|
198
|
+
// Shared by the three Plan Advisor detectors: union the given nodes' own stageIds, or fall back
|
|
199
|
+
// to the whole execution's stage set when none have coverage (never a partial blend). `fallback`
|
|
200
|
+
// is a param so callers compute stageIdsForSqlExec once per detect() and reuse it per finding.
|
|
224
201
|
export function unionStageIds(nodes , fallback ) {
|
|
225
202
|
const union = new Set ();
|
|
226
203
|
for (const node of nodes) for (const sid of node.stageIds ?? []) union.add(sid);
|
|
227
204
|
return union.size > 0 ? [...union].sort((a, b) => a - b) : fallback;
|
|
228
205
|
}
|
|
229
206
|
|
|
230
|
-
// Bottom-up shape computation for duplicate-subtree detection
|
|
231
|
-
//
|
|
232
|
-
//
|
|
233
|
-
//
|
|
234
|
-
// encodes operator name + sorted metric *names* (never values, per spec) +
|
|
235
|
-
// children fingerprints in original order, so two subtrees with the same shape
|
|
236
|
-
// but different row counts/literals still collide, which is the point (we
|
|
237
|
-
// have no expr/plan/codegen IDs to strip in the first place, since `metrics`
|
|
238
|
-
// never carried them).
|
|
207
|
+
// Bottom-up shape computation for duplicate-subtree detection and the
|
|
208
|
+
// cachingOpportunity composite detector. `size` is the subtree node count; the default
|
|
209
|
+
// fingerprint encodes operator name + sorted metric NAMES (never values, per spec) + child
|
|
210
|
+
// fingerprints, so two subtrees with the same shape but different values still collide.
|
|
239
211
|
//
|
|
240
|
-
//
|
|
241
|
-
//
|
|
242
|
-
//
|
|
243
|
-
// This is what lets a caller ask "does this specific node's own detail +
|
|
244
|
-
// structural shape match another node's", without descendant filter/scan
|
|
245
|
-
// detail (literals, paths) ever entering the comparison. The existing
|
|
246
|
-
// `duplicatePlanSubtree` call site passes no options: identical behavior,
|
|
247
|
-
// zero regression to its existing tests.
|
|
212
|
+
// opts.includeDetail (default false) folds root's own detail (via opts.normalizeDetail) into
|
|
213
|
+
// ONLY root's fingerprint, never a child's: lets a caller match "this node's own detail + shape"
|
|
214
|
+
// without descendant scan detail entering the comparison.
|
|
248
215
|
|
|
249
216
|
|
|
250
217
|
export function computePlanShapes(
|
|
@@ -256,9 +223,22 @@ export function computePlanShapes(
|
|
|
256
223
|
const allNodes = [];
|
|
257
224
|
function visit(node , isRoot ) {
|
|
258
225
|
allNodes.push(node);
|
|
259
|
-
|
|
226
|
+
// A read half wrapping a write half (the Exchange split from
|
|
227
|
+
// resolvePlanTree, see event-handlers.ts) is one logical Spark operator
|
|
228
|
+
// for subtree-shape purposes. Without this, every real Exchange in a
|
|
229
|
+
// matched subtree would count twice, inflating duplicatePlanSubtree's
|
|
230
|
+
// reported subtreeSize and shifting its groupIndex-derived findingIds
|
|
231
|
+
// for an otherwise-unchanged plan. Skip straight through the write
|
|
232
|
+
// wrapper: size comes from its real children, and metricNames from its
|
|
233
|
+
// real metrics (the read half's own metrics are always empty), so
|
|
234
|
+
// fingerprint distinctiveness between different real Exchanges is
|
|
235
|
+
// preserved too.
|
|
236
|
+
const writeHalf = node.exchangeRole === 'read' ? node.children[0] : null;
|
|
237
|
+
const realChildren = writeHalf ? writeHalf.children : (node.children ?? []);
|
|
238
|
+
const realMetrics = writeHalf ? writeHalf.metrics : node.metrics;
|
|
239
|
+
const childShapes = realChildren.map((c) => visit(c, false));
|
|
260
240
|
const size = 1 + childShapes.reduce((sum, c) => sum + c.size, 0);
|
|
261
|
-
const metricNames = (
|
|
241
|
+
const metricNames = (realMetrics ?? []).map((m) => m.name).sort().join(',');
|
|
262
242
|
const childFingerprints = childShapes.map((c) => c.fingerprint).join(',');
|
|
263
243
|
const fingerprint = isRoot && includeDetail
|
|
264
244
|
? `${node.name}[${metricNames}]<${normalize(node.detail ?? '')}>{${childFingerprints}}`
|
|
@@ -271,16 +251,10 @@ export function computePlanShapes(
|
|
|
271
251
|
return { shapeOf, allNodes };
|
|
272
252
|
}
|
|
273
253
|
|
|
274
|
-
// Normalizes an anchor join/union node's
|
|
275
|
-
//
|
|
276
|
-
// (
|
|
277
|
-
//
|
|
278
|
-
// expr ids, plan/codegen-stage ids, and AQE's runtime BuildLeft/BuildRight
|
|
279
|
-
// broadcast-side choice (which can flip between executions of the logically
|
|
280
|
-
// identical join based on runtime stats); it then canonicalizes commutative
|
|
281
|
-
// equality operand order so `A.x = B.y` and `B.y = A.x` collide. Everything
|
|
282
|
-
// else (join type, columns, literal values) is kept: that's the
|
|
283
|
-
// semantically meaningful part a leaf-relation-set identity would miss.
|
|
254
|
+
// Normalizes an anchor join/union node's detail for the cachingOpportunity composite
|
|
255
|
+
// fingerprint. Strips per-analysis numbering (expr ids, plan/codegen ids) and AQE's runtime
|
|
256
|
+
// BuildLeft/BuildRight choice (can flip between runs), then canonicalizes commutative equality
|
|
257
|
+
// operand order so `A.x = B.y` and `B.y = A.x` collide. Join type, columns, literals are kept.
|
|
284
258
|
export function normalizeDetail(detail ) {
|
|
285
259
|
let s = detail
|
|
286
260
|
.replace(/#\d+L?/g, '')
|
|
@@ -296,34 +270,21 @@ export function normalizeDetail(detail ) {
|
|
|
296
270
|
|
|
297
271
|
const JOIN_NAME_RE = /Join/i;
|
|
298
272
|
|
|
299
|
-
// Structural operator kind for cachingOpportunity's composite detection:
|
|
300
|
-
//
|
|
301
|
-
//
|
|
302
|
-
// plan-summary.js's visitJoin recognizes); 'union' is Spark's exact `Union`
|
|
303
|
-
// node name. CartesianProduct is deliberately excluded (out of scope, same
|
|
304
|
-
// as plan-summary.js's join handling).
|
|
273
|
+
// Structural operator kind for cachingOpportunity's composite detection: 'join' covers every
|
|
274
|
+
// Spark join physical operator; 'union' is Spark's exact `Union` node. CartesianProduct is
|
|
275
|
+
// deliberately excluded (out of scope, as in plan-summary.ts).
|
|
305
276
|
export function planOperatorKind(name ) {
|
|
306
277
|
if (JOIN_NAME_RE.test(name)) return 'join';
|
|
307
278
|
if (name === 'Union') return 'union';
|
|
308
279
|
return null;
|
|
309
280
|
}
|
|
310
281
|
|
|
311
|
-
// Single bottom-up pass
|
|
312
|
-
//
|
|
313
|
-
//
|
|
314
|
-
//
|
|
315
|
-
//
|
|
316
|
-
//
|
|
317
|
-
// computePlanShapes's default path) and merges each subtree's leaf-relation
|
|
318
|
-
// byte map (via scanRelationId, same identity cachingOpportunity's existing
|
|
319
|
-
// leaf aggregation uses) bottom-up. When a node is a join/union, its anchor
|
|
320
|
-
// fingerprint folds only its OWN normalized detail (per computePlanShapes's
|
|
321
|
-
// opts.includeDetail contract), computed here inline, in O(1), from the
|
|
322
|
-
// node's own detail plus its already-computed child fingerprints, not via a
|
|
323
|
-
// nested computePlanShapes call. `ancestorNodes` (strict ancestors, root
|
|
324
|
-
// first) is threaded down for free via the recursion's own call stack, so
|
|
325
|
-
// later nested-composite dedupe (cachingOpportunity.detect()) doesn't need a
|
|
326
|
-
// separate tree walk to determine containment.
|
|
282
|
+
// Single bottom-up O(n) pass producing one composite candidate per join/union node. Deliberately
|
|
283
|
+
// does NOT call computePlanShapes per node (subtrees overlap, that would be O(n·k) on
|
|
284
|
+
// star/snowflake joins): computes the plain fingerprint once and merges each subtree's
|
|
285
|
+
// leaf-relation byte map bottom-up. A join/union's anchor fingerprint folds only its OWN
|
|
286
|
+
// normalized detail, inline in O(1). `ancestorNodes` threads down via the call stack so
|
|
287
|
+
// nested-composite dedupe needs no separate tree walk.
|
|
327
288
|
|
|
328
289
|
|
|
329
290
|
|
|
@@ -374,17 +335,10 @@ export function findCompositeCandidates(root ) {
|
|
|
374
335
|
}
|
|
375
336
|
|
|
376
337
|
|
|
377
|
-
// First scanned
|
|
378
|
-
//
|
|
379
|
-
//
|
|
380
|
-
//
|
|
381
|
-
// duplicate groups that scan different tables (real-log bug: two unrelated
|
|
382
|
-
// "BroadcastExchange over a Project/Filter/Scan" patterns, one per dimension
|
|
383
|
-
// table, produced byte-identical findings). This is best-effort/informational
|
|
384
|
-
// only, not a uniqueness guarantee: it's null for scan-less subtrees (JDBC/
|
|
385
|
-
// Kafka/LocalRelation) and can coincide when two groups share their first-
|
|
386
|
-
// encountered leaf. See findDuplicateSubtrees's groupIndex for the actual
|
|
387
|
-
// discriminator findingId() relies on.
|
|
338
|
+
// First scanned relation identity in a subtree (pre-order), or null when none. Surfaced on
|
|
339
|
+
// duplicatePlanSubtree findings as `sampleRelation` so a user can tell apart same-shaped groups
|
|
340
|
+
// that scan different tables. Best-effort only: null for scan-less subtrees, can coincide across
|
|
341
|
+
// groups; findDuplicateSubtrees's groupIndex is the actual discriminator findingId relies on.
|
|
388
342
|
function firstLeafRelationId(node ) {
|
|
389
343
|
let found = null;
|
|
390
344
|
walkPlanTree(node, (n) => {
|
|
@@ -394,14 +348,10 @@ function firstLeafRelationId(node ) {
|
|
|
394
348
|
return found;
|
|
395
349
|
}
|
|
396
350
|
|
|
397
|
-
// Groups nodes by fingerprint, keeping
|
|
398
|
-
//
|
|
399
|
-
//
|
|
400
|
-
//
|
|
401
|
-
// smaller, fully-nested duplicate group inside an already-accepted match is
|
|
402
|
-
// dropped (a 5-node duplicate should not also emit findings for its 3-node
|
|
403
|
-
// sub-subtrees); occurrences of a smaller group that fall OUTSIDE any
|
|
404
|
-
// accepted larger match still count normally.
|
|
351
|
+
// Groups nodes by fingerprint, keeping groups of size >= minOccurrences whose subtree size is
|
|
352
|
+
// >= minSubtreeSize. De-overlap: process largest-subtree-first, and once a group is accepted mark
|
|
353
|
+
// every node in its matches "claimed" so a smaller fully-nested duplicate is dropped; occurrences
|
|
354
|
+
// of a smaller group OUTSIDE any accepted match still count.
|
|
405
355
|
|
|
406
356
|
|
|
407
357
|
|
|
@@ -417,7 +367,12 @@ export function findDuplicateSubtrees(
|
|
|
417
367
|
{ minSubtreeSize, minOccurrences } ,
|
|
418
368
|
) {
|
|
419
369
|
const { shapeOf, allNodes } = computePlanShapes(root);
|
|
420
|
-
|
|
370
|
+
// Defensive, not load-bearing: computePlanShapes's visit() already skips
|
|
371
|
+
// straight through a read node to its write half's real children, so no
|
|
372
|
+
// write-half node is ever pushed into allNodes in the first place, this
|
|
373
|
+
// filter can structurally never exclude anything. Kept in case that
|
|
374
|
+
// invariant ever changes upstream.
|
|
375
|
+
const eligible = allNodes.filter(n => shapeOf.get(n) .size >= minSubtreeSize && n.exchangeRole !== 'write');
|
|
421
376
|
|
|
422
377
|
const groups = new Map ();
|
|
423
378
|
for (const n of eligible) {
|
|
@@ -445,14 +400,10 @@ export function findDuplicateSubtrees(
|
|
|
445
400
|
rootName: unclaimed[0].name,
|
|
446
401
|
subtreeSize: shapeOf.get(unclaimed[0]) .size,
|
|
447
402
|
occurrences: unclaimed.length,
|
|
448
|
-
isExchangeRoot:
|
|
403
|
+
isExchangeRoot: isExchangeNode(unclaimed[0]),
|
|
449
404
|
sampleRelation: firstLeafRelationId(unclaimed[0]),
|
|
450
|
-
// Deterministic position within this execution's group list
|
|
451
|
-
// is best-effort (null
|
|
452
|
-
// JDBC/Kafka/LocalRelation sources, or identical when two groups happen to share
|
|
453
|
-
// their first-encountered leaf) and is NOT sufficient on its own to guarantee two
|
|
454
|
-
// structurally-distinct groups get distinct finding ids; groupIndex is the actual
|
|
455
|
-
// uniqueness guarantee findingId() relies on.
|
|
405
|
+
// Deterministic position within this execution's group list: the actual uniqueness
|
|
406
|
+
// guarantee findingId relies on, since sampleRelation is best-effort (null or coincident).
|
|
456
407
|
groupIndex: results.length,
|
|
457
408
|
nodes: unclaimed,
|
|
458
409
|
});
|
|
@@ -460,21 +411,32 @@ export function findDuplicateSubtrees(
|
|
|
460
411
|
return results;
|
|
461
412
|
}
|
|
462
413
|
|
|
463
|
-
//
|
|
464
|
-
//
|
|
465
|
-
//
|
|
466
|
-
//
|
|
414
|
+
// Fingerprint matching compares operator + metric names only, not literal values or expr IDs
|
|
415
|
+
// (see the finding's validationRequired text), so a small pattern repeated the bare minimum
|
|
416
|
+
// number of times is the case most likely to be coincidental rather than real duplicated work.
|
|
417
|
+
// A bigger matched subtree, or more repeats, are each on their own strong corroborating evidence
|
|
418
|
+
// that the match is real: the odds of two semantically-different query branches producing an
|
|
419
|
+
// identical operator-name sequence shrink fast as the sequence grows or repeats.
|
|
420
|
+
function duplicateSubtreeConfidence(
|
|
421
|
+
subtreeSize ,
|
|
422
|
+
occurrences ,
|
|
423
|
+
thresholds ,
|
|
424
|
+
) {
|
|
425
|
+
if (subtreeSize <= thresholds.minSubtreeSize && occurrences <= thresholds.minOccurrences) return 'low';
|
|
426
|
+
if (subtreeSize >= thresholds.minSubtreeSize * 2 || occurrences >= thresholds.minOccurrences + 2) return 'high';
|
|
427
|
+
return 'medium';
|
|
428
|
+
}
|
|
429
|
+
|
|
430
|
+
// Exact metric names Spark emits, verified against a real SQLExecutionStart's sparkPlanInfo.
|
|
431
|
+
// The write-side byte metric is "written output", not "size of written files".
|
|
467
432
|
const FILES_READ_COUNT = 'number of files read';
|
|
468
433
|
const FILES_READ_BYTES = 'size of files read';
|
|
469
434
|
const FILES_WRITTEN_COUNT = 'number of written files';
|
|
470
435
|
const FILES_WRITTEN_BYTES = 'written output';
|
|
471
436
|
|
|
472
|
-
// Byte size of a join-side subtree for broadcast sizing
|
|
473
|
-
//
|
|
474
|
-
//
|
|
475
|
-
// beneath it (e.g. an Exchange's "data size" already reflects everything it
|
|
476
|
-
// shuffled), so summing further down would double-count. Only recurses into
|
|
477
|
-
// children when the current node carries no such metric.
|
|
437
|
+
// Byte size of a join-side subtree for broadcast sizing. Stops
|
|
438
|
+
// descending at a node with a "data size" metric: that value already aggregates everything
|
|
439
|
+
// beneath it, so summing further would double-count. Only recurses when no such metric.
|
|
478
440
|
function sumBoundarySize(node ) {
|
|
479
441
|
const m = (node.metrics ?? []).find(x => x.name === 'data size');
|
|
480
442
|
if (m) return m.value;
|
|
@@ -483,10 +445,8 @@ function sumBoundarySize(node ) {
|
|
|
483
445
|
return sum;
|
|
484
446
|
}
|
|
485
447
|
|
|
486
|
-
// Nodes that
|
|
487
|
-
//
|
|
488
|
-
// so the implicated stageIds line up with the size that was actually
|
|
489
|
-
// compared, however deep that turns out to live.
|
|
448
|
+
// Nodes that fed sumBoundarySize's total: mirrors its recursion exactly so implicated stageIds
|
|
449
|
+
// line up with the size actually compared.
|
|
490
450
|
function boundarySizeContributors(node ) {
|
|
491
451
|
const m = (node.metrics ?? []).find((x) => x.name === 'data size');
|
|
492
452
|
if (m) return [node];
|
|
@@ -516,10 +476,8 @@ function maxMedianRatio(
|
|
|
516
476
|
|
|
517
477
|
|
|
518
478
|
|
|
519
|
-
// Machine-readable detector metadata for the evidence report
|
|
520
|
-
//
|
|
521
|
-
// portable report record exactly which detector + threshold set produced each
|
|
522
|
-
// finding, so evidence stays reproducible as detectors evolve.
|
|
479
|
+
// Machine-readable detector metadata for the evidence report (no `detect` closure), so a
|
|
480
|
+
// portable report records which detector + thresholds produced each finding.
|
|
523
481
|
export function detectorCatalog() {
|
|
524
482
|
return DETECTORS.map((d) => ({
|
|
525
483
|
type: d.type,
|
|
@@ -530,21 +488,11 @@ export function detectorCatalog() {
|
|
|
530
488
|
}));
|
|
531
489
|
}
|
|
532
490
|
|
|
533
|
-
// True task-duration skew ratio
|
|
534
|
-
//
|
|
535
|
-
//
|
|
536
|
-
//
|
|
537
|
-
//
|
|
538
|
-
//
|
|
539
|
-
// Fields widened to optional: the `skew` detector below only ever calls this
|
|
540
|
-
// with an already-finalized `DetectorStage` (all four always numeric by the
|
|
541
|
-
// time `analyze()` runs), but `cli/budgets.ts`'s `checkSkew` calls it
|
|
542
|
-
// directly against `AppModel.stages`: real `Stage` records for a stage that
|
|
543
|
-
// never received a `StageCompleted` event (e.g. an unfinished run) genuinely
|
|
544
|
-
// lack these fields (see `types.ts`'s `Stage`). The `as number` casts below
|
|
545
|
-
// preserve the original behavior byte-for-byte: dividing through an absent
|
|
546
|
-
// field still naturally produces `NaN` (as it always has for untyped JS
|
|
547
|
-
// callers), rather than adding a new guard that would change the result.
|
|
491
|
+
// True task-duration skew ratio: P95/median once enough tasks to trust P95, else max/median.
|
|
492
|
+
// Null when no measurable median (p50 === 0). Exported so cli/budgets.ts recomputes the same
|
|
493
|
+
// ratio rather than reading findings floored at ratioWarn.
|
|
494
|
+
// Fields optional: budgets.ts calls this against raw AppModel.stages, whose stages may lack
|
|
495
|
+
// these fields (unfinished run). The `as number` casts keep behavior: an absent field yields NaN.
|
|
548
496
|
export function computeSkewRatio(
|
|
549
497
|
stage ,
|
|
550
498
|
minTasksForP95 ,
|
|
@@ -556,18 +504,10 @@ export function computeSkewRatio(
|
|
|
556
504
|
: { ratio: (max ) / (p50 ), metric: 'max/median' };
|
|
557
505
|
}
|
|
558
506
|
|
|
559
|
-
// Absolute-magnitude floor
|
|
560
|
-
// a
|
|
561
|
-
//
|
|
562
|
-
//
|
|
563
|
-
// absolute waste is real in a run that only took seconds; a fixed-ms floor
|
|
564
|
-
// can't scale between those. Used by the `skew` and `straggler` entries
|
|
565
|
-
// below, each gated via `clippedWasteMs` (below) against the *same
|
|
566
|
-
// occupancy-clipped* wall-clock figure
|
|
567
|
-
// src/impact-estimator.ts displays as that finding's savings, not the raw
|
|
568
|
-
// pre-clip delta, which can stay large after clipping collapses the
|
|
569
|
-
// recoverable time to near zero (the stage's own longest task already
|
|
570
|
-
// accounts for nearly all of its wall-clock window).
|
|
507
|
+
// Absolute-magnitude floor as a % of app runtime, not a fixed ms constant: a skew/straggler
|
|
508
|
+
// ratio on a few ms is noise in an hours-long run but real in a seconds-long one; a fixed-ms
|
|
509
|
+
// floor can't scale. Used by skew/straggler, gated via clippedWasteMs against the same
|
|
510
|
+
// occupancy-clipped figure impact-estimator.ts displays as savings.
|
|
571
511
|
// NOT SOURCED: floor percentages are our own noise floor, unvalidated.
|
|
572
512
|
function computeAppDurationMs(ctx ) {
|
|
573
513
|
const app = ctx?.app;
|
|
@@ -576,37 +516,38 @@ function computeAppDurationMs(ctx ) {
|
|
|
576
516
|
return durationMs > 0 ? durationMs : null;
|
|
577
517
|
}
|
|
578
518
|
|
|
579
|
-
// Unknown app timing
|
|
580
|
-
// finding; it just skips the floor gate, preserving prior ratio-only
|
|
581
|
-
// behavior when total runtime can't be computed.
|
|
519
|
+
// Unknown app timing never suppresses a finding; it just skips the floor gate.
|
|
582
520
|
function meetsRuntimeFloor(wasteMs , appDurationMs , floorPct ) {
|
|
583
521
|
return appDurationMs == null || wasteMs >= appDurationMs * floorPct;
|
|
584
522
|
}
|
|
585
523
|
|
|
586
|
-
// Runs a raw waste delta through the same
|
|
587
|
-
//
|
|
588
|
-
//
|
|
589
|
-
// wall-clock time rather than a delta that a physical floor (the stage's own
|
|
590
|
-
// longest task) may leave almost entirely unrecoverable. Falls back to the
|
|
591
|
-
// raw delta when occupancy data isn't available for this stage (ctx omitted,
|
|
592
|
-
// or the stage was excluded from the occupancy sweep for having <= 0
|
|
593
|
-
// duration), same as the pre-existing ratio-only behavior for unknown app
|
|
594
|
-
// timing above.
|
|
524
|
+
// Runs a raw waste delta through the same occupancy clip impact-estimator.ts applies before
|
|
525
|
+
// display, so the runtime floor checks recoverable wall-clock, not a delta a physical floor
|
|
526
|
+
// leaves unrecoverable. Falls back to the raw delta when occupancy data is unavailable.
|
|
595
527
|
function clippedWasteMs(wasteMs , stageId , ctx ) {
|
|
596
528
|
if (!ctx) return wasteMs;
|
|
597
529
|
const est = estimateSingleStage(wasteMs, stageId, ctx.stages , ctx.occupancy);
|
|
598
530
|
return est ? est.wallClock.high : wasteMs;
|
|
599
531
|
}
|
|
600
532
|
|
|
601
|
-
// cacheUtilization's
|
|
602
|
-
//
|
|
603
|
-
// pickDominantReason above). Both variants share the same confidence/
|
|
604
|
-
// validationRequired text: the ratio is a point-in-time storage snapshot
|
|
605
|
-
// from stage-submission events (src/event-handlers.js's mergeStageRddInfo),
|
|
606
|
-
// not a runtime block-access read-count.
|
|
533
|
+
// Shared by cacheUtilization's two variants: the ratio is a point-in-time storage snapshot from
|
|
534
|
+
// stage-submission events, not a runtime block-access read-count.
|
|
607
535
|
const CACHE_UTILIZATION_VALIDATION =
|
|
608
536
|
"This ratio is a point-in-time storage snapshot from stage-submission events, not a runtime read-count. Confirm against the Spark UI's Storage tab before acting.";
|
|
609
537
|
|
|
538
|
+
// Both cachedRatio and diskRatio are percentages of an RDD's partition/byte count: with few
|
|
539
|
+
// partitions, one partition flipping cached/evicted (or disk-resident/memory-resident) swings the
|
|
540
|
+
// reported percentage by a large amount, so the point estimate is noisy. More partitions average
|
|
541
|
+
// that noise out into a stable ratio. numPartitions is the only sample-size signal
|
|
542
|
+
// DetectorRddInfo carries, so it drives confidence for both variants rather than a flat guess.
|
|
543
|
+
// 10/50 mirror this file's other "trust the sample" cutoffs (skew's minTasksForP95: 20,
|
|
544
|
+
// coreLocality's minTasks: 50).
|
|
545
|
+
function cacheSampleConfidence(numPartitions ) {
|
|
546
|
+
if (numPartitions < 10) return 'low';
|
|
547
|
+
if (numPartitions >= 50) return 'high';
|
|
548
|
+
return 'medium';
|
|
549
|
+
}
|
|
550
|
+
|
|
610
551
|
function partialCacheFinding(rdd , cachedRatio , impactBand ) {
|
|
611
552
|
const rddName = rdd.name || `RDD ${rdd.id}`;
|
|
612
553
|
const cachedPct = Math.round(cachedRatio * 100);
|
|
@@ -615,7 +556,7 @@ function partialCacheFinding(rdd , cachedRatio , impactBa
|
|
|
615
556
|
type: 'cacheUtilization', variant: 'partialCache', stageId: null,
|
|
616
557
|
rddId: rdd.id, rddName,
|
|
617
558
|
impactBand, metric: 'cachedRatio', value: cachedPct,
|
|
618
|
-
confidence:
|
|
559
|
+
confidence: cacheSampleConfidence(rdd.numPartitions),
|
|
619
560
|
validationRequired: CACHE_UTILIZATION_VALIDATION,
|
|
620
561
|
memorySize: rdd.memorySize, diskSize: rdd.diskSize,
|
|
621
562
|
numCachedPartitions: rdd.numCachedPartitions, numPartitions: rdd.numPartitions,
|
|
@@ -630,7 +571,7 @@ function diskSpilloverFinding(rdd , diskRatio , impactBan
|
|
|
630
571
|
type: 'cacheUtilization', variant: 'diskSpillover', stageId: null,
|
|
631
572
|
rddId: rdd.id, rddName,
|
|
632
573
|
impactBand, metric: 'diskRatio', value: diskPct,
|
|
633
|
-
confidence:
|
|
574
|
+
confidence: cacheSampleConfidence(rdd.numPartitions),
|
|
634
575
|
validationRequired: CACHE_UTILIZATION_VALIDATION,
|
|
635
576
|
memorySize: rdd.memorySize, diskSize: rdd.diskSize,
|
|
636
577
|
numCachedPartitions: rdd.numCachedPartitions, numPartitions: rdd.numPartitions,
|
|
@@ -638,19 +579,11 @@ function diskSpilloverFinding(rdd , diskRatio , impactBan
|
|
|
638
579
|
};
|
|
639
580
|
}
|
|
640
581
|
|
|
641
|
-
// Entry shape for every item
|
|
642
|
-
//
|
|
643
|
-
//
|
|
644
|
-
//
|
|
645
|
-
//
|
|
646
|
-
// the only real callers), and unifying those four into one `TTarget` would
|
|
647
|
-
// need either an unsound cast or a discriminated-union-of-detectors redesign
|
|
648
|
-
// this migration task doesn't ask for. Each entry below still gets a
|
|
649
|
-
// precisely-typed `detect` by annotating its own `target`/`ctx` parameters
|
|
650
|
-
// directly: object-literal methods (this `detect(target) {}` shorthand, not
|
|
651
|
-
// an arrow function assigned to a property) are checked bivariantly against
|
|
652
|
-
// an interface's method parameter types, so a narrower, concrete annotation
|
|
653
|
-
// here does not conflict with `Detector`'s `unknown` declaration.
|
|
582
|
+
// Entry shape for every DETECTORS item. TTarget stays `unknown` at the array level: detect's
|
|
583
|
+
// real first-arg varies by scope (DetectorStage/DetectorSqlExec/DetectorCtx/{ app }), and
|
|
584
|
+
// unifying them would need an unsound cast or a discriminated-union redesign. Each entry gets a
|
|
585
|
+
// precise detect by annotating its own params: object-literal method params are checked
|
|
586
|
+
// bivariantly, so a narrower annotation here doesn't conflict with the `unknown` declaration.
|
|
654
587
|
|
|
655
588
|
|
|
656
589
|
|
|
@@ -658,14 +591,11 @@ function diskSpilloverFinding(rdd , diskRatio , impactBan
|
|
|
658
591
|
|
|
659
592
|
|
|
660
593
|
|
|
661
|
-
|
|
662
|
-
|
|
663
|
-
|
|
664
|
-
|
|
665
|
-
|
|
594
|
+
|
|
666
595
|
|
|
667
596
|
|
|
668
597
|
|
|
598
|
+
|
|
669
599
|
|
|
670
600
|
|
|
671
601
|
|
|
@@ -678,12 +608,108 @@ function diskSpilloverFinding(rdd , diskRatio , impactBan
|
|
|
678
608
|
// *Ratio = multiplicative factor
|
|
679
609
|
// *Share/*Rate/*Util = 0–1 fraction (normalized)
|
|
680
610
|
|
|
681
|
-
//
|
|
682
|
-
//
|
|
683
|
-
// impact-band floor instead of hand-copying the literals.
|
|
611
|
+
// straggler's noise-floor thresholds (NOT SOURCED: unvalidated), exported so impact-band.ts
|
|
612
|
+
// reuses the same figures instead of hand-copying.
|
|
684
613
|
export const STRAGGLER_FLOOR_PCT_WARN = 0.005;
|
|
685
614
|
export const STRAGGLER_FLOOR_PCT_CRIT = 0.02;
|
|
686
615
|
|
|
616
|
+
// A ratio just past ratioWarn is the case most likely to be ordinary task-duration variance
|
|
617
|
+
// rather than real skew; a ratio many multiples past it (a 50x P95/median vs. a 3.1x one) is
|
|
618
|
+
// unambiguous. 1.5x/5x mirror this file's other "just past the floor vs. clearly past it" splits
|
|
619
|
+
// (duplicateSubtreeConfidence's 2x, cacheSampleConfidence's 10/50-partition cutoffs).
|
|
620
|
+
function skewConfidence(ratio , ratioWarn ) {
|
|
621
|
+
if (ratio <= ratioWarn * 1.5) return 'low';
|
|
622
|
+
if (ratio >= ratioWarn * 5) return 'high';
|
|
623
|
+
return 'medium';
|
|
624
|
+
}
|
|
625
|
+
|
|
626
|
+
// warnPct100/lowInfoPct100 are this detector's own two thresholds; scale confidence as a multiple
|
|
627
|
+
// of whichever one gates the branch that fired, the same way skewConfidence scales off ratioWarn.
|
|
628
|
+
// High-GC: a pct just past warnPct100 (10%) is likely normal variance, 3x past it is unambiguous.
|
|
629
|
+
// Low-GC ("cost", over-provisioned): a pct just under lowInfoPct100 (5%) is borderline, a pct near
|
|
630
|
+
// zero is unambiguous idle GC.
|
|
631
|
+
function gcConfidence(
|
|
632
|
+
pct ,
|
|
633
|
+
thresholds ,
|
|
634
|
+
direction ,
|
|
635
|
+
) {
|
|
636
|
+
if (direction === 'high') {
|
|
637
|
+
if (pct <= thresholds.warnPct100 * 1.5) return 'low';
|
|
638
|
+
if (pct >= thresholds.warnPct100 * 3) return 'high';
|
|
639
|
+
return 'medium';
|
|
640
|
+
}
|
|
641
|
+
if (pct >= thresholds.lowInfoPct100 * 0.66) return 'low';
|
|
642
|
+
if (pct <= thresholds.lowInfoPct100 * 0.2) return 'high';
|
|
643
|
+
return 'medium';
|
|
644
|
+
}
|
|
645
|
+
|
|
646
|
+
// warnFloor gates the finding, so a value just past it is the weakest evidence this detector can
|
|
647
|
+
// produce. highFloor is critPct: straggler's own second (speculative-share) tier, 2x warnPct
|
|
648
|
+
// (0.10 -> 0.20); reused as the high-confidence bar for the stragglerShare path too since that
|
|
649
|
+
// metric has no dedicated critical tier of its own (see the "no dedicated critical tier" comment
|
|
650
|
+
// on the straggler detector) but is the same 0-1 task-share magnitude.
|
|
651
|
+
function stragglerConfidence(shareValue , warnFloor , highFloor ) {
|
|
652
|
+
if (shareValue < warnFloor * 1.5) return 'low';
|
|
653
|
+
if (shareValue >= highFloor) return 'high';
|
|
654
|
+
return 'medium';
|
|
655
|
+
}
|
|
656
|
+
|
|
657
|
+
// minWasteMs is speculationWaste's own floor; a run that barely clears it (under 1.5x) is the
|
|
658
|
+
// weakest case, one that clears it several times over (4x, i.e. 4 minutes against a 1-minute
|
|
659
|
+
// floor) is unambiguous.
|
|
660
|
+
function speculationWasteConfidence(wastedMs , minWasteMs ) {
|
|
661
|
+
if (wastedMs <= minWasteMs * 1.5) return 'low';
|
|
662
|
+
if (wastedMs >= minWasteMs * 4) return 'high';
|
|
663
|
+
return 'medium';
|
|
664
|
+
}
|
|
665
|
+
|
|
666
|
+
// The finding already gates on wastedMBSeconds > wasteBufferMultiplier * usedMBSeconds, i.e. a
|
|
667
|
+
// ratio of 1 at the floor; scale confidence off that same ratio the way skewConfidence scales off
|
|
668
|
+
// ratioWarn, instead of introducing a second, unrelated multiplier.
|
|
669
|
+
function memoryWasteConfidence(wastedMBSeconds , usedMBSeconds , wasteBufferMultiplier ) {
|
|
670
|
+
const ratio = usedMBSeconds > 0 ? wastedMBSeconds / (wasteBufferMultiplier * usedMBSeconds) : Infinity;
|
|
671
|
+
if (ratio <= 1.5) return 'low';
|
|
672
|
+
if (ratio >= 3) return 'high';
|
|
673
|
+
return 'medium';
|
|
674
|
+
}
|
|
675
|
+
|
|
676
|
+
// Two independent weak spots can each undercut this finding: a ratio just past warnRatio (could
|
|
677
|
+
// be one bad stage), or too few sampled tasks (mirrors cacheSampleConfidence's use of
|
|
678
|
+
// numPartitions as a sample-size signal). Report whichever signal is weaker rather than
|
|
679
|
+
// averaging them away. critRatio is this detector's own existing second tier, reused directly as
|
|
680
|
+
// the ratio high-bar; minTasks*2/*4 mirror the same "2x floor is still weak, 4x is strong" spread
|
|
681
|
+
// used elsewhere in this file.
|
|
682
|
+
function coreLocalityConfidence(
|
|
683
|
+
ratio ,
|
|
684
|
+
totalTasks ,
|
|
685
|
+
thresholds ,
|
|
686
|
+
) {
|
|
687
|
+
const rank = { low: 0, medium: 1, high: 2 } ;
|
|
688
|
+
const ratioTier = ratio < thresholds.warnRatio * 1.5 ? 'low' : ratio >= thresholds.critRatio ? 'high' : 'medium';
|
|
689
|
+
const sampleTier = totalTasks < thresholds.minTasks * 2 ? 'low' : totalTasks >= thresholds.minTasks * 4 ? 'high' : 'medium';
|
|
690
|
+
return rank[ratioTier] <= rank[sampleTier] ? ratioTier : sampleTier;
|
|
691
|
+
}
|
|
692
|
+
|
|
693
|
+
// warningPct/criticalPct are this detector's own two tiers (0.30/0.60); a churn rate just past
|
|
694
|
+
// warningPct is the borderline call the impact-band split already treats as the weaker tier, so
|
|
695
|
+
// reuse criticalPct directly as the high-confidence bar instead of inventing a third figure.
|
|
696
|
+
function autoscalingChurnConfidence(shortLivedPct , warningPct , criticalPct ) {
|
|
697
|
+
if (shortLivedPct <= warningPct * 1.5) return 'low';
|
|
698
|
+
if (shortLivedPct >= criticalPct) return 'high';
|
|
699
|
+
return 'medium';
|
|
700
|
+
}
|
|
701
|
+
|
|
702
|
+
// minExecutions is the bare minimum occurrence count this detector will even emit a finding for;
|
|
703
|
+
// a match at exactly that count is the weakest reuse signal (as likely to be coincidental overlap
|
|
704
|
+
// as real shared work), while 3x the floor is several independent executions all hitting the same
|
|
705
|
+
// relation/composite shape, unambiguous. Mirrors duplicateSubtreeConfidence's occurrences handling
|
|
706
|
+
// for the same reason: repetition count is the strength signal for a structural-match detector.
|
|
707
|
+
function cachingReuseConfidence(occurrences , minExecutions ) {
|
|
708
|
+
if (occurrences <= minExecutions) return 'low';
|
|
709
|
+
if (occurrences >= minExecutions * 3) return 'high';
|
|
710
|
+
return 'medium';
|
|
711
|
+
}
|
|
712
|
+
|
|
687
713
|
export const DETECTORS = [
|
|
688
714
|
{
|
|
689
715
|
type: 'skew', scope: 'stage', order: 30, fixEffort: 'code', version: 1,
|
|
@@ -700,9 +726,8 @@ export const DETECTORS = [
|
|
|
700
726
|
if (result === null) return null;
|
|
701
727
|
const { ratio, metric } = result;
|
|
702
728
|
if (ratio <= this.thresholds.ratioWarn) return null;
|
|
703
|
-
// Same absolute delta
|
|
704
|
-
//
|
|
705
|
-
// so the gate agrees with what's actually displayed.
|
|
729
|
+
// Same absolute delta impact-estimator.ts's 'skew' case reports as savings; clipped the
|
|
730
|
+
// same way before the floor check so the gate agrees with what's displayed.
|
|
706
731
|
const wasteMs = Math.max(0, metric === 'P95/median' ? stage.taskDurationP95 - stage.taskDurationP50 : stage.taskDurationMax - stage.taskDurationP50);
|
|
707
732
|
const appDurationMs = computeAppDurationMs(ctx);
|
|
708
733
|
const floorWasteMs = clippedWasteMs(wasteMs, stage.id, ctx);
|
|
@@ -712,6 +737,8 @@ export const DETECTORS = [
|
|
|
712
737
|
type: 'skew', stageId: stage.id,
|
|
713
738
|
impactBand: 'warning',
|
|
714
739
|
metric, value,
|
|
740
|
+
confidence: skewConfidence(ratio, this.thresholds.ratioWarn),
|
|
741
|
+
validationRequired: 'This finding is gated by a 0.5% runtime-floor threshold, our own noise floor for this metric.',
|
|
715
742
|
recommendation: `Task duration ratio (${metric}) is ${value}×: for join-driven skew, enable AQE skew-join handling (spark.sql.adaptive.skewJoin.enabled); otherwise salt the key or repartition on a better key to reduce task skew.`,
|
|
716
743
|
};
|
|
717
744
|
},
|
|
@@ -753,11 +780,9 @@ export const DETECTORS = [
|
|
|
753
780
|
});
|
|
754
781
|
}
|
|
755
782
|
}
|
|
756
|
-
// TaskStageSkew: straggler cost vs stage wall-clock. Skip near-zero duration.
|
|
757
|
-
//
|
|
758
|
-
//
|
|
759
|
-
// exactly zero on every firing (see impact-estimator.ts's costOnly branch below),
|
|
760
|
-
// so there is no wall-clock-backed impact-band tier left to gate on.
|
|
783
|
+
// TaskStageSkew: straggler cost vs stage wall-clock. Skip near-zero duration. Always info
|
|
784
|
+
// like its siblings: this trigger forces the occupancy-clipped estimate to exactly zero on
|
|
785
|
+
// every firing, so there's no wall-clock-backed tier left to gate on.
|
|
761
786
|
const stageDurationMs = (stage.completedAt ?? 0) - (stage.submittedAt ?? 0);
|
|
762
787
|
if (stageDurationMs > 0) {
|
|
763
788
|
const ratio = stage.taskDurationMax / stageDurationMs;
|
|
@@ -808,9 +833,8 @@ export const DETECTORS = [
|
|
|
808
833
|
const out = [];
|
|
809
834
|
const { shuffleReadP50: p50, shuffleReadMax: max, shuffleReadBytes: total, taskCount } = stage;
|
|
810
835
|
if (max > this.thresholds.skewRatio * p50 && max > this.thresholds.skewFloorBytes) {
|
|
811
|
-
// p50 can be 0 (
|
|
812
|
-
//
|
|
813
|
-
// back to median-free phrasing instead of dividing by p50.
|
|
836
|
+
// p50 can be 0 (over half the shuffle partitions empty): a ratio against zero renders
|
|
837
|
+
// "Infinity×", so fall back to median-free phrasing.
|
|
814
838
|
const ratioText = p50 > 0
|
|
815
839
|
? `${Math.round(max / p50 * 10) / 10}× the median (${formatBytes(p50)})`
|
|
816
840
|
: `far larger than the median (${formatBytes(p50)}, effectively empty)`;
|
|
@@ -828,6 +852,10 @@ export const DETECTORS = [
|
|
|
828
852
|
});
|
|
829
853
|
}
|
|
830
854
|
if (max >= this.thresholds.maxPartBytes) {
|
|
855
|
+
// Fixed 'critical': an OOM/crash-risk safety signal, not a time-waste one. Exempted in
|
|
856
|
+
// impact-band.ts's deriveImpactBand from the wall-clock-based overwrite every other
|
|
857
|
+
// finding here gets, so a long-running job can't demote an active crash risk to 'info'
|
|
858
|
+
// just because the modeled time savings are a small fraction of total runtime.
|
|
831
859
|
out.push({
|
|
832
860
|
type: 'partitionSizing', stageId: stage.id, impactBand: 'critical',
|
|
833
861
|
rule: 'maxPartitionTooBig', metric: 'shuffleReadMax', value: max,
|
|
@@ -864,18 +892,18 @@ export const DETECTORS = [
|
|
|
864
892
|
{
|
|
865
893
|
type: 'gc', scope: 'stage', order: 50, fixEffort: 'config', version: 1,
|
|
866
894
|
docAnchor: '#bottleneck-gc',
|
|
895
|
+
validationRequired: 'This finding is gated by a 10-second minimum-runtime floor, our own noise floor for this metric.',
|
|
867
896
|
thresholds: {
|
|
868
897
|
warnPct100: 10,
|
|
869
|
-
// Descending tier:
|
|
898
|
+
// Descending tier: ExecutorGcHeuristic, ported as-is.
|
|
870
899
|
lowInfoPct100: 5,
|
|
871
|
-
// NOT SOURCED: our own noise floor so a stage that barely ran
|
|
872
|
-
// near 0 or wildly inflated by a tiny denominator) does not flag,
|
|
873
|
-
// in either direction.
|
|
900
|
+
// NOT SOURCED: our own noise floor so a stage that barely ran doesn't flag either direction.
|
|
874
901
|
minRunTimeMs: 10000,
|
|
875
902
|
},
|
|
876
903
|
detect(
|
|
877
904
|
|
|
878
905
|
|
|
906
|
+
|
|
879
907
|
|
|
880
908
|
stage ,
|
|
881
909
|
) {
|
|
@@ -887,6 +915,7 @@ export const DETECTORS = [
|
|
|
887
915
|
type: 'gc', stageId: stage.id,
|
|
888
916
|
impactBand: 'warning',
|
|
889
917
|
metric: 'gcPct', value,
|
|
918
|
+
confidence: gcConfidence(pct, this.thresholds, 'high'), validationRequired: this.validationRequired,
|
|
890
919
|
recommendation: `GC consumed ${value}% of executor run time: reduce object creation, use primitive types, avoid UDFs, increase executor memory.`,
|
|
891
920
|
};
|
|
892
921
|
}
|
|
@@ -898,6 +927,7 @@ export const DETECTORS = [
|
|
|
898
927
|
type: 'gc', stageId: stage.id, direction: 'low',
|
|
899
928
|
impactBand: 'info',
|
|
900
929
|
metric: 'gcPct', value,
|
|
930
|
+
confidence: gcConfidence(pct, this.thresholds, 'low'), validationRequired: this.validationRequired,
|
|
901
931
|
recommendation: `GC consumed only ${value}% of executor run time: memory may be over-provisioned; consider reducing spark.executor.memory for cost savings.`,
|
|
902
932
|
};
|
|
903
933
|
}
|
|
@@ -909,12 +939,9 @@ export const DETECTORS = [
|
|
|
909
939
|
docAnchor: '#bottleneck-slow-host',
|
|
910
940
|
thresholds: {
|
|
911
941
|
minHosts: 3, minTasks: 15, ratioWarn: 2.0, minShare: 0.20, shareWarn: 0.75, taskShareWarn: 0.50, ratioTiers: [1.33, 1.78, 3.16, 10],
|
|
912
|
-
// Absolute-magnitude floors (mirrors computeSpillMagnitude's ratio+floor
|
|
913
|
-
//
|
|
914
|
-
//
|
|
915
|
-
// real slow-host problem. 1s is well above typical per-task scheduling
|
|
916
|
-
// jitter but well below the tens-of-seconds+ means genuine slow-host
|
|
917
|
-
// stages exhibit; 64MB mirrors the spill detector's disk-skew floor.
|
|
942
|
+
// Absolute-magnitude floors (mirrors computeSpillMagnitude's ratio+floor pattern): on short
|
|
943
|
+
// stages, sub-second/sub-64MB host differences produce huge noise ratios. 1s is above
|
|
944
|
+
// per-task jitter but below genuine slow-host stages; 64MB mirrors the spill disk-skew floor.
|
|
918
945
|
floorMs: 1000, floorBytes: 64 * MB,
|
|
919
946
|
},
|
|
920
947
|
detect(
|
|
@@ -946,7 +973,7 @@ export const DETECTORS = [
|
|
|
946
973
|
// `value` is a ratio; the estimator needs the absolute per-host mean.
|
|
947
974
|
hostMeanMs: h.mean,
|
|
948
975
|
host: h.host, hostTaskShare: Math.round(share * 100) / 100,
|
|
949
|
-
recommendation:
|
|
976
|
+
recommendation: `${h.host} may just hold data locality for its tasks or carry one heavy stage, not necessarily a hardware fault: check what it was running, and consider enabling spark.speculation to relaunch a lagging task automatically.`,
|
|
950
977
|
});
|
|
951
978
|
}
|
|
952
979
|
}
|
|
@@ -1012,10 +1039,8 @@ export const DETECTORS = [
|
|
|
1012
1039
|
type: 'stageSlowness', scope: 'stage', order: 65, fixEffort: 'code', version: 2,
|
|
1013
1040
|
docAnchor: '#bottleneck-stage-slowness',
|
|
1014
1041
|
thresholds: { infoMin: 15 },
|
|
1015
|
-
// Cross-detector suppression (
|
|
1016
|
-
//
|
|
1017
|
-
// Requires this entry to be declared AFTER slowHost in DETECTORS so
|
|
1018
|
-
// slowHost findings are already in `out`.
|
|
1042
|
+
// Cross-detector suppression (see "Detector contract" in detector-contract.md). Requires this
|
|
1043
|
+
// entry to be declared AFTER slowHost in DETECTORS so slowHost findings are already in `out`.
|
|
1019
1044
|
suppressWhen(finding, out) {
|
|
1020
1045
|
return out.some(o => o.type === 'slowHost' && o.stageId === finding.stageId);
|
|
1021
1046
|
},
|
|
@@ -1023,9 +1048,8 @@ export const DETECTORS = [
|
|
|
1023
1048
|
|
|
1024
1049
|
stage ,
|
|
1025
1050
|
) {
|
|
1026
|
-
// Basis is real wall-clock stage duration, not per-executor average
|
|
1027
|
-
//
|
|
1028
|
-
// stageDurationMs computation.
|
|
1051
|
+
// Basis is real wall-clock stage duration, not per-executor average; the impact-estimator
|
|
1052
|
+
// formula reuses this exact stageDurationMs computation.
|
|
1029
1053
|
const stageDurationMs = (stage.completedAt ?? 0) - (stage.submittedAt ?? 0);
|
|
1030
1054
|
if (!(stageDurationMs > 0)) return null;
|
|
1031
1055
|
const durationMinutes = stageDurationMs / 60000;
|
|
@@ -1036,7 +1060,7 @@ export const DETECTORS = [
|
|
|
1036
1060
|
return {
|
|
1037
1061
|
type: 'stageSlowness', stageId: stage.id, impactBand,
|
|
1038
1062
|
metric: 'stageDurationMinutes', value,
|
|
1039
|
-
recommendation: `This stage ran ${value} minutes with no more specific cause flagged:
|
|
1063
|
+
recommendation: `This stage ran ${value} minutes with no more specific cause flagged: often a partition-count problem, raise parallelism via spark.sql.shuffle.partitions or spark.default.parallelism, or check for a large per-task data volume driving heavy shuffle and spill.`,
|
|
1040
1064
|
};
|
|
1041
1065
|
},
|
|
1042
1066
|
},
|
|
@@ -1050,6 +1074,9 @@ export const DETECTORS = [
|
|
|
1050
1074
|
type: 'stageFailed', stageId: stage.id, impactBand: 'critical',
|
|
1051
1075
|
variant: 'stageFailure',
|
|
1052
1076
|
metric: 'stageFailureReason', value: stage.stageFailureReason,
|
|
1077
|
+
numTasks: stage.taskCount,
|
|
1078
|
+
memoryBytesSpilled: stage.memoryBytesSpilled,
|
|
1079
|
+
failedTaskDetails: stage.failedTaskSamples ?? [],
|
|
1053
1080
|
recommendation: `This stage attempt failed outright. Inspect the driver log for the failure reason and the job that triggered it.`,
|
|
1054
1081
|
};
|
|
1055
1082
|
},
|
|
@@ -1081,9 +1108,8 @@ export const DETECTORS = [
|
|
|
1081
1108
|
{
|
|
1082
1109
|
type: 'straggler', scope: 'stage', order: 70, fixEffort: 'code', version: 1,
|
|
1083
1110
|
docAnchor: '#bottleneck-straggler',
|
|
1084
|
-
// floorPctWarn/floorPctCrit are re-exported as STRAGGLER_FLOOR_PCT_WARN/CRIT
|
|
1085
|
-
//
|
|
1086
|
-
// in sync, don't hand-edit one without the other.
|
|
1111
|
+
// floorPctWarn/floorPctCrit are re-exported as STRAGGLER_FLOOR_PCT_WARN/CRIT and reused as
|
|
1112
|
+
// impact-band.ts's global noise floor: keep the two in sync.
|
|
1087
1113
|
thresholds: { minTasks: 10, shareWarn: 0.05, warnPct: 0.10, critPct: 0.20, floorPctWarn: STRAGGLER_FLOOR_PCT_WARN, floorPctCrit: STRAGGLER_FLOOR_PCT_CRIT },
|
|
1088
1114
|
detect(
|
|
1089
1115
|
|
|
@@ -1097,12 +1123,9 @@ export const DETECTORS = [
|
|
|
1097
1123
|
if ((stage.speculativeTasks ?? 0) === 0 && stragglerShare <= this.thresholds.shareWarn) return null;
|
|
1098
1124
|
const useSpeculative = (stage.speculativeTasks ?? 0) > 0;
|
|
1099
1125
|
const speculativeShare = useSpeculative ? stage.speculativeTasks / stage.taskCount : 0;
|
|
1100
|
-
// Same absolute delta
|
|
1101
|
-
//
|
|
1102
|
-
//
|
|
1103
|
-
// duration models near-zero savings, so it must not outrank 'info'.
|
|
1104
|
-
// Clipped the same way before the floor check so the gate agrees with
|
|
1105
|
-
// what's actually displayed.
|
|
1126
|
+
// Same absolute delta impact-estimator.ts's straggler/stageShape case reports as savings: a
|
|
1127
|
+
// high straggler/speculative share on a stage whose tasks barely vary models near-zero
|
|
1128
|
+
// savings, so it must not outrank 'info'. Clipped the same way before the floor check.
|
|
1106
1129
|
const wasteMs = Math.max(0, stage.taskDurationMax - stage.taskDurationP50);
|
|
1107
1130
|
const appDurationMs = computeAppDurationMs(ctx);
|
|
1108
1131
|
const floorWasteMs = clippedWasteMs(wasteMs, stage.id, ctx);
|
|
@@ -1110,21 +1133,14 @@ export const DETECTORS = [
|
|
|
1110
1133
|
const meetsCritFloor = meetsRuntimeFloor(floorWasteMs, appDurationMs, this.thresholds.floorPctCrit);
|
|
1111
1134
|
const speculativeTier = speculativeShare >= this.thresholds.critPct && meetsCritFloor ? 'critical'
|
|
1112
1135
|
: speculativeShare >= this.thresholds.warnPct && meetsWarnFloor ? 'warning' : 'info';
|
|
1113
|
-
// Straggler share has no dedicated critical tier per
|
|
1114
|
-
// docs-site/contributor-guide/architecture/detector-contract.md; it can only push to warning.
|
|
1136
|
+
// Straggler share has no dedicated critical tier per detector-contract.md; only warning.
|
|
1115
1137
|
const stragglerTier = stragglerShare > this.thresholds.shareWarn && meetsWarnFloor ? 'warning' : 'info';
|
|
1116
|
-
// Fixed fallback: overwritten by deriveImpactBand
|
|
1117
|
-
//
|
|
1118
|
-
// surfaces on the rare miss (stage excluded from the occupancy sweep).
|
|
1138
|
+
// Fixed fallback: overwritten by deriveImpactBand when this finding gets a real wallClock
|
|
1139
|
+
// estimate (the common case). Only surfaces on the rare occupancy-sweep miss.
|
|
1119
1140
|
const impactBand = 'info';
|
|
1120
|
-
// Report whichever signal actually drove the finding, not just whether
|
|
1121
|
-
// speculative
|
|
1122
|
-
//
|
|
1123
|
-
// speculativeTasks count (real-log bug: a 50%-straggler-share stage
|
|
1124
|
-
// with 1 speculative task reported an impact band of 'warning' but metric
|
|
1125
|
-
// 'speculativeTasks: 1', hiding the actual cause). Ties keep the prior
|
|
1126
|
-
// default (speculative-driven) so existing speculative-only findings
|
|
1127
|
-
// are unaffected.
|
|
1141
|
+
// Report whichever signal actually drove the finding, not just whether speculation was on:
|
|
1142
|
+
// a high stragglerShare with few speculative retries must not be reported as a low-value
|
|
1143
|
+
// speculativeTasks count. Ties keep the speculative-driven default.
|
|
1128
1144
|
const useSpeculativeMetric = useSpeculative && !(IMPACT_BAND_ORDER[stragglerTier] < IMPACT_BAND_ORDER[speculativeTier]);
|
|
1129
1145
|
const value = useSpeculativeMetric ? stage.speculativeTasks : Math.round(stragglerShare * 100);
|
|
1130
1146
|
const detail = useSpeculativeMetric
|
|
@@ -1137,18 +1153,21 @@ export const DETECTORS = [
|
|
|
1137
1153
|
unit: useSpeculativeMetric ? 'count' : 'pct',
|
|
1138
1154
|
speculativeTasks: stage.speculativeTasks ?? 0,
|
|
1139
1155
|
stragglerCount: stage.stragglerCount ?? 0,
|
|
1140
|
-
|
|
1156
|
+
confidence: useSpeculativeMetric
|
|
1157
|
+
? stragglerConfidence(speculativeShare, this.thresholds.warnPct, this.thresholds.critPct)
|
|
1158
|
+
: stragglerConfidence(stragglerShare, this.thresholds.shareWarn, this.thresholds.critPct),
|
|
1159
|
+
validationRequired: 'This finding is gated by 0.5%/2% runtime-floor thresholds, our own noise floor for this metric.',
|
|
1160
|
+
recommendation: `${detail}: rule out a GC pause or a slow shuffle fetch before assuming a hardware issue; if a skewed key is the real cause, that's a candidate for AQE's skew-join handling.`,
|
|
1141
1161
|
};
|
|
1142
1162
|
},
|
|
1143
1163
|
},
|
|
1144
1164
|
{
|
|
1145
1165
|
type: 'speculationWaste', scope: 'stage', order: 71, fixEffort: 'config', version: 1,
|
|
1146
|
-
docAnchor: '#bottleneck-straggler',
|
|
1166
|
+
docAnchor: '#bottleneck-straggler',
|
|
1147
1167
|
thresholds: { minWasted: 5, minWasteMs: 60000 },
|
|
1148
1168
|
detect(
|
|
1149
1169
|
|
|
1150
1170
|
|
|
1151
|
-
|
|
1152
1171
|
|
|
1153
1172
|
stage ,
|
|
1154
1173
|
) {
|
|
@@ -1159,7 +1178,7 @@ export const DETECTORS = [
|
|
|
1159
1178
|
type: 'speculationWaste', stageId: stage.id,
|
|
1160
1179
|
impactBand: 'warning',
|
|
1161
1180
|
metric: 'speculationWasteMs', value: wastedMs,
|
|
1162
|
-
confidence: this.
|
|
1181
|
+
confidence: speculationWasteConfidence(wastedMs, this.thresholds.minWasteMs),
|
|
1163
1182
|
recommendation: `Speculative execution discarded ${Math.round(wastedMs / 1000)}s of executor time in this stage; if task durations are naturally variable rather than genuine stragglers, consider tuning spark.speculation.multiplier/quantile.`,
|
|
1164
1183
|
};
|
|
1165
1184
|
},
|
|
@@ -1179,6 +1198,9 @@ export const DETECTORS = [
|
|
|
1179
1198
|
type: 'retryWaste', stageId: stage.id,
|
|
1180
1199
|
impactBand: 'warning',
|
|
1181
1200
|
metric: 'retryWasteMs', value: wastedMs,
|
|
1201
|
+
numTasks: stage.taskCount,
|
|
1202
|
+
memoryBytesSpilled: stage.memoryBytesSpilled,
|
|
1203
|
+
retriedTaskDetails: stage.retryTaskSamples ?? [],
|
|
1182
1204
|
recommendation: `Retried task attempts wasted ${Math.round(wastedMs / 1000)}s of executor time (${wasted} attempt${wasted === 1 ? '' : 's'}) even though the stage completed: investigate executor loss or fetch failures.`,
|
|
1183
1205
|
extended: `${wasted} task attempts were superseded by a later retry, wasting ${Math.round(wastedMs / 1000)}s of executor time. Common causes: executor loss (OOM-kill, node death) or shuffle FetchFailed forcing a stage-map recompute. Check driver logs for the dominant reason (see the Failures widget) even if the final failure rate looks low; retries hide the true cost.`,
|
|
1184
1206
|
};
|
|
@@ -1206,14 +1228,13 @@ export const DETECTORS = [
|
|
|
1206
1228
|
},
|
|
1207
1229
|
},
|
|
1208
1230
|
{
|
|
1209
|
-
// No docAnchor
|
|
1210
|
-
//
|
|
1211
|
-
// section for this tool-specific "capture stopped early" signal.
|
|
1231
|
+
// No docAnchor: the upstream spark-tuning-reference docs have no section for this
|
|
1232
|
+
// tool-specific "capture stopped early" signal.
|
|
1212
1233
|
type: 'incompleteRun', scope: 'app', order: 5, fixEffort: 'code', version: 1,
|
|
1213
1234
|
thresholds: {},
|
|
1214
1235
|
recommendation: 'This event log never recorded an ApplicationEnd event: the capture stopped before the run finished (an in-flight job, a rotated-away log, or a cut-short capture). Findings and metrics elsewhere on this board reflect only what was captured up to that point, not the full run.',
|
|
1215
1236
|
detect( ctx ) {
|
|
1216
|
-
if (ctx.app.startTime == null || ctx.app.endTime != null) return null;
|
|
1237
|
+
if (!ctx.app || ctx.app.startTime == null || ctx.app.endTime != null) return null;
|
|
1217
1238
|
return {
|
|
1218
1239
|
type: 'incompleteRun', stageId: null, impactBand: 'warning',
|
|
1219
1240
|
metric: 'applicationEnd', value: 'missing',
|
|
@@ -1230,18 +1251,22 @@ export const DETECTORS = [
|
|
|
1230
1251
|
ctx ,
|
|
1231
1252
|
) {
|
|
1232
1253
|
const { app, stages } = ctx;
|
|
1233
|
-
|
|
1254
|
+
// Nullish (not falsy) check: a literal startTime:0 must not be treated as "missing".
|
|
1255
|
+
if (!app || app.startTime == null || stages.size === 0) return null;
|
|
1234
1256
|
let firstTaskLaunch = Infinity;
|
|
1235
1257
|
for (const stage of stages.values()) {
|
|
1236
1258
|
if (stage.submittedAt > 0 && stage.submittedAt < firstTaskLaunch) firstTaskLaunch = stage.submittedAt;
|
|
1237
1259
|
}
|
|
1260
|
+
// No stage ever recorded a submission timestamp: no basis to measure a startup gap against.
|
|
1261
|
+
// Exposed now that a literal app.startTime:0 no longer short-circuits this detector entirely.
|
|
1262
|
+
if (!Number.isFinite(firstTaskLaunch)) return null;
|
|
1238
1263
|
const gapSeconds = (firstTaskLaunch - app.startTime) / 1000;
|
|
1239
1264
|
if (gapSeconds <= this.thresholds.gapSeconds) return null;
|
|
1240
1265
|
const value = Math.round(gapSeconds);
|
|
1241
1266
|
return {
|
|
1242
1267
|
type: 'coldStart', stageId: null, impactBand: 'warning',
|
|
1243
1268
|
metric: 'startupGapSeconds', value,
|
|
1244
|
-
recommendation: `
|
|
1269
|
+
recommendation: `The first task waited ${value}s for executors to become available: keep a warm pool of idle executors, or if using dynamic allocation, raise the minimum/initial executor count so it doesn't scale up from zero.`,
|
|
1245
1270
|
};
|
|
1246
1271
|
},
|
|
1247
1272
|
},
|
|
@@ -1253,25 +1278,27 @@ export const DETECTORS = [
|
|
|
1253
1278
|
|
|
1254
1279
|
ctx ,
|
|
1255
1280
|
) {
|
|
1256
|
-
const { app, executorsAdded, executorsRemoved } = ctx;
|
|
1257
|
-
|
|
1281
|
+
const { app, executorsAdded, executorsRemoved, runAggregates } = ctx;
|
|
1282
|
+
// Nullish (not falsy) check: a literal startTime:0 must not be treated as "missing".
|
|
1283
|
+
if (!app || executorsAdded.length === 0 || app.startTime == null || app.endTime == null) return null;
|
|
1258
1284
|
const appDuration = app.endTime - app.startTime;
|
|
1259
1285
|
if (appDuration <= 0) return null;
|
|
1260
|
-
|
|
1261
|
-
|
|
1262
|
-
|
|
1263
|
-
|
|
1264
|
-
|
|
1265
|
-
|
|
1266
|
-
|
|
1267
|
-
|
|
1268
|
-
|
|
1269
|
-
|
|
1286
|
+
// computePeakConcurrentCores (not executorsAdded.length/computeTotalCores): real concurrent
|
|
1287
|
+
// capacity, not a cumulative sum that double-counts a churned-through executor against its
|
|
1288
|
+
// replacement's (spot preemption, dynamicAllocation replacement).
|
|
1289
|
+
const totalCores = computePeakConcurrentCores(app, executorsAdded, executorsRemoved);
|
|
1290
|
+
if (totalCores <= 0) return null;
|
|
1291
|
+
const capacityCoreMs = totalCores * appDuration;
|
|
1292
|
+
// Busy core-time (from the whole-run core-time-series, same signal memoryUtilization's
|
|
1293
|
+
// idleCores variant already uses), not executor lifetime: an executor that exists for the
|
|
1294
|
+
// whole run but sits fully idle must not score as 100% used. Missing runAggregates (older
|
|
1295
|
+
// callers, synthetic fixtures) reads as 0 busy time rather than falling back to the
|
|
1296
|
+
// lifetime-based measure this replaces.
|
|
1297
|
+
const busyCoreMs = runAggregates?.busyCoreMs ?? 0;
|
|
1298
|
+
const utilization = busyCoreMs / capacityCoreMs;
|
|
1270
1299
|
if (utilization >= this.thresholds.minUtil) return null;
|
|
1271
1300
|
|
|
1272
1301
|
// CPU-time-based utilization (sparkMeasure): metric only, no threshold.
|
|
1273
|
-
// Total cores: prefer executor-added Total Cores (real), else config cores.
|
|
1274
|
-
const totalCores = computeTotalCores(app, executorsAdded);
|
|
1275
1302
|
let cpuUtilizationPct = null;
|
|
1276
1303
|
if (totalCores > 0) {
|
|
1277
1304
|
let cpuMs = 0;
|
|
@@ -1296,10 +1323,10 @@ export const DETECTORS = [
|
|
|
1296
1323
|
type: 'memoryUtilization', scope: 'app', order: 102, fixEffort: 'config', version: 1,
|
|
1297
1324
|
docAnchor: '#bottleneck-memory-utilization',
|
|
1298
1325
|
thresholds: {
|
|
1299
|
-
idleCoreWarn: 0.50, //
|
|
1300
|
-
bandTooSmall: 0.95, //
|
|
1326
|
+
idleCoreWarn: 0.50, // WastedCoresAlertsReducer
|
|
1327
|
+
bandTooSmall: 0.95, // MemoryAlertsReducer: used/allocated
|
|
1301
1328
|
bandTooHigh: 0.70, // below this => over-provisioned (cost signal)
|
|
1302
|
-
wasteBufferMultiplier: 1.5, //
|
|
1329
|
+
wasteBufferMultiplier: 1.5, // UNVERIFIED
|
|
1303
1330
|
},
|
|
1304
1331
|
detect(
|
|
1305
1332
|
|
|
@@ -1309,19 +1336,20 @@ export const DETECTORS = [
|
|
|
1309
1336
|
|
|
1310
1337
|
ctx ,
|
|
1311
1338
|
) {
|
|
1312
|
-
const { app, executorsAdded, runAggregates, stages } = ctx;
|
|
1339
|
+
const { app, executorsAdded, executorsRemoved, runAggregates, stages } = ctx;
|
|
1313
1340
|
const out = [];
|
|
1314
|
-
// Nullish (not falsy) check:
|
|
1315
|
-
// detectors, a literal startTime:0 must not be treated as "missing".
|
|
1341
|
+
// Nullish (not falsy) check: a literal startTime:0 must not be treated as "missing".
|
|
1316
1342
|
if (app?.startTime == null || app?.endTime == null) return out;
|
|
1317
1343
|
const appDurationMs = app.endTime - app.startTime;
|
|
1318
1344
|
if (appDurationMs <= 0) return out;
|
|
1319
1345
|
|
|
1320
|
-
//
|
|
1321
|
-
|
|
1322
|
-
|
|
1323
|
-
|
|
1324
|
-
|
|
1346
|
+
// Peak concurrent executors/cores (not executorsAdded.length/computeTotalCores): a
|
|
1347
|
+
// cumulative sum or count double-counts a churned-through executor against its replacement's
|
|
1348
|
+
// (spot preemption, dynamicAllocation replacement), inflating idle-rate and waste-model figures.
|
|
1349
|
+
const peakExecutors = computePeakConcurrentExecutorCount(executorsAdded, executorsRemoved);
|
|
1350
|
+
const totalCores = computePeakConcurrentCores(app, executorsAdded, executorsRemoved);
|
|
1351
|
+
// Hoisted above 1a (also 1b/1c's input) so the idle-cores finding carries the allocated
|
|
1352
|
+
// memory its MB-seconds estimate needs.
|
|
1325
1353
|
const allocatedMB = app.resources?.executor?.memoryMB ?? null;
|
|
1326
1354
|
|
|
1327
1355
|
// ── 1a idle-cores rate ────────────────────────────────────────────────
|
|
@@ -1333,8 +1361,7 @@ export const DETECTORS = [
|
|
|
1333
1361
|
out.push({
|
|
1334
1362
|
type: 'memoryUtilization', variant: 'idleCores', stageId: null,
|
|
1335
1363
|
impactBand: 'warning', metric: 'idleCoreRate', value,
|
|
1336
|
-
// Raw (unrounded) rate plus
|
|
1337
|
-
// wasted-MB-seconds model: `value` above is a rounded percentage.
|
|
1364
|
+
// Raw (unrounded) rate plus sizing inputs for the impact estimator: `value` is rounded pct.
|
|
1338
1365
|
idleRateFraction: idleRate, allocatedMB, peakExecutors, appDurationMs,
|
|
1339
1366
|
recommendation: `${value}% of allocated core-time ran no task: reduce cluster size or enable dynamic allocation.`,
|
|
1340
1367
|
});
|
|
@@ -1362,10 +1389,8 @@ export const DETECTORS = [
|
|
|
1362
1389
|
const allocatedBytes = allocatedMB * 1024 * 1024;
|
|
1363
1390
|
for (const [execId, heap] of peakHeapByExec) {
|
|
1364
1391
|
const ratio = heap / allocatedBytes;
|
|
1365
|
-
// The two bands are opposite signals
|
|
1366
|
-
//
|
|
1367
|
-
// OOM-risk case from the over-provisioning waste case without re-deriving
|
|
1368
|
-
// the ratio against the thresholds.
|
|
1392
|
+
// The two bands are opposite signals: an explicit `rule` discriminator lets consumers
|
|
1393
|
+
// tell OOM-risk from over-provisioning without re-deriving the ratio.
|
|
1369
1394
|
if (ratio > this.thresholds.bandTooSmall) {
|
|
1370
1395
|
out.push({
|
|
1371
1396
|
type: 'memoryUtilization', variant: 'memoryBand', rule: 'heapNearCapacity',
|
|
@@ -1387,7 +1412,7 @@ export const DETECTORS = [
|
|
|
1387
1412
|
}
|
|
1388
1413
|
}
|
|
1389
1414
|
|
|
1390
|
-
// ── 1c Spark Memory Limit waste model (
|
|
1415
|
+
// ── 1c Spark Memory Limit waste model (UNVERIFIED buffer) ─
|
|
1391
1416
|
if (allocatedMB != null && peakExecutors > 0) {
|
|
1392
1417
|
const allocatedMBSeconds = peakExecutors * allocatedMB * (appDurationMs / 1000);
|
|
1393
1418
|
let usedRunTimeMs = 0;
|
|
@@ -1399,8 +1424,8 @@ export const DETECTORS = [
|
|
|
1399
1424
|
out.push({
|
|
1400
1425
|
type: 'memoryUtilization', variant: 'wasteModel', stageId: null,
|
|
1401
1426
|
impactBand: 'info', metric: 'wastedMBSeconds', value,
|
|
1402
|
-
confidence:
|
|
1403
|
-
validationRequired: 'Memory-waste estimate uses allocated-vs-used memory-time and
|
|
1427
|
+
confidence: memoryWasteConfidence(wastedMBSeconds, usedMBSeconds, this.thresholds.wasteBufferMultiplier),
|
|
1428
|
+
validationRequired: 'Memory-waste estimate uses allocated-vs-used memory-time and a 1.5x buffer: confirm against the Spark UI before acting.',
|
|
1404
1429
|
recommendation: `Allocated executor memory sat largely idle over the run (~${value.toLocaleString('en-US')} MB-seconds wasted): review spark.executor.memory and executor count.`,
|
|
1405
1430
|
});
|
|
1406
1431
|
}
|
|
@@ -1410,13 +1435,9 @@ export const DETECTORS = [
|
|
|
1410
1435
|
},
|
|
1411
1436
|
},
|
|
1412
1437
|
{
|
|
1413
|
-
// Per-RDD cache-utilization proxies (this repo's own design: Spark
|
|
1414
|
-
//
|
|
1415
|
-
//
|
|
1416
|
-
// #85). Two independent, per-RDD tiered checks over `ctx.app.rddInfo`
|
|
1417
|
-
// storage snapshots: partial caching (numCachedPartitions < numPartitions)
|
|
1418
|
-
// and disk spillover (diskSize share of a MEMORY_AND_DISK*-requesting
|
|
1419
|
-
// RDD's cached footprint). An RDD can produce both findings in one pass.
|
|
1438
|
+
// Per-RDD cache-utilization proxies (this repo's own design: Spark event logs carry no
|
|
1439
|
+
// block-access events, so a literal cache hit rate isn't derivable). Two per-RDD tiered
|
|
1440
|
+
// checks over rddInfo snapshots: partial caching and disk spillover. An RDD can produce both.
|
|
1420
1441
|
type: 'cacheUtilization', scope: 'app', order: 103, fixEffort: 'code', version: 1,
|
|
1421
1442
|
docAnchor: '#memory-model',
|
|
1422
1443
|
thresholds: {
|
|
@@ -1458,11 +1479,9 @@ export const DETECTORS = [
|
|
|
1458
1479
|
},
|
|
1459
1480
|
},
|
|
1460
1481
|
{
|
|
1461
|
-
// Non-local task ratio across stage.localityStats (RACK_LOCAL + ANY vs
|
|
1462
|
-
//
|
|
1463
|
-
// the
|
|
1464
|
-
// variant above. NO_PREF stays in the denominator only: it's what
|
|
1465
|
-
// shuffle-read stages legitimately report with no locality problem.
|
|
1482
|
+
// Non-local task ratio across stage.localityStats (RACK_LOCAL + ANY vs all tasks), the other
|
|
1483
|
+
// half of the "Wasted Cores Ratio" (idle-core half is memoryUtilization's idleCores).
|
|
1484
|
+
// NO_PREF stays in the denominator only: shuffle-read stages legitimately report it.
|
|
1466
1485
|
type: 'coreLocality', scope: 'app', order: 103, fixEffort: 'config', version: 1,
|
|
1467
1486
|
docAnchor: '#bottleneck-utilization',
|
|
1468
1487
|
thresholds: { minTasks: 50, warnRatio: 0.15, critRatio: 0.35 },
|
|
@@ -1472,11 +1491,8 @@ export const DETECTORS = [
|
|
|
1472
1491
|
) {
|
|
1473
1492
|
const { totalTasks, nonLocalTasks, ratio } = computeCoreLocalityRatio([...ctx.stages.values()]);
|
|
1474
1493
|
if (totalTasks == null || totalTasks < this.thresholds.minTasks) return null;
|
|
1475
|
-
// computeCoreLocalityRatio
|
|
1476
|
-
//
|
|
1477
|
-
// both at once); the `totalTasks == null` guard above already rules that
|
|
1478
|
-
// out, so `ratio` is guaranteed non-null here even though the function's
|
|
1479
|
-
// declared return type keeps the two nullable independently.
|
|
1494
|
+
// computeCoreLocalityRatio only returns ratio:null together with totalTasks:null (shared
|
|
1495
|
+
// EMPTY sentinel); the totalTasks guard above rules that out, so ratio is non-null here.
|
|
1480
1496
|
if (ratio < this.thresholds.warnRatio) return null;
|
|
1481
1497
|
|
|
1482
1498
|
const value = Math.round(ratio * 100);
|
|
@@ -1484,33 +1500,28 @@ export const DETECTORS = [
|
|
|
1484
1500
|
type: 'coreLocality', stageId: null,
|
|
1485
1501
|
impactBand: ratio >= this.thresholds.critRatio ? 'critical' : 'warning',
|
|
1486
1502
|
metric: 'nonLocalRatio', value,
|
|
1487
|
-
// Raw count behind the ratio, for the impact estimator
|
|
1488
|
-
// figure. Non-null whenever totalTasks is (both come from the same EMPTY sentinel).
|
|
1503
|
+
// Raw count behind the ratio, for the impact estimator. Non-null whenever totalTasks is.
|
|
1489
1504
|
nonLocalTaskCount: nonLocalTasks ,
|
|
1490
|
-
confidence:
|
|
1491
|
-
validationRequired: '
|
|
1505
|
+
confidence: coreLocalityConfidence(ratio , totalTasks, this.thresholds),
|
|
1506
|
+
validationRequired: 'This finding is gated by 15%/35% non-local-ratio thresholds (and a 50-task minimum), our own noise floor for this metric.',
|
|
1492
1507
|
recommendation: `${value}% of tasks (${nonLocalTasks }) ran without process- or node-local data placement: check spark.locality.wait settings and executor/data colocation.`,
|
|
1493
1508
|
};
|
|
1494
1509
|
},
|
|
1495
1510
|
},
|
|
1496
1511
|
{
|
|
1497
|
-
// Short-lived executors:
|
|
1498
|
-
//
|
|
1499
|
-
//
|
|
1500
|
-
// `utilization` above (no new data extraction), but measures lifetime
|
|
1501
|
-
// against a threshold instead of aggregate active-time.
|
|
1512
|
+
// Short-lived executors: stood up and torn down before doing useful work (wasteful
|
|
1513
|
+
// re-provisioning, not normal scale-down). Reuses utilization's add/remove matching, but
|
|
1514
|
+
// measures lifetime against a threshold instead of aggregate active-time.
|
|
1502
1515
|
type: 'autoscalingChurn', scope: 'app', order: 103, fixEffort: 'config', version: 1,
|
|
1503
|
-
confidence: 'low',
|
|
1504
1516
|
thresholds: { shortLivedMs: 120_000, warningPct: 0.30, criticalPct: 0.60, minExecutors: 5 },
|
|
1505
1517
|
detect(
|
|
1506
1518
|
|
|
1507
1519
|
|
|
1508
|
-
|
|
1509
1520
|
|
|
1510
1521
|
ctx ,
|
|
1511
1522
|
) {
|
|
1512
1523
|
const { app, executorsAdded, executorsRemoved } = ctx;
|
|
1513
|
-
if (executorsAdded.length === 0 || app.endTime == null) return null;
|
|
1524
|
+
if (!app || executorsAdded.length === 0 || app.endTime == null) return null;
|
|
1514
1525
|
if (executorsAdded.length < this.thresholds.minExecutors) return null;
|
|
1515
1526
|
|
|
1516
1527
|
const removedAt = new Map ();
|
|
@@ -1534,16 +1545,14 @@ export const DETECTORS = [
|
|
|
1534
1545
|
metric: 'shortLivedExecutorPct', value: pct,
|
|
1535
1546
|
// Raw count behind the percentage, for the impact estimator's startup-overhead figure.
|
|
1536
1547
|
shortLivedExecutorCount: shortLivedCount,
|
|
1537
|
-
confidence: this.
|
|
1548
|
+
confidence: autoscalingChurnConfidence(shortLivedPct, this.thresholds.warningPct, this.thresholds.criticalPct),
|
|
1538
1549
|
recommendation: `${pct}% of executors ran for under 2 minutes before being removed. This looks like wasteful re-provisioning rather than normal scale-down; consider raising spark.dynamicAllocation.executorIdleTimeout or widening the minExecutors/maxExecutors bounds to reduce flapping.`,
|
|
1539
1550
|
};
|
|
1540
1551
|
},
|
|
1541
1552
|
},
|
|
1542
1553
|
{
|
|
1543
|
-
// Cross-execution relation reuse
|
|
1544
|
-
//
|
|
1545
|
-
// SQL workloads; see docs/adr/0009-caching-opportunity-relation-reuse.md).
|
|
1546
|
-
// Flags an input relation scanned by two or more SQL executions in one run.
|
|
1554
|
+
// Cross-execution relation reuse: flags an input relation scanned by two or more SQL
|
|
1555
|
+
// executions in one run, firing on real relation names (parquet:..., jdbc:...).
|
|
1547
1556
|
type: 'cachingOpportunity', scope: 'app', order: 105, fixEffort: 'code', version: 1,
|
|
1548
1557
|
docAnchor: '#bottleneck-utilization',
|
|
1549
1558
|
thresholds: { minExecutions: 2 },
|
|
@@ -1566,12 +1575,10 @@ export const DETECTORS = [
|
|
|
1566
1575
|
|
|
1567
1576
|
|
|
1568
1577
|
|
|
1569
|
-
// relationId -> { format, relation, executionIds:Set, executionBytes: Map<execId, bytes> }
|
|
1570
1578
|
const byRelation = new Map ();
|
|
1571
1579
|
for (const exec of sql.values()) {
|
|
1572
1580
|
if (!exec.planTree) continue;
|
|
1573
|
-
// Dedupe relations within one execution (self-joins count once), summing
|
|
1574
|
-
// this execution's read bytes per relation across its scan nodes.
|
|
1581
|
+
// Dedupe relations within one execution (self-joins count once), summing read bytes per relation.
|
|
1575
1582
|
const perExec = new Map ();
|
|
1576
1583
|
walkPlanTree(exec.planTree, (node) => {
|
|
1577
1584
|
const rid = scanRelationId(node.name ?? '', node.detail ?? '');
|
|
@@ -1591,15 +1598,13 @@ export const DETECTORS = [
|
|
|
1591
1598
|
}
|
|
1592
1599
|
}
|
|
1593
1600
|
|
|
1594
|
-
// fingerprint -> { operator, exampleNode, executionIds:Set, executionBytes:Map<execId,bytes>, ancestorFingerprints:Set<fingerprint>, leafRelationRids:Set<rid> }
|
|
1595
1601
|
const byComposite = new Map ();
|
|
1596
1602
|
for (const exec of sql.values()) {
|
|
1597
1603
|
if (!exec.planTree) continue;
|
|
1598
1604
|
const candidates = findCompositeCandidates(exec.planTree);
|
|
1599
1605
|
const fingerprintByNode = new Map (candidates.map((c) => [c.node, c.fingerprint]));
|
|
1600
1606
|
|
|
1601
|
-
// Dedupe identical fingerprints within this execution (repeated
|
|
1602
|
-
// identical composite counts once, mirroring the leaf perExec dedupe).
|
|
1607
|
+
// Dedupe identical fingerprints within this execution (repeated composite counts once).
|
|
1603
1608
|
const perExecComposite = new Map ();
|
|
1604
1609
|
for (const c of candidates) {
|
|
1605
1610
|
let agg = perExecComposite.get(c.fingerprint);
|
|
@@ -1636,12 +1641,10 @@ export const DETECTORS = [
|
|
|
1636
1641
|
}
|
|
1637
1642
|
}
|
|
1638
1643
|
|
|
1639
|
-
// Qualifying =
|
|
1640
|
-
//
|
|
1641
|
-
// subsumed: fully (equal execution sets) or partially (residual).
|
|
1644
|
+
// Qualifying = enough distinct executions on its own. Nested-dedupe: a qualifying composite
|
|
1645
|
+
// with a qualifying ANCESTOR is subsumed, fully (equal sets) or partially (residual).
|
|
1642
1646
|
const isQualifying = (fp ) =>
|
|
1643
1647
|
byComposite.has(fp) && byComposite.get(fp) .executionIds.size >= this.thresholds.minExecutions;
|
|
1644
|
-
// fingerprint -> { finalExecutionIds:Set, suppressed:boolean }
|
|
1645
1648
|
const compositeResolutions = new Map ();
|
|
1646
1649
|
for (const [fingerprint, agg] of byComposite) {
|
|
1647
1650
|
if (!isQualifying(fingerprint)) { compositeResolutions.set(fingerprint, { finalExecutionIds: agg.executionIds, suppressed: true }); continue; }
|
|
@@ -1658,7 +1661,7 @@ export const DETECTORS = [
|
|
|
1658
1661
|
const compositeVerb = { join: ['joined', 'join'], union: ['unioned', 'union'] };
|
|
1659
1662
|
|
|
1660
1663
|
const out = [];
|
|
1661
|
-
// rid -> Set<execId> covered by an emitted composite
|
|
1664
|
+
// rid -> Set<execId> covered by an emitted composite, for leaf suppression.
|
|
1662
1665
|
const coveredExecutionsByRid = new Map ();
|
|
1663
1666
|
for (const [fingerprint, agg] of byComposite) {
|
|
1664
1667
|
const resolution = compositeResolutions.get(fingerprint) ;
|
|
@@ -1683,7 +1686,7 @@ export const DETECTORS = [
|
|
|
1683
1686
|
metric: 'executionReuse', value,
|
|
1684
1687
|
format: 'derived', relations, operator: agg.operator, relation: relationDisplay,
|
|
1685
1688
|
executionIds: finalExecutionIds, totalReadBytes,
|
|
1686
|
-
confidence:
|
|
1689
|
+
confidence: cachingReuseConfidence(value, this.thresholds.minExecutions),
|
|
1687
1690
|
validationRequired:
|
|
1688
1691
|
'Composite reuse is inferred from a structural plan-shape match (operator + normalized ' +
|
|
1689
1692
|
'join/filter condition + child shapes) across SQL executions; confirm these executions ' +
|
|
@@ -1714,7 +1717,7 @@ export const DETECTORS = [
|
|
|
1714
1717
|
relation: agg.relation, format: agg.format,
|
|
1715
1718
|
executionIds: residualExecutionIds.sort((a, b) => a - b),
|
|
1716
1719
|
totalReadBytes,
|
|
1717
|
-
confidence:
|
|
1720
|
+
confidence: cachingReuseConfidence(value, this.thresholds.minExecutions),
|
|
1718
1721
|
validationRequired:
|
|
1719
1722
|
'Relation-reuse is inferred from the pre-AQE plan scan identity across SQL ' +
|
|
1720
1723
|
'executions; confirm the reads are the same data and cacheable within one ' +
|
|
@@ -1744,10 +1747,8 @@ export const DETECTORS = [
|
|
|
1744
1747
|
let totalTasks = 0, failedTasks = 0;
|
|
1745
1748
|
for (const s of stages.values()) { totalTasks += s.taskCount ?? 0; failedTasks += s.failedTasks ?? 0; }
|
|
1746
1749
|
const taskFailureRate = totalTasks > 0 ? failedTasks / totalTasks : 0;
|
|
1747
|
-
// Average wall-clock
|
|
1748
|
-
//
|
|
1749
|
-
// missing either timestamp are excluded rather than counted as zero-length;
|
|
1750
|
-
// with no timed failed job at all the average is 0 (never NaN).
|
|
1750
|
+
// Average wall-clock of failed jobs, for the impact estimator's cost-only figure. Jobs
|
|
1751
|
+
// missing either timestamp are excluded (not counted as zero); with none timed the average is 0.
|
|
1751
1752
|
const timedFailedJobs = failedJobList.filter(j => j.submissionTime != null && j.completionTime != null);
|
|
1752
1753
|
const avgJobDurationMs = timedFailedJobs.length > 0
|
|
1753
1754
|
? timedFailedJobs.reduce((s, j) => s + (j.completionTime - j.submissionTime ), 0) / timedFailedJobs.length
|
|
@@ -1858,15 +1859,17 @@ export const DETECTORS = [
|
|
|
1858
1859
|
const nodes = [];
|
|
1859
1860
|
for (const n of g.nodes) walkPlanTree(n, (node) => nodes.push(node));
|
|
1860
1861
|
const stageIds = unionStageIds(nodes, fallbackStageIds);
|
|
1862
|
+
// resolvePlanTree always sets id; safe downstream of it.
|
|
1863
|
+
const planNodeIds = nodes.map((n) => n.id ).filter(Boolean);
|
|
1861
1864
|
const touching = g.sampleRelation ? ` (touching ${g.sampleRelation})` : '';
|
|
1862
1865
|
return {
|
|
1863
|
-
type: 'duplicatePlanSubtree', executionId: sqlExec.id, stageIds,
|
|
1864
|
-
// Fixed fallback: overwritten by deriveImpactBand
|
|
1865
|
-
//
|
|
1866
|
-
// surfaces on the rare miss (stage excluded from the occupancy sweep).
|
|
1866
|
+
type: 'duplicatePlanSubtree', executionId: sqlExec.id, stageIds, planNodeIds,
|
|
1867
|
+
// Fixed fallback: overwritten by deriveImpactBand when this finding gets a real
|
|
1868
|
+
// wallClock estimate (the common case). Only surfaces on the rare occupancy-sweep miss.
|
|
1867
1869
|
impactBand: 'warning', metric: 'subtreeOccurrences', value: g.occurrences,
|
|
1868
1870
|
rootName: g.rootName, subtreeSize: g.subtreeSize, sampleRelation: g.sampleRelation,
|
|
1869
|
-
groupIndex: g.groupIndex,
|
|
1871
|
+
groupIndex: g.groupIndex,
|
|
1872
|
+
confidence: duplicateSubtreeConfidence(g.subtreeSize, g.occurrences, this.thresholds),
|
|
1870
1873
|
validationRequired: 'Duplicate-subtree matching compares operator names and metric names only, not literal values or expr IDs: confirm the repeated work is real in the Spark SQL plan tab before acting.',
|
|
1871
1874
|
recommendation: g.isExchangeRoot
|
|
1872
1875
|
? `A ${g.subtreeSize}-node subtree rooted at ${pathBasename(g.rootName)} repeats ${g.occurrences}x in this plan${touching}: this looks like a possible missed exchange reuse; check whether the same shuffle could be computed once and reused.`
|
|
@@ -1908,8 +1911,9 @@ export const DETECTORS = [
|
|
|
1908
1911
|
const fallbackStageIds = stageIdsForSqlExec(sqlExec.id, ctx.stages);
|
|
1909
1912
|
return hits.map(h => {
|
|
1910
1913
|
const stageIds = unionStageIds([h.node], fallbackStageIds);
|
|
1914
|
+
const planNodeIds = h.node.id ? [h.node.id] : [];
|
|
1911
1915
|
return {
|
|
1912
|
-
type: 'smallFiles', executionId: sqlExec.id, stageIds,
|
|
1916
|
+
type: 'smallFiles', executionId: sqlExec.id, stageIds, planNodeIds,
|
|
1913
1917
|
impactBand: 'warning',
|
|
1914
1918
|
metric: 'avgFileSizeBytes', value: Math.round(h.avgBytes),
|
|
1915
1919
|
fileCount: h.fileCount, direction: h.direction, nodeName: h.nodeName,
|
|
@@ -1921,10 +1925,9 @@ export const DETECTORS = [
|
|
|
1921
1925
|
},
|
|
1922
1926
|
},
|
|
1923
1927
|
{
|
|
1924
|
-
// Entry-level type is an identifier only; it never appears on an
|
|
1925
|
-
//
|
|
1926
|
-
//
|
|
1927
|
-
// direction rules (dataflint JoinToBroadcastAlert / BroadcastTooLargeAlert).
|
|
1928
|
+
// Entry-level type is an identifier only; it never appears on an emitted finding. Findings
|
|
1929
|
+
// carry 'underBroadcast'/'overBroadcast' since one shared plan-walk covers both
|
|
1930
|
+
// opposite-direction rules (JoinToBroadcastAlert / BroadcastTooLargeAlert).
|
|
1928
1931
|
type: 'broadcastSizing', scope: 'sql', order: 132, fixEffort: 'config', version: 2,
|
|
1929
1932
|
docAnchor: '#bottleneck-broadcast-sizing',
|
|
1930
1933
|
thresholds: {
|
|
@@ -1959,6 +1962,8 @@ export const DETECTORS = [
|
|
|
1959
1962
|
const contributors = [...boundarySizeContributors(childA), ...boundarySizeContributors(childB)];
|
|
1960
1963
|
out.push({
|
|
1961
1964
|
type: 'underBroadcast', executionId: sqlExec.id, stageIds: unionStageIds(contributors, fallbackStageIds),
|
|
1965
|
+
// resolvePlanTree always sets id; safe downstream of it.
|
|
1966
|
+
planNodeIds: contributors.map((n) => n.id ).filter(Boolean),
|
|
1962
1967
|
impactBand: 'info', metric: 'smallerSideBytes', value: smaller,
|
|
1963
1968
|
largerSideBytes: larger,
|
|
1964
1969
|
recommendation: `The smaller input to this Sort Merge Join (${formatBytes(smaller)}) is well under the broadcast threshold relative to the larger side (${formatBytes(larger)}): this could have been a broadcast join. Consider a broadcast() hint or raising spark.sql.autoBroadcastJoinThreshold.`,
|
|
@@ -1966,17 +1971,18 @@ export const DETECTORS = [
|
|
|
1966
1971
|
}
|
|
1967
1972
|
}
|
|
1968
1973
|
}
|
|
1969
|
-
if (node.name
|
|
1974
|
+
if (isBroadcastExchangeNode(node.name)) {
|
|
1970
1975
|
const m = (node.metrics ?? []).find(x => x.name === 'data size');
|
|
1971
1976
|
if (m && m.value > overBroadcastBytes) {
|
|
1972
|
-
//
|
|
1973
|
-
//
|
|
1974
|
-
//
|
|
1975
|
-
// executor-side metrics, so only the child is unioned in.
|
|
1977
|
+
// BroadcastExchange's own metrics are driver-computed and never on any TaskEnd
|
|
1978
|
+
// (node.stageIds always empty in real data); its child carries the executor-side
|
|
1979
|
+
// metrics, so only the child unions in.
|
|
1976
1980
|
const child = (node.children ?? [])[0];
|
|
1977
1981
|
out.push({
|
|
1978
1982
|
type: 'overBroadcast', executionId: sqlExec.id,
|
|
1979
1983
|
stageIds: unionStageIds(child ? [child] : [], fallbackStageIds),
|
|
1984
|
+
// resolvePlanTree always sets id; safe downstream of it.
|
|
1985
|
+
planNodeIds: [node.id ].filter(Boolean),
|
|
1980
1986
|
impactBand: 'warning', metric: 'broadcastBytes', value: m.value,
|
|
1981
1987
|
recommendation: `This broadcast (${formatBytes(m.value)}) exceeds the 1 GB threshold: check for a misapplied broadcast hint or a misconfigured spark.sql.autoBroadcastJoinThreshold.`,
|
|
1982
1988
|
});
|