sparkforensics-cli 0.1.0 → 0.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +6 -0
- package/bin/sparkforensics-analyze.mjs +113 -48
- package/export-template/docs/404.html +25 -0
- package/export-template/docs/assets/app.CndaAS6v.js +1 -0
- package/export-template/docs/assets/aqe-loop.IwQSATHw.svg +1 -0
- package/export-template/docs/assets/aqe-loop.dark.DGbaxqJE.svg +1 -0
- package/export-template/docs/assets/broadcast-vs-shuffle.Db4WY1XK.svg +1 -0
- package/export-template/docs/assets/broadcast-vs-shuffle.dark.C7Bxs0mG.svg +1 -0
- package/export-template/docs/assets/cache-lifecycle.dark.B-hS7AgU.svg +1 -0
- package/export-template/docs/assets/cache-lifecycle.rEOVYQNU.svg +1 -0
- package/export-template/docs/assets/chunks/@localSearchIndexroot.DNY8bVcl.js +1 -0
- package/export-template/docs/assets/chunks/VPLocalSearchBox.yJbZbsEo.js +9 -0
- package/export-template/docs/assets/chunks/duplicate-plan-subtree.dark.Cdp70QhV.js +1 -0
- package/export-template/docs/assets/chunks/framework.DSg0KOwT.js +20 -0
- package/export-template/docs/assets/chunks/retry-escalation-ladder.dark.DHipdJgZ.js +1 -0
- package/export-template/docs/assets/chunks/theme.Df2VAG9w.js +2 -0
- package/export-template/docs/assets/cold-start-timeline.DxC_Sc7w.svg +1 -0
- package/export-template/docs/assets/cold-start-timeline.dark.CZ17YcAG.svg +1 -0
- package/export-template/docs/assets/columnar-layout.PghGeOEA.svg +1 -0
- package/export-template/docs/assets/columnar-layout.dark.BVNlz0ff.svg +1 -0
- package/export-template/docs/assets/container-memory.DIO0AnIm.svg +1 -0
- package/export-template/docs/assets/container-memory.dark.CP-5zuCl.svg +1 -0
- package/export-template/docs/assets/contributor-guide_architecture_board-widgets.md.B-OsL91z.js +1 -0
- package/export-template/docs/assets/contributor-guide_architecture_board-widgets.md.B-OsL91z.lean.js +1 -0
- package/export-template/docs/assets/contributor-guide_architecture_detector-contract.md.BOeH4d1J.js +1 -0
- package/export-template/docs/assets/contributor-guide_architecture_detector-contract.md.BOeH4d1J.lean.js +1 -0
- package/export-template/docs/assets/contributor-guide_architecture_drill-down.md.BtPdlM7r.js +1 -0
- package/export-template/docs/assets/contributor-guide_architecture_drill-down.md.BtPdlM7r.lean.js +1 -0
- package/export-template/docs/assets/contributor-guide_architecture_impact-estimation.md.DYCDPgkh.js +1 -0
- package/export-template/docs/assets/contributor-guide_architecture_impact-estimation.md.DYCDPgkh.lean.js +1 -0
- package/export-template/docs/assets/contributor-guide_architecture_index.md.3TO9ic6w.js +1 -0
- package/export-template/docs/assets/contributor-guide_architecture_index.md.3TO9ic6w.lean.js +1 -0
- package/export-template/docs/assets/contributor-guide_architecture_overview.md.CehiRmGn.js +1 -0
- package/export-template/docs/assets/contributor-guide_architecture_overview.md.CehiRmGn.lean.js +1 -0
- package/export-template/docs/assets/contributor-guide_architecture_state-and-history.md.m3S3UdMk.js +1 -0
- package/export-template/docs/assets/contributor-guide_architecture_state-and-history.md.m3S3UdMk.lean.js +1 -0
- package/export-template/docs/assets/contributor-guide_architecture_widget-rendering.md.DbqPf2OT.js +1 -0
- package/export-template/docs/assets/contributor-guide_architecture_widget-rendering.md.DbqPf2OT.lean.js +1 -0
- package/export-template/docs/assets/contributor-guide_architecture_worker-protocol.md.B93qJ_tT.js +6 -0
- package/export-template/docs/assets/contributor-guide_architecture_worker-protocol.md.B93qJ_tT.lean.js +1 -0
- package/export-template/docs/assets/contributor-guide_contributing.md.CvRsdr6J.js +1 -0
- package/export-template/docs/assets/contributor-guide_contributing.md.CvRsdr6J.lean.js +1 -0
- package/export-template/docs/assets/contributor-guide_development-setup.md.DvAN_9mK.js +12 -0
- package/export-template/docs/assets/contributor-guide_development-setup.md.DvAN_9mK.lean.js +1 -0
- package/export-template/docs/assets/contributor-guide_testing.md.6rIKqSyY.js +1 -0
- package/export-template/docs/assets/contributor-guide_testing.md.6rIKqSyY.lean.js +1 -0
- package/export-template/docs/assets/dag-stages.DSz_S937.svg +1 -0
- package/export-template/docs/assets/dag-stages.dark.F72UzxH4.svg +1 -0
- package/export-template/docs/assets/driver-executor.D5pQ7YN1.svg +1 -0
- package/export-template/docs/assets/driver-executor.dark.BmX9cPvh.svg +1 -0
- package/export-template/docs/assets/duplicate-plan-subtree.B4cvN6fj.svg +1 -0
- package/export-template/docs/assets/duplicate-plan-subtree.dark.Dw8wS0Ag.svg +1 -0
- package/export-template/docs/assets/index.md.CHJVslga.js +1 -0
- package/export-template/docs/assets/index.md.CHJVslga.lean.js +1 -0
- package/export-template/docs/assets/inter-italic-cyrillic-ext.r48I6akx.woff2 +0 -0
- package/export-template/docs/assets/inter-italic-cyrillic.By2_1cv3.woff2 +0 -0
- package/export-template/docs/assets/inter-italic-greek-ext.1u6EdAuj.woff2 +0 -0
- package/export-template/docs/assets/inter-italic-greek.DJ8dCoTZ.woff2 +0 -0
- package/export-template/docs/assets/inter-italic-latin-ext.CN1xVJS-.woff2 +0 -0
- package/export-template/docs/assets/inter-italic-latin.C2AdPX0b.woff2 +0 -0
- package/export-template/docs/assets/inter-italic-vietnamese.BSbpV94h.woff2 +0 -0
- package/export-template/docs/assets/inter-roman-cyrillic-ext.BBPuwvHQ.woff2 +0 -0
- package/export-template/docs/assets/inter-roman-cyrillic.C5lxZ8CY.woff2 +0 -0
- package/export-template/docs/assets/inter-roman-greek-ext.CqjqNYQ-.woff2 +0 -0
- package/export-template/docs/assets/inter-roman-greek.BBVDIX6e.woff2 +0 -0
- package/export-template/docs/assets/inter-roman-latin-ext.4ZJIpNVo.woff2 +0 -0
- package/export-template/docs/assets/inter-roman-latin.Di8DUHzh.woff2 +0 -0
- package/export-template/docs/assets/inter-roman-vietnamese.BjW4sHH5.woff2 +0 -0
- package/export-template/docs/assets/join-strategy.C_FvrCEo.svg +1 -0
- package/export-template/docs/assets/join-strategy.dark.ChMLnNII.svg +1 -0
- package/export-template/docs/assets/memory-borrowing.BqQRJg0u.svg +1 -0
- package/export-template/docs/assets/memory-borrowing.dark.Yhh20O9C.svg +1 -0
- package/export-template/docs/assets/memory-regions.XHvO7jHG.svg +1 -0
- package/export-template/docs/assets/memory-regions.dark.D4TP9_08.svg +1 -0
- package/export-template/docs/assets/repartition-vs-coalesce.BovLRrpj.svg +1 -0
- package/export-template/docs/assets/repartition-vs-coalesce.dark.BhAczKZQ.svg +1 -0
- package/export-template/docs/assets/retry-escalation-ladder.DyTKJJmZ.svg +1 -0
- package/export-template/docs/assets/retry-escalation-ladder.dark.BdsabtU3.svg +1 -0
- package/export-template/docs/assets/shuffle-map-reduce.KuOEZVmg.svg +1 -0
- package/export-template/docs/assets/shuffle-map-reduce.dark.BgQZnFSb.svg +1 -0
- package/export-template/docs/assets/spill-classification.BU2euYDO.svg +1 -0
- package/export-template/docs/assets/spill-classification.dark.D7i1M40d.svg +1 -0
- package/export-template/docs/assets/style.DXOMCXxn.css +1 -0
- package/export-template/docs/assets/tuning-reference_anti-patterns.md.Df1YMIHu.js +1 -0
- package/export-template/docs/assets/tuning-reference_anti-patterns.md.Df1YMIHu.lean.js +1 -0
- package/export-template/docs/assets/tuning-reference_aqe.md.BIsCtLzm.js +1 -0
- package/export-template/docs/assets/tuning-reference_aqe.md.BIsCtLzm.lean.js +1 -0
- package/export-template/docs/assets/tuning-reference_bottleneck-broadcast-sizing.md.CEstB3Ia.js +1 -0
- package/export-template/docs/assets/tuning-reference_bottleneck-broadcast-sizing.md.CEstB3Ia.lean.js +1 -0
- package/export-template/docs/assets/tuning-reference_bottleneck-cold-start.md.CEuy-72y.js +7 -0
- package/export-template/docs/assets/tuning-reference_bottleneck-cold-start.md.CEuy-72y.lean.js +1 -0
- package/export-template/docs/assets/tuning-reference_bottleneck-duplicate-plan-subtree.md.CIohQDfn.js +1 -0
- package/export-template/docs/assets/tuning-reference_bottleneck-duplicate-plan-subtree.md.CIohQDfn.lean.js +1 -0
- package/export-template/docs/assets/tuning-reference_bottleneck-failures.md.4z5BXGJ2.js +6 -0
- package/export-template/docs/assets/tuning-reference_bottleneck-failures.md.4z5BXGJ2.lean.js +1 -0
- package/export-template/docs/assets/tuning-reference_bottleneck-gc.md.DSxzZRK7.js +6 -0
- package/export-template/docs/assets/tuning-reference_bottleneck-gc.md.DSxzZRK7.lean.js +1 -0
- package/export-template/docs/assets/tuning-reference_bottleneck-job-failure-rate.md.BaJl__1W.js +8 -0
- package/export-template/docs/assets/tuning-reference_bottleneck-job-failure-rate.md.BaJl__1W.lean.js +1 -0
- package/export-template/docs/assets/tuning-reference_bottleneck-memory-utilization.md.DbP-SJZc.js +7 -0
- package/export-template/docs/assets/tuning-reference_bottleneck-memory-utilization.md.DbP-SJZc.lean.js +1 -0
- package/export-template/docs/assets/tuning-reference_bottleneck-retry-waste.md.D5JMjVOt.js +1 -0
- package/export-template/docs/assets/tuning-reference_bottleneck-retry-waste.md.D5JMjVOt.lean.js +1 -0
- package/export-template/docs/assets/tuning-reference_bottleneck-shuffle.md.CM-nTmIH.js +12 -0
- package/export-template/docs/assets/tuning-reference_bottleneck-shuffle.md.CM-nTmIH.lean.js +1 -0
- package/export-template/docs/assets/tuning-reference_bottleneck-skew.md.BdUwiDhn.js +14 -0
- package/export-template/docs/assets/tuning-reference_bottleneck-skew.md.BdUwiDhn.lean.js +1 -0
- package/export-template/docs/assets/tuning-reference_bottleneck-slow-host.md.BlIo6UDW.js +7 -0
- package/export-template/docs/assets/tuning-reference_bottleneck-slow-host.md.BlIo6UDW.lean.js +1 -0
- package/export-template/docs/assets/tuning-reference_bottleneck-small-files.md.B8kloyx8.js +5 -0
- package/export-template/docs/assets/tuning-reference_bottleneck-small-files.md.B8kloyx8.lean.js +1 -0
- package/export-template/docs/assets/tuning-reference_bottleneck-spill.md.PNH7mITt.js +6 -0
- package/export-template/docs/assets/tuning-reference_bottleneck-spill.md.PNH7mITt.lean.js +1 -0
- package/export-template/docs/assets/tuning-reference_bottleneck-straggler.md.DY36fHN5.js +7 -0
- package/export-template/docs/assets/tuning-reference_bottleneck-straggler.md.DY36fHN5.lean.js +1 -0
- package/export-template/docs/assets/tuning-reference_bottleneck-tiny-tasks.md.QTV7O8kU.js +8 -0
- package/export-template/docs/assets/tuning-reference_bottleneck-tiny-tasks.md.QTV7O8kU.lean.js +1 -0
- package/export-template/docs/assets/tuning-reference_bottleneck-utilization.md.DTiueZC3.js +5 -0
- package/export-template/docs/assets/tuning-reference_bottleneck-utilization.md.DTiueZC3.lean.js +1 -0
- package/export-template/docs/assets/tuning-reference_caching.md.B7aQ8asB.js +1 -0
- package/export-template/docs/assets/tuning-reference_caching.md.B7aQ8asB.lean.js +1 -0
- package/export-template/docs/assets/tuning-reference_cluster-config.md.ZVmDGsQ3.js +1 -0
- package/export-template/docs/assets/tuning-reference_cluster-config.md.ZVmDGsQ3.lean.js +1 -0
- package/export-template/docs/assets/tuning-reference_config.md.UvveiWG3.js +1 -0
- package/export-template/docs/assets/tuning-reference_config.md.UvveiWG3.lean.js +1 -0
- package/export-template/docs/assets/tuning-reference_data-formats.md.bjCAWH3N.js +1 -0
- package/export-template/docs/assets/tuning-reference_data-formats.md.bjCAWH3N.lean.js +1 -0
- package/export-template/docs/assets/tuning-reference_index.md.BQ_NooMV.js +1 -0
- package/export-template/docs/assets/tuning-reference_index.md.BQ_NooMV.lean.js +1 -0
- package/export-template/docs/assets/tuning-reference_intro.md.CobD-lGB.js +1 -0
- package/export-template/docs/assets/tuning-reference_intro.md.CobD-lGB.lean.js +1 -0
- package/export-template/docs/assets/tuning-reference_joins.md.BtKs_CuW.js +1 -0
- package/export-template/docs/assets/tuning-reference_joins.md.BtKs_CuW.lean.js +1 -0
- package/export-template/docs/assets/tuning-reference_memory-model.md.DhT-n4y3.js +1 -0
- package/export-template/docs/assets/tuning-reference_memory-model.md.DhT-n4y3.lean.js +1 -0
- package/export-template/docs/assets/tuning-reference_metrics.md.mLOh7Apj.js +1 -0
- package/export-template/docs/assets/tuning-reference_metrics.md.mLOh7Apj.lean.js +1 -0
- package/export-template/docs/assets/tuning-reference_partitioning.md.q0zKF_8X.js +1 -0
- package/export-template/docs/assets/tuning-reference_partitioning.md.q0zKF_8X.lean.js +1 -0
- package/export-template/docs/assets/tuning-reference_pyspark.md.DDCfvN9t.js +6 -0
- package/export-template/docs/assets/tuning-reference_pyspark.md.DDCfvN9t.lean.js +1 -0
- package/export-template/docs/assets/tuning-reference_shuffle.md.BZZ7R4Ix.js +1 -0
- package/export-template/docs/assets/tuning-reference_shuffle.md.BZZ7R4Ix.lean.js +1 -0
- package/export-template/docs/assets/tuning-reference_spark-architecture.md.Dwzm5avO.js +1 -0
- package/export-template/docs/assets/tuning-reference_spark-architecture.md.Dwzm5avO.lean.js +1 -0
- package/export-template/docs/assets/tuning-reference_table-formats.md.D6wj-2dX.js +1 -0
- package/export-template/docs/assets/tuning-reference_table-formats.md.D6wj-2dX.lean.js +1 -0
- package/export-template/docs/assets/udf-execution-models.BUFDICuG.svg +1 -0
- package/export-template/docs/assets/udf-execution-models.dark.YTNS6GDq.svg +1 -0
- package/export-template/docs/assets/user-guide_alternative-log-retrieval.md.sU3KGarf.js +1 -0
- package/export-template/docs/assets/user-guide_alternative-log-retrieval.md.sU3KGarf.lean.js +1 -0
- package/export-template/docs/assets/user-guide_getting-started.md.DtEM37MK.js +3 -0
- package/export-template/docs/assets/user-guide_getting-started.md.DtEM37MK.lean.js +1 -0
- package/export-template/docs/assets/user-guide_mcp-tools.md.C8MiIu7F.js +125 -0
- package/export-template/docs/assets/user-guide_mcp-tools.md.C8MiIu7F.lean.js +1 -0
- package/export-template/docs/assets/user-guide_run-comparison.md.S0TWWmLY.js +1 -0
- package/export-template/docs/assets/user-guide_run-comparison.md.S0TWWmLY.lean.js +1 -0
- package/export-template/docs/assets/user-guide_understanding-findings.md.D0R_Y-R2.js +1 -0
- package/export-template/docs/assets/user-guide_understanding-findings.md.D0R_Y-R2.lean.js +1 -0
- package/export-template/docs/contributor-guide/architecture/board-widgets.html +25 -0
- package/export-template/docs/contributor-guide/architecture/detector-contract.html +25 -0
- package/export-template/docs/contributor-guide/architecture/drill-down.html +25 -0
- package/export-template/docs/contributor-guide/architecture/impact-estimation.html +25 -0
- package/export-template/docs/contributor-guide/architecture/index.html +25 -0
- package/export-template/docs/contributor-guide/architecture/overview.html +25 -0
- package/export-template/docs/contributor-guide/architecture/state-and-history.html +25 -0
- package/export-template/docs/contributor-guide/architecture/widget-rendering.html +25 -0
- package/export-template/docs/contributor-guide/architecture/worker-protocol.html +30 -0
- package/export-template/docs/contributor-guide/contributing.html +25 -0
- package/export-template/docs/contributor-guide/development-setup.html +36 -0
- package/export-template/docs/contributor-guide/testing.html +25 -0
- package/export-template/docs/favicon.svg +4 -0
- package/export-template/docs/hashmap.json +1 -0
- package/export-template/docs/index.html +25 -0
- package/export-template/docs/package.json +1 -0
- package/export-template/docs/tuning-reference/anti-patterns.html +25 -0
- package/export-template/docs/tuning-reference/aqe.html +25 -0
- package/export-template/docs/tuning-reference/bottleneck-broadcast-sizing.html +25 -0
- package/export-template/docs/tuning-reference/bottleneck-cold-start.html +31 -0
- package/export-template/docs/tuning-reference/bottleneck-duplicate-plan-subtree.html +25 -0
- package/export-template/docs/tuning-reference/bottleneck-failures.html +30 -0
- package/export-template/docs/tuning-reference/bottleneck-gc.html +30 -0
- package/export-template/docs/tuning-reference/bottleneck-job-failure-rate.html +32 -0
- package/export-template/docs/tuning-reference/bottleneck-memory-utilization.html +31 -0
- package/export-template/docs/tuning-reference/bottleneck-retry-waste.html +25 -0
- package/export-template/docs/tuning-reference/bottleneck-shuffle.html +36 -0
- package/export-template/docs/tuning-reference/bottleneck-skew.html +38 -0
- package/export-template/docs/tuning-reference/bottleneck-slow-host.html +31 -0
- package/export-template/docs/tuning-reference/bottleneck-small-files.html +29 -0
- package/export-template/docs/tuning-reference/bottleneck-spill.html +30 -0
- package/export-template/docs/tuning-reference/bottleneck-straggler.html +31 -0
- package/export-template/docs/tuning-reference/bottleneck-tiny-tasks.html +32 -0
- package/export-template/docs/tuning-reference/bottleneck-utilization.html +29 -0
- package/export-template/docs/tuning-reference/caching.html +25 -0
- package/export-template/docs/tuning-reference/cluster-config.html +25 -0
- package/export-template/docs/tuning-reference/config.html +25 -0
- package/export-template/docs/tuning-reference/data-formats.html +25 -0
- package/export-template/docs/tuning-reference/index.html +25 -0
- package/export-template/docs/tuning-reference/intro.html +25 -0
- package/export-template/docs/tuning-reference/joins.html +25 -0
- package/export-template/docs/tuning-reference/memory-model.html +25 -0
- package/export-template/docs/tuning-reference/metrics.html +25 -0
- package/export-template/docs/tuning-reference/partitioning.html +25 -0
- package/export-template/docs/tuning-reference/pyspark.html +30 -0
- package/export-template/docs/tuning-reference/shuffle.html +25 -0
- package/export-template/docs/tuning-reference/spark-architecture.html +25 -0
- package/export-template/docs/tuning-reference/table-formats.html +25 -0
- package/export-template/docs/user-guide/alternative-log-retrieval.html +25 -0
- package/export-template/docs/user-guide/getting-started.html +27 -0
- package/export-template/docs/user-guide/mcp-tools.html +149 -0
- package/export-template/docs/user-guide/run-comparison.html +25 -0
- package/export-template/docs/user-guide/understanding-findings.html +25 -0
- package/export-template/docs/vp-icons.css +0 -0
- package/export-template/favicon.svg +4 -0
- package/export-template/index.html +115 -0
- package/export-template/parser-worker-QqyEE4m9.js +64 -0
- package/package.json +16 -3
- package/vendor-core/analyzer.js +74 -74
- package/vendor-core/cli/budgets.js +13 -27
- package/vendor-core/cli/collect-run.js +43 -19
- package/vendor-core/core-count.js +25 -27
- package/vendor-core/core-locality-ratio.js +4 -11
- package/vendor-core/core-time-series.js +6 -12
- package/vendor-core/core-usage-locality.js +3 -4
- package/vendor-core/detectors.js +256 -375
- package/vendor-core/docs-config.js +69 -21
- package/vendor-core/docs-content/chapters/01-intro.md +32 -0
- package/vendor-core/docs-content/chapters/02-spark-architecture.md +76 -0
- package/vendor-core/docs-content/chapters/03-memory-model.md +73 -0
- package/vendor-core/docs-content/chapters/04-partitioning.md +65 -0
- package/vendor-core/docs-content/chapters/05-joins.md +62 -0
- package/vendor-core/docs-content/chapters/06-shuffle.md +59 -0
- package/vendor-core/docs-content/chapters/07-data-formats.md +81 -0
- package/vendor-core/docs-content/chapters/07b-table-formats.md +56 -0
- package/vendor-core/docs-content/chapters/08-caching.md +58 -0
- package/vendor-core/docs-content/chapters/09-pyspark.md +78 -0
- package/vendor-core/docs-content/chapters/10-aqe.md +167 -0
- package/vendor-core/docs-content/chapters/11-cluster-config.md +170 -0
- package/vendor-core/docs-content/chapters/12-anti-patterns.md +171 -0
- package/vendor-core/docs-content/chapters/14-metrics.md +87 -0
- package/vendor-core/docs-content/chapters/15-config.md +93 -0
- package/vendor-core/docs-content/chapters/nav-index.json +370 -0
- package/vendor-core/docs-content/detection/cache.md +6 -0
- package/vendor-core/docs-content/detection/cfg.md +15 -0
- package/vendor-core/docs-content/detection/chrn.md +7 -0
- package/vendor-core/docs-content/detection/cold.md +4 -0
- package/vendor-core/docs-content/detection/cstor.md +4 -0
- package/vendor-core/docs-content/detection/fail.md +5 -0
- package/vendor-core/docs-content/detection/gc.md +4 -0
- package/vendor-core/docs-content/detection/host.md +5 -0
- package/vendor-core/docs-content/detection/incmp.md +6 -0
- package/vendor-core/docs-content/detection/jobs.md +4 -0
- package/vendor-core/docs-content/detection/local.md +7 -0
- package/vendor-core/docs-content/detection/mem.md +10 -0
- package/vendor-core/docs-content/detection/part.md +5 -0
- package/vendor-core/docs-content/detection/plan.md +14 -0
- package/vendor-core/docs-content/detection/retry.md +4 -0
- package/vendor-core/docs-content/detection/sfail.md +5 -0
- package/vendor-core/docs-content/detection/shape.md +5 -0
- package/vendor-core/docs-content/detection/shfl.md +4 -0
- package/vendor-core/docs-content/detection/skew.md +6 -0
- package/vendor-core/docs-content/detection/slow.md +6 -0
- package/vendor-core/docs-content/detection/spec.md +7 -0
- package/vendor-core/docs-content/detection/spill.md +7 -0
- package/vendor-core/docs-content/detection/strag.md +5 -0
- package/vendor-core/docs-content/detection/tiny.md +4 -0
- package/vendor-core/docs-content/detection/util.md +4 -0
- package/vendor-core/docs-content/diagrams/aqe-loop.dark.svg +1 -0
- package/vendor-core/docs-content/diagrams/aqe-loop.svg +1 -0
- package/vendor-core/docs-content/diagrams/broadcast-vs-shuffle.dark.svg +1 -0
- package/vendor-core/docs-content/diagrams/broadcast-vs-shuffle.svg +1 -0
- package/vendor-core/docs-content/diagrams/cache-lifecycle.dark.svg +1 -0
- package/vendor-core/docs-content/diagrams/cache-lifecycle.svg +1 -0
- package/vendor-core/docs-content/diagrams/cold-start-timeline.dark.svg +1 -0
- package/vendor-core/docs-content/diagrams/cold-start-timeline.svg +1 -0
- package/vendor-core/docs-content/diagrams/columnar-layout.dark.svg +1 -0
- package/vendor-core/docs-content/diagrams/columnar-layout.svg +1 -0
- package/vendor-core/docs-content/diagrams/container-memory.dark.svg +1 -0
- package/vendor-core/docs-content/diagrams/container-memory.svg +1 -0
- package/vendor-core/docs-content/diagrams/dag-stages.dark.svg +1 -0
- package/vendor-core/docs-content/diagrams/dag-stages.svg +1 -0
- package/vendor-core/docs-content/diagrams/driver-executor.dark.svg +1 -0
- package/vendor-core/docs-content/diagrams/driver-executor.svg +1 -0
- package/vendor-core/docs-content/diagrams/duplicate-plan-subtree.dark.svg +1 -0
- package/vendor-core/docs-content/diagrams/duplicate-plan-subtree.svg +1 -0
- package/vendor-core/docs-content/diagrams/join-strategy.dark.svg +1 -0
- package/vendor-core/docs-content/diagrams/join-strategy.svg +1 -0
- package/vendor-core/docs-content/diagrams/memory-borrowing.dark.svg +1 -0
- package/vendor-core/docs-content/diagrams/memory-borrowing.svg +1 -0
- package/vendor-core/docs-content/diagrams/memory-regions.dark.svg +1 -0
- package/vendor-core/docs-content/diagrams/memory-regions.svg +1 -0
- package/vendor-core/docs-content/diagrams/repartition-vs-coalesce.dark.svg +1 -0
- package/vendor-core/docs-content/diagrams/repartition-vs-coalesce.svg +1 -0
- package/vendor-core/docs-content/diagrams/retry-escalation-ladder.dark.svg +1 -0
- package/vendor-core/docs-content/diagrams/retry-escalation-ladder.svg +1 -0
- package/vendor-core/docs-content/diagrams/shuffle-map-reduce.dark.svg +1 -0
- package/vendor-core/docs-content/diagrams/shuffle-map-reduce.svg +1 -0
- package/vendor-core/docs-content/diagrams/spill-classification.dark.svg +1 -0
- package/vendor-core/docs-content/diagrams/spill-classification.svg +1 -0
- package/vendor-core/docs-content/diagrams/udf-execution-models.dark.svg +1 -0
- package/vendor-core/docs-content/diagrams/udf-execution-models.svg +1 -0
- package/vendor-core/docs-content/tuning/broadcast-sizing.md +78 -0
- package/vendor-core/docs-content/tuning/cold-start.md +81 -0
- package/vendor-core/docs-content/tuning/duplicate-plan-subtree.md +45 -0
- package/vendor-core/docs-content/tuning/failures.md +124 -0
- package/vendor-core/docs-content/tuning/gc.md +110 -0
- package/vendor-core/docs-content/tuning/job-failure-rate.md +101 -0
- package/vendor-core/docs-content/tuning/memory-utilization.md +58 -0
- package/vendor-core/docs-content/tuning/retry-waste.md +90 -0
- package/vendor-core/docs-content/tuning/shuffle.md +154 -0
- package/vendor-core/docs-content/tuning/skew.md +123 -0
- package/vendor-core/docs-content/tuning/slow-host.md +117 -0
- package/vendor-core/docs-content/tuning/small-files.md +99 -0
- package/vendor-core/docs-content/tuning/spill.md +114 -0
- package/vendor-core/docs-content/tuning/straggler.md +103 -0
- package/vendor-core/docs-content/tuning/tiny-tasks.md +94 -0
- package/vendor-core/docs-content/tuning/utilization.md +90 -0
- package/vendor-core/docs-site-config.js +10 -17
- package/vendor-core/efficiency-model.js +7 -13
- package/vendor-core/etl-phases.js +3 -5
- package/vendor-core/event-handlers.js +232 -134
- package/vendor-core/event-schemas.js +48 -114
- package/vendor-core/evidence-availability.js +5 -10
- package/vendor-core/evidence-report.js +72 -122
- package/vendor-core/export-data.js +48 -0
- package/vendor-core/finding-action-label.js +4 -10
- package/vendor-core/finding-filter-predicate.js +3 -7
- package/vendor-core/finding-generic-recommendation.js +112 -0
- package/vendor-core/finding-names.js +51 -0
- package/vendor-core/format-utils.js +112 -38
- package/vendor-core/impact-band.js +18 -24
- package/vendor-core/impact-estimator.js +38 -74
- package/vendor-core/ingest.js +7 -13
- package/vendor-core/job-groups.js +3 -6
- package/vendor-core/list-runs.js +278 -0
- package/vendor-core/load-vendored.js +6 -12
- package/vendor-core/log-header-peek.js +81 -0
- package/vendor-core/lz4-block.js +4 -6
- package/vendor-core/mcp-server-factory.js +38 -8
- package/vendor-core/mcp-tools.js +105 -76
- package/vendor-core/model-assembler.js +8 -16
- package/vendor-core/occupancy.js +5 -9
- package/vendor-core/parser-worker.js +18 -27
- package/vendor-core/plan-dot.js +2 -5
- package/vendor-core/plan-duration-attribution.js +78 -29
- package/vendor-core/plan-graph-model.js +126 -69
- package/vendor-core/plan-node-detail.js +31 -17
- package/vendor-core/plan-summary.js +19 -8
- package/vendor-core/recommendation-rollup.js +35 -39
- package/vendor-core/redact.js +72 -16
- package/vendor-core/rolling-log-reassembly.js +4 -6
- package/vendor-core/run-comparison.js +65 -70
- package/vendor-core/scaling-sim.js +5 -7
- package/vendor-core/session-snapshot.js +1 -1
- package/vendor-core/shs-fetch.js +4 -6
- package/vendor-core/shs-load.js +9 -13
- package/vendor-core/shs-request.js +1 -1
- package/vendor-core/stage-quantiles.js +14 -0
- package/vendor-core/types.js +78 -18
- package/vendor-core/wasted-core-hours.js +7 -12
package/vendor-core/detectors.js
CHANGED
|
@@ -1,33 +1,23 @@
|
|
|
1
1
|
import { pathBasename, formatBytes, nsToMs, IMPACT_BAND_ORDER } from './format-utils.js';
|
|
2
2
|
import { scanRelationId } from './plan-summary.js';
|
|
3
|
-
import {
|
|
3
|
+
import { computePeakConcurrentCores, computePeakConcurrentExecutorCount } from './core-count.js';
|
|
4
4
|
import { walkPlanTree } from './plan-tree-walk.js';
|
|
5
5
|
import { computeCoreLocalityRatio } from './core-locality-ratio.js';
|
|
6
6
|
import { estimateSingleStage, } from './occupancy.js';
|
|
7
|
+
import { isExchangeNode, isBroadcastExchangeNode } from './plan-node-detail.js';
|
|
7
8
|
|
|
8
9
|
|
|
9
10
|
const MB = 1024 * 1024;
|
|
10
11
|
const GB = 1024 * MB;
|
|
11
12
|
const TB = 1024 * GB;
|
|
12
13
|
|
|
13
|
-
// ---------------------------------------------------------------------------
|
|
14
14
|
// Local runtime shapes.
|
|
15
15
|
//
|
|
16
|
-
//
|
|
17
|
-
//
|
|
18
|
-
//
|
|
19
|
-
//
|
|
20
|
-
//
|
|
21
|
-
// `event-handlers.ts`'s stage/sql/app records actually carry, accessed
|
|
22
|
-
// directly (arithmetic, comparisons) rather than read-and-display, so an
|
|
23
|
-
// index-signature-shaped type would force an `unknown` cast at nearly every
|
|
24
|
-
// field access. Every field below is verified against a real access in this
|
|
25
|
-
// file (or a helper it calls); none are speculative.
|
|
26
|
-
//
|
|
27
|
-
// `analyzer.js` (the only real caller of `detect()`, still plain JS) passes
|
|
28
|
-
// whatever the parser actually produced, so these types describe reality,
|
|
29
|
-
// not a narrowing of some existing stricter type; there is nothing unsound
|
|
30
|
-
// about them being independent of `types.ts`'s `Stage`/`SqlExecution`.
|
|
16
|
+
// types.ts's Stage/SqlExecution/SparkAppInfo describe the posted AppModel surface for the view
|
|
17
|
+
// layer (with a catch-all index signature). This file needs the FULL set of fields finalizeStage
|
|
18
|
+
// computes and event-handlers.ts records carry, accessed directly (arithmetic, comparisons), so
|
|
19
|
+
// an index-signature type would force an `unknown` cast at nearly every access. Every field below
|
|
20
|
+
// is verified against a real access here.
|
|
31
21
|
|
|
32
22
|
|
|
33
23
|
|
|
@@ -35,9 +25,18 @@ const TB = 1024 * GB;
|
|
|
35
25
|
|
|
36
26
|
|
|
37
27
|
|
|
38
|
-
|
|
39
|
-
|
|
40
|
-
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
// Per-executor snapshot from a StageExecutorMetrics event: a loose bag of Spark's
|
|
39
|
+
// ExecutorMetrics field names, only a few of which any detector reads.
|
|
41
40
|
|
|
42
41
|
|
|
43
42
|
|
|
@@ -81,6 +80,8 @@ const TB = 1024 * GB;
|
|
|
81
80
|
|
|
82
81
|
|
|
83
82
|
|
|
83
|
+
|
|
84
|
+
|
|
84
85
|
|
|
85
86
|
|
|
86
87
|
|
|
@@ -116,27 +117,19 @@ const TB = 1024 * GB;
|
|
|
116
117
|
|
|
117
118
|
|
|
118
119
|
|
|
119
|
-
|
|
120
|
-
|
|
120
|
+
|
|
121
121
|
|
|
122
122
|
|
|
123
123
|
|
|
124
124
|
|
|
125
|
-
// The full context
|
|
126
|
-
//
|
|
127
|
-
// 'app' scope (`d.detect(ctx)`). 'config' scope gets a narrower `{ app }`
|
|
128
|
-
// (see auditConfig in analyzer.js), typed per-entry below instead of here.
|
|
125
|
+
// The full context analyze() passes as every stage/sql detect()'s second arg, and as the sole
|
|
126
|
+
// arg for 'app' scope. 'config' scope gets a narrower `{ app }` (auditConfig), typed per-entry.
|
|
129
127
|
//
|
|
130
|
-
// `app` is non-nullable here (unlike
|
|
131
|
-
//
|
|
132
|
-
//
|
|
133
|
-
//
|
|
134
|
-
//
|
|
135
|
-
// (harmless on a non-nullable value), and `coldStart` reads `app.startTime`
|
|
136
|
-
// with no guard at all, which only type-checks if `app` is non-nullable.
|
|
137
|
-
// `auditConfig`'s separate config-scope target type keeps `app` nullable
|
|
138
|
-
// instead, since `auditConfig(appModel.app)` (src/analyzer.js) really can be
|
|
139
|
-
// called with a `null` app and every config-scope entry optional-chains it.
|
|
128
|
+
// `app` is typed as non-nullable here (unlike AppModel.app), but incompleteRun, coldStart,
|
|
129
|
+
// utilization, and autoscalingChurn defensively guard against null at runtime to tolerate
|
|
130
|
+
// malformed/incomplete logs. Despite the type annotation, app may be null in edge cases, and
|
|
131
|
+
// these detectors handle it gracefully. auditConfig's config-scope target keeps `app` nullable
|
|
132
|
+
// instead.
|
|
140
133
|
|
|
141
134
|
|
|
142
135
|
|
|
@@ -145,17 +138,13 @@ const TB = 1024 * GB;
|
|
|
145
138
|
|
|
146
139
|
|
|
147
140
|
|
|
148
|
-
|
|
149
|
-
|
|
150
|
-
|
|
151
|
-
|
|
152
|
-
|
|
141
|
+
|
|
142
|
+
|
|
153
143
|
|
|
154
144
|
|
|
155
145
|
|
|
156
|
-
//
|
|
157
|
-
//
|
|
158
|
-
// (`SparkAppInfo | null`), independent of the full `DetectorCtx` above.
|
|
146
|
+
// auditConfig(app) calls every config-scope detect({ app }) with whatever appModel.app is
|
|
147
|
+
// (SparkAppInfo | null), independent of DetectorCtx.
|
|
159
148
|
|
|
160
149
|
|
|
161
150
|
|
|
@@ -167,8 +156,7 @@ const TB = 1024 * GB;
|
|
|
167
156
|
|
|
168
157
|
|
|
169
158
|
|
|
170
|
-
//
|
|
171
|
-
// Returns { magnitude } | null. Magnitude ∈ 'severe'|'high'|'medium'.
|
|
159
|
+
// SpillPressureDetector (5a) + SpillSkewDetector (5b).
|
|
172
160
|
function computeSpillMagnitude(
|
|
173
161
|
stage ,
|
|
174
162
|
t ,
|
|
@@ -195,15 +183,9 @@ function pickDominantReason(reasons )
|
|
|
195
183
|
return [...reasons].sort((a, b) => b.count - a.count)[0].reason;
|
|
196
184
|
}
|
|
197
185
|
|
|
198
|
-
// Shared by every scope:'sql' detector
|
|
199
|
-
//
|
|
200
|
-
//
|
|
201
|
-
// stage's own (correctly populated) `sqlExecutionId`. Parameter type is
|
|
202
|
-
// intentionally the minimal shape needed (not `DetectorStage`): callers
|
|
203
|
-
// outside this file (Topbar.tsx, plan-node-detail.ts, plan-graph-model.ts)
|
|
204
|
-
// pass `appModel.stages`, typed `Map<StageId, Stage>` per types.ts, which
|
|
205
|
-
// carries `id`/`sqlExecutionId` but not this file's fuller `DetectorStage`
|
|
206
|
-
// shape.
|
|
186
|
+
// Shared by every scope:'sql' detector. sql.get(id).stageIds is always empty (parser-worker
|
|
187
|
+
// never populates it), so stage linkage is derived from each stage's own sqlExecutionId.
|
|
188
|
+
// Minimal param shape (not DetectorStage): external callers pass Map<StageId, Stage>.
|
|
207
189
|
export function stageIdsForSqlExec(
|
|
208
190
|
executionId ,
|
|
209
191
|
stages ,
|
|
@@ -213,38 +195,23 @@ export function stageIdsForSqlExec(
|
|
|
213
195
|
return out;
|
|
214
196
|
}
|
|
215
197
|
|
|
216
|
-
// Shared by the three Plan Advisor detectors
|
|
217
|
-
//
|
|
218
|
-
//
|
|
219
|
-
// coverage. A finding never partially blends a narrowed set with the
|
|
220
|
-
// execution-wide one (see docs/architecture.md's plan-node-to-stage mapping
|
|
221
|
-
// section). `fallback` is a param, not computed here, so callers can compute
|
|
222
|
-
// `stageIdsForSqlExec` once per `detect()` call and reuse it across every
|
|
223
|
-
// finding in that call instead of re-walking `stages` per finding.
|
|
198
|
+
// Shared by the three Plan Advisor detectors: union the given nodes' own stageIds, or fall back
|
|
199
|
+
// to the whole execution's stage set when none have coverage (never a partial blend). `fallback`
|
|
200
|
+
// is a param so callers compute stageIdsForSqlExec once per detect() and reuse it per finding.
|
|
224
201
|
export function unionStageIds(nodes , fallback ) {
|
|
225
202
|
const union = new Set ();
|
|
226
203
|
for (const node of nodes) for (const sid of node.stageIds ?? []) union.add(sid);
|
|
227
204
|
return union.size > 0 ? [...union].sort((a, b) => a - b) : fallback;
|
|
228
205
|
}
|
|
229
206
|
|
|
230
|
-
// Bottom-up shape computation for duplicate-subtree detection
|
|
231
|
-
//
|
|
232
|
-
//
|
|
233
|
-
//
|
|
234
|
-
// encodes operator name + sorted metric *names* (never values, per spec) +
|
|
235
|
-
// children fingerprints in original order, so two subtrees with the same shape
|
|
236
|
-
// but different row counts/literals still collide, which is the point (we
|
|
237
|
-
// have no expr/plan/codegen IDs to strip in the first place, since `metrics`
|
|
238
|
-
// never carried them).
|
|
207
|
+
// Bottom-up shape computation for duplicate-subtree detection and the
|
|
208
|
+
// cachingOpportunity composite detector. `size` is the subtree node count; the default
|
|
209
|
+
// fingerprint encodes operator name + sorted metric NAMES (never values, per spec) + child
|
|
210
|
+
// fingerprints, so two subtrees with the same shape but different values still collide.
|
|
239
211
|
//
|
|
240
|
-
//
|
|
241
|
-
//
|
|
242
|
-
//
|
|
243
|
-
// This is what lets a caller ask "does this specific node's own detail +
|
|
244
|
-
// structural shape match another node's", without descendant filter/scan
|
|
245
|
-
// detail (literals, paths) ever entering the comparison. The existing
|
|
246
|
-
// `duplicatePlanSubtree` call site passes no options: identical behavior,
|
|
247
|
-
// zero regression to its existing tests.
|
|
212
|
+
// opts.includeDetail (default false) folds root's own detail (via opts.normalizeDetail) into
|
|
213
|
+
// ONLY root's fingerprint, never a child's: lets a caller match "this node's own detail + shape"
|
|
214
|
+
// without descendant scan detail entering the comparison.
|
|
248
215
|
|
|
249
216
|
|
|
250
217
|
export function computePlanShapes(
|
|
@@ -256,9 +223,22 @@ export function computePlanShapes(
|
|
|
256
223
|
const allNodes = [];
|
|
257
224
|
function visit(node , isRoot ) {
|
|
258
225
|
allNodes.push(node);
|
|
259
|
-
|
|
226
|
+
// A read half wrapping a write half (the Exchange split from
|
|
227
|
+
// resolvePlanTree, see event-handlers.ts) is one logical Spark operator
|
|
228
|
+
// for subtree-shape purposes. Without this, every real Exchange in a
|
|
229
|
+
// matched subtree would count twice, inflating duplicatePlanSubtree's
|
|
230
|
+
// reported subtreeSize and shifting its groupIndex-derived findingIds
|
|
231
|
+
// for an otherwise-unchanged plan. Skip straight through the write
|
|
232
|
+
// wrapper: size comes from its real children, and metricNames from its
|
|
233
|
+
// real metrics (the read half's own metrics are always empty), so
|
|
234
|
+
// fingerprint distinctiveness between different real Exchanges is
|
|
235
|
+
// preserved too.
|
|
236
|
+
const writeHalf = node.exchangeRole === 'read' ? node.children[0] : null;
|
|
237
|
+
const realChildren = writeHalf ? writeHalf.children : (node.children ?? []);
|
|
238
|
+
const realMetrics = writeHalf ? writeHalf.metrics : node.metrics;
|
|
239
|
+
const childShapes = realChildren.map((c) => visit(c, false));
|
|
260
240
|
const size = 1 + childShapes.reduce((sum, c) => sum + c.size, 0);
|
|
261
|
-
const metricNames = (
|
|
241
|
+
const metricNames = (realMetrics ?? []).map((m) => m.name).sort().join(',');
|
|
262
242
|
const childFingerprints = childShapes.map((c) => c.fingerprint).join(',');
|
|
263
243
|
const fingerprint = isRoot && includeDetail
|
|
264
244
|
? `${node.name}[${metricNames}]<${normalize(node.detail ?? '')}>{${childFingerprints}}`
|
|
@@ -271,16 +251,10 @@ export function computePlanShapes(
|
|
|
271
251
|
return { shapeOf, allNodes };
|
|
272
252
|
}
|
|
273
253
|
|
|
274
|
-
// Normalizes an anchor join/union node's
|
|
275
|
-
//
|
|
276
|
-
// (
|
|
277
|
-
//
|
|
278
|
-
// expr ids, plan/codegen-stage ids, and AQE's runtime BuildLeft/BuildRight
|
|
279
|
-
// broadcast-side choice (which can flip between executions of the logically
|
|
280
|
-
// identical join based on runtime stats); it then canonicalizes commutative
|
|
281
|
-
// equality operand order so `A.x = B.y` and `B.y = A.x` collide. Everything
|
|
282
|
-
// else (join type, columns, literal values) is kept: that's the
|
|
283
|
-
// semantically meaningful part a leaf-relation-set identity would miss.
|
|
254
|
+
// Normalizes an anchor join/union node's detail for the cachingOpportunity composite
|
|
255
|
+
// fingerprint. Strips per-analysis numbering (expr ids, plan/codegen ids) and AQE's runtime
|
|
256
|
+
// BuildLeft/BuildRight choice (can flip between runs), then canonicalizes commutative equality
|
|
257
|
+
// operand order so `A.x = B.y` and `B.y = A.x` collide. Join type, columns, literals are kept.
|
|
284
258
|
export function normalizeDetail(detail ) {
|
|
285
259
|
let s = detail
|
|
286
260
|
.replace(/#\d+L?/g, '')
|
|
@@ -296,34 +270,21 @@ export function normalizeDetail(detail ) {
|
|
|
296
270
|
|
|
297
271
|
const JOIN_NAME_RE = /Join/i;
|
|
298
272
|
|
|
299
|
-
// Structural operator kind for cachingOpportunity's composite detection:
|
|
300
|
-
//
|
|
301
|
-
//
|
|
302
|
-
// plan-summary.js's visitJoin recognizes); 'union' is Spark's exact `Union`
|
|
303
|
-
// node name. CartesianProduct is deliberately excluded (out of scope, same
|
|
304
|
-
// as plan-summary.js's join handling).
|
|
273
|
+
// Structural operator kind for cachingOpportunity's composite detection: 'join' covers every
|
|
274
|
+
// Spark join physical operator; 'union' is Spark's exact `Union` node. CartesianProduct is
|
|
275
|
+
// deliberately excluded (out of scope, as in plan-summary.ts).
|
|
305
276
|
export function planOperatorKind(name ) {
|
|
306
277
|
if (JOIN_NAME_RE.test(name)) return 'join';
|
|
307
278
|
if (name === 'Union') return 'union';
|
|
308
279
|
return null;
|
|
309
280
|
}
|
|
310
281
|
|
|
311
|
-
// Single bottom-up pass
|
|
312
|
-
//
|
|
313
|
-
//
|
|
314
|
-
//
|
|
315
|
-
//
|
|
316
|
-
//
|
|
317
|
-
// computePlanShapes's default path) and merges each subtree's leaf-relation
|
|
318
|
-
// byte map (via scanRelationId, same identity cachingOpportunity's existing
|
|
319
|
-
// leaf aggregation uses) bottom-up. When a node is a join/union, its anchor
|
|
320
|
-
// fingerprint folds only its OWN normalized detail (per computePlanShapes's
|
|
321
|
-
// opts.includeDetail contract), computed here inline, in O(1), from the
|
|
322
|
-
// node's own detail plus its already-computed child fingerprints, not via a
|
|
323
|
-
// nested computePlanShapes call. `ancestorNodes` (strict ancestors, root
|
|
324
|
-
// first) is threaded down for free via the recursion's own call stack, so
|
|
325
|
-
// later nested-composite dedupe (cachingOpportunity.detect()) doesn't need a
|
|
326
|
-
// separate tree walk to determine containment.
|
|
282
|
+
// Single bottom-up O(n) pass producing one composite candidate per join/union node. Deliberately
|
|
283
|
+
// does NOT call computePlanShapes per node (subtrees overlap, that would be O(n·k) on
|
|
284
|
+
// star/snowflake joins): computes the plain fingerprint once and merges each subtree's
|
|
285
|
+
// leaf-relation byte map bottom-up. A join/union's anchor fingerprint folds only its OWN
|
|
286
|
+
// normalized detail, inline in O(1). `ancestorNodes` threads down via the call stack so
|
|
287
|
+
// nested-composite dedupe needs no separate tree walk.
|
|
327
288
|
|
|
328
289
|
|
|
329
290
|
|
|
@@ -374,17 +335,10 @@ export function findCompositeCandidates(root ) {
|
|
|
374
335
|
}
|
|
375
336
|
|
|
376
337
|
|
|
377
|
-
// First scanned
|
|
378
|
-
//
|
|
379
|
-
//
|
|
380
|
-
//
|
|
381
|
-
// duplicate groups that scan different tables (real-log bug: two unrelated
|
|
382
|
-
// "BroadcastExchange over a Project/Filter/Scan" patterns, one per dimension
|
|
383
|
-
// table, produced byte-identical findings). This is best-effort/informational
|
|
384
|
-
// only, not a uniqueness guarantee: it's null for scan-less subtrees (JDBC/
|
|
385
|
-
// Kafka/LocalRelation) and can coincide when two groups share their first-
|
|
386
|
-
// encountered leaf. See findDuplicateSubtrees's groupIndex for the actual
|
|
387
|
-
// discriminator findingId() relies on.
|
|
338
|
+
// First scanned relation identity in a subtree (pre-order), or null when none. Surfaced on
|
|
339
|
+
// duplicatePlanSubtree findings as `sampleRelation` so a user can tell apart same-shaped groups
|
|
340
|
+
// that scan different tables. Best-effort only: null for scan-less subtrees, can coincide across
|
|
341
|
+
// groups; findDuplicateSubtrees's groupIndex is the actual discriminator findingId relies on.
|
|
388
342
|
function firstLeafRelationId(node ) {
|
|
389
343
|
let found = null;
|
|
390
344
|
walkPlanTree(node, (n) => {
|
|
@@ -394,14 +348,10 @@ function firstLeafRelationId(node ) {
|
|
|
394
348
|
return found;
|
|
395
349
|
}
|
|
396
350
|
|
|
397
|
-
// Groups nodes by fingerprint, keeping
|
|
398
|
-
//
|
|
399
|
-
//
|
|
400
|
-
//
|
|
401
|
-
// smaller, fully-nested duplicate group inside an already-accepted match is
|
|
402
|
-
// dropped (a 5-node duplicate should not also emit findings for its 3-node
|
|
403
|
-
// sub-subtrees); occurrences of a smaller group that fall OUTSIDE any
|
|
404
|
-
// accepted larger match still count normally.
|
|
351
|
+
// Groups nodes by fingerprint, keeping groups of size >= minOccurrences whose subtree size is
|
|
352
|
+
// >= minSubtreeSize. De-overlap: process largest-subtree-first, and once a group is accepted mark
|
|
353
|
+
// every node in its matches "claimed" so a smaller fully-nested duplicate is dropped; occurrences
|
|
354
|
+
// of a smaller group OUTSIDE any accepted match still count.
|
|
405
355
|
|
|
406
356
|
|
|
407
357
|
|
|
@@ -417,7 +367,12 @@ export function findDuplicateSubtrees(
|
|
|
417
367
|
{ minSubtreeSize, minOccurrences } ,
|
|
418
368
|
) {
|
|
419
369
|
const { shapeOf, allNodes } = computePlanShapes(root);
|
|
420
|
-
|
|
370
|
+
// Defensive, not load-bearing: computePlanShapes's visit() already skips
|
|
371
|
+
// straight through a read node to its write half's real children, so no
|
|
372
|
+
// write-half node is ever pushed into allNodes in the first place, this
|
|
373
|
+
// filter can structurally never exclude anything. Kept in case that
|
|
374
|
+
// invariant ever changes upstream.
|
|
375
|
+
const eligible = allNodes.filter(n => shapeOf.get(n) .size >= minSubtreeSize && n.exchangeRole !== 'write');
|
|
421
376
|
|
|
422
377
|
const groups = new Map ();
|
|
423
378
|
for (const n of eligible) {
|
|
@@ -445,14 +400,10 @@ export function findDuplicateSubtrees(
|
|
|
445
400
|
rootName: unclaimed[0].name,
|
|
446
401
|
subtreeSize: shapeOf.get(unclaimed[0]) .size,
|
|
447
402
|
occurrences: unclaimed.length,
|
|
448
|
-
isExchangeRoot:
|
|
403
|
+
isExchangeRoot: isExchangeNode(unclaimed[0]),
|
|
449
404
|
sampleRelation: firstLeafRelationId(unclaimed[0]),
|
|
450
|
-
// Deterministic position within this execution's group list
|
|
451
|
-
// is best-effort (null
|
|
452
|
-
// JDBC/Kafka/LocalRelation sources, or identical when two groups happen to share
|
|
453
|
-
// their first-encountered leaf) and is NOT sufficient on its own to guarantee two
|
|
454
|
-
// structurally-distinct groups get distinct finding ids; groupIndex is the actual
|
|
455
|
-
// uniqueness guarantee findingId() relies on.
|
|
405
|
+
// Deterministic position within this execution's group list: the actual uniqueness
|
|
406
|
+
// guarantee findingId relies on, since sampleRelation is best-effort (null or coincident).
|
|
456
407
|
groupIndex: results.length,
|
|
457
408
|
nodes: unclaimed,
|
|
458
409
|
});
|
|
@@ -460,21 +411,16 @@ export function findDuplicateSubtrees(
|
|
|
460
411
|
return results;
|
|
461
412
|
}
|
|
462
413
|
|
|
463
|
-
// Exact metric names Spark emits
|
|
464
|
-
//
|
|
465
|
-
// NOTE: the write-side byte metric is "written output", not "size of written
|
|
466
|
-
// files" as an earlier draft of this detector's spec assumed.
|
|
414
|
+
// Exact metric names Spark emits, verified against a real SQLExecutionStart's sparkPlanInfo.
|
|
415
|
+
// The write-side byte metric is "written output", not "size of written files".
|
|
467
416
|
const FILES_READ_COUNT = 'number of files read';
|
|
468
417
|
const FILES_READ_BYTES = 'size of files read';
|
|
469
418
|
const FILES_WRITTEN_COUNT = 'number of written files';
|
|
470
419
|
const FILES_WRITTEN_BYTES = 'written output';
|
|
471
420
|
|
|
472
|
-
// Byte size of a join-side subtree for broadcast sizing
|
|
473
|
-
//
|
|
474
|
-
//
|
|
475
|
-
// beneath it (e.g. an Exchange's "data size" already reflects everything it
|
|
476
|
-
// shuffled), so summing further down would double-count. Only recurses into
|
|
477
|
-
// children when the current node carries no such metric.
|
|
421
|
+
// Byte size of a join-side subtree for broadcast sizing. Stops
|
|
422
|
+
// descending at a node with a "data size" metric: that value already aggregates everything
|
|
423
|
+
// beneath it, so summing further would double-count. Only recurses when no such metric.
|
|
478
424
|
function sumBoundarySize(node ) {
|
|
479
425
|
const m = (node.metrics ?? []).find(x => x.name === 'data size');
|
|
480
426
|
if (m) return m.value;
|
|
@@ -483,10 +429,8 @@ function sumBoundarySize(node ) {
|
|
|
483
429
|
return sum;
|
|
484
430
|
}
|
|
485
431
|
|
|
486
|
-
// Nodes that
|
|
487
|
-
//
|
|
488
|
-
// so the implicated stageIds line up with the size that was actually
|
|
489
|
-
// compared, however deep that turns out to live.
|
|
432
|
+
// Nodes that fed sumBoundarySize's total: mirrors its recursion exactly so implicated stageIds
|
|
433
|
+
// line up with the size actually compared.
|
|
490
434
|
function boundarySizeContributors(node ) {
|
|
491
435
|
const m = (node.metrics ?? []).find((x) => x.name === 'data size');
|
|
492
436
|
if (m) return [node];
|
|
@@ -516,10 +460,8 @@ function maxMedianRatio(
|
|
|
516
460
|
|
|
517
461
|
|
|
518
462
|
|
|
519
|
-
// Machine-readable detector metadata for the evidence report
|
|
520
|
-
//
|
|
521
|
-
// portable report record exactly which detector + threshold set produced each
|
|
522
|
-
// finding, so evidence stays reproducible as detectors evolve.
|
|
463
|
+
// Machine-readable detector metadata for the evidence report (no `detect` closure), so a
|
|
464
|
+
// portable report records which detector + thresholds produced each finding.
|
|
523
465
|
export function detectorCatalog() {
|
|
524
466
|
return DETECTORS.map((d) => ({
|
|
525
467
|
type: d.type,
|
|
@@ -530,21 +472,11 @@ export function detectorCatalog() {
|
|
|
530
472
|
}));
|
|
531
473
|
}
|
|
532
474
|
|
|
533
|
-
// True task-duration skew ratio
|
|
534
|
-
//
|
|
535
|
-
//
|
|
536
|
-
//
|
|
537
|
-
//
|
|
538
|
-
//
|
|
539
|
-
// Fields widened to optional: the `skew` detector below only ever calls this
|
|
540
|
-
// with an already-finalized `DetectorStage` (all four always numeric by the
|
|
541
|
-
// time `analyze()` runs), but `cli/budgets.ts`'s `checkSkew` calls it
|
|
542
|
-
// directly against `AppModel.stages`: real `Stage` records for a stage that
|
|
543
|
-
// never received a `StageCompleted` event (e.g. an unfinished run) genuinely
|
|
544
|
-
// lack these fields (see `types.ts`'s `Stage`). The `as number` casts below
|
|
545
|
-
// preserve the original behavior byte-for-byte: dividing through an absent
|
|
546
|
-
// field still naturally produces `NaN` (as it always has for untyped JS
|
|
547
|
-
// callers), rather than adding a new guard that would change the result.
|
|
475
|
+
// True task-duration skew ratio: P95/median once enough tasks to trust P95, else max/median.
|
|
476
|
+
// Null when no measurable median (p50 === 0). Exported so cli/budgets.ts recomputes the same
|
|
477
|
+
// ratio rather than reading findings floored at ratioWarn.
|
|
478
|
+
// Fields optional: budgets.ts calls this against raw AppModel.stages, whose stages may lack
|
|
479
|
+
// these fields (unfinished run). The `as number` casts keep behavior: an absent field yields NaN.
|
|
548
480
|
export function computeSkewRatio(
|
|
549
481
|
stage ,
|
|
550
482
|
minTasksForP95 ,
|
|
@@ -556,18 +488,10 @@ export function computeSkewRatio(
|
|
|
556
488
|
: { ratio: (max ) / (p50 ), metric: 'max/median' };
|
|
557
489
|
}
|
|
558
490
|
|
|
559
|
-
// Absolute-magnitude floor
|
|
560
|
-
// a
|
|
561
|
-
//
|
|
562
|
-
//
|
|
563
|
-
// absolute waste is real in a run that only took seconds; a fixed-ms floor
|
|
564
|
-
// can't scale between those. Used by the `skew` and `straggler` entries
|
|
565
|
-
// below, each gated via `clippedWasteMs` (below) against the *same
|
|
566
|
-
// occupancy-clipped* wall-clock figure
|
|
567
|
-
// src/impact-estimator.ts displays as that finding's savings, not the raw
|
|
568
|
-
// pre-clip delta, which can stay large after clipping collapses the
|
|
569
|
-
// recoverable time to near zero (the stage's own longest task already
|
|
570
|
-
// accounts for nearly all of its wall-clock window).
|
|
491
|
+
// Absolute-magnitude floor as a % of app runtime, not a fixed ms constant: a skew/straggler
|
|
492
|
+
// ratio on a few ms is noise in an hours-long run but real in a seconds-long one; a fixed-ms
|
|
493
|
+
// floor can't scale. Used by skew/straggler, gated via clippedWasteMs against the same
|
|
494
|
+
// occupancy-clipped figure impact-estimator.ts displays as savings.
|
|
571
495
|
// NOT SOURCED: floor percentages are our own noise floor, unvalidated.
|
|
572
496
|
function computeAppDurationMs(ctx ) {
|
|
573
497
|
const app = ctx?.app;
|
|
@@ -576,34 +500,22 @@ function computeAppDurationMs(ctx ) {
|
|
|
576
500
|
return durationMs > 0 ? durationMs : null;
|
|
577
501
|
}
|
|
578
502
|
|
|
579
|
-
// Unknown app timing
|
|
580
|
-
// finding; it just skips the floor gate, preserving prior ratio-only
|
|
581
|
-
// behavior when total runtime can't be computed.
|
|
503
|
+
// Unknown app timing never suppresses a finding; it just skips the floor gate.
|
|
582
504
|
function meetsRuntimeFloor(wasteMs , appDurationMs , floorPct ) {
|
|
583
505
|
return appDurationMs == null || wasteMs >= appDurationMs * floorPct;
|
|
584
506
|
}
|
|
585
507
|
|
|
586
|
-
// Runs a raw waste delta through the same
|
|
587
|
-
//
|
|
588
|
-
//
|
|
589
|
-
// wall-clock time rather than a delta that a physical floor (the stage's own
|
|
590
|
-
// longest task) may leave almost entirely unrecoverable. Falls back to the
|
|
591
|
-
// raw delta when occupancy data isn't available for this stage (ctx omitted,
|
|
592
|
-
// or the stage was excluded from the occupancy sweep for having <= 0
|
|
593
|
-
// duration), same as the pre-existing ratio-only behavior for unknown app
|
|
594
|
-
// timing above.
|
|
508
|
+
// Runs a raw waste delta through the same occupancy clip impact-estimator.ts applies before
|
|
509
|
+
// display, so the runtime floor checks recoverable wall-clock, not a delta a physical floor
|
|
510
|
+
// leaves unrecoverable. Falls back to the raw delta when occupancy data is unavailable.
|
|
595
511
|
function clippedWasteMs(wasteMs , stageId , ctx ) {
|
|
596
512
|
if (!ctx) return wasteMs;
|
|
597
513
|
const est = estimateSingleStage(wasteMs, stageId, ctx.stages , ctx.occupancy);
|
|
598
514
|
return est ? est.wallClock.high : wasteMs;
|
|
599
515
|
}
|
|
600
516
|
|
|
601
|
-
// cacheUtilization's
|
|
602
|
-
//
|
|
603
|
-
// pickDominantReason above). Both variants share the same confidence/
|
|
604
|
-
// validationRequired text: the ratio is a point-in-time storage snapshot
|
|
605
|
-
// from stage-submission events (src/event-handlers.js's mergeStageRddInfo),
|
|
606
|
-
// not a runtime block-access read-count.
|
|
517
|
+
// Shared by cacheUtilization's two variants: the ratio is a point-in-time storage snapshot from
|
|
518
|
+
// stage-submission events, not a runtime block-access read-count.
|
|
607
519
|
const CACHE_UTILIZATION_VALIDATION =
|
|
608
520
|
"This ratio is a point-in-time storage snapshot from stage-submission events, not a runtime read-count. Confirm against the Spark UI's Storage tab before acting.";
|
|
609
521
|
|
|
@@ -638,19 +550,11 @@ function diskSpilloverFinding(rdd , diskRatio , impactBan
|
|
|
638
550
|
};
|
|
639
551
|
}
|
|
640
552
|
|
|
641
|
-
// Entry shape for every item
|
|
642
|
-
//
|
|
643
|
-
//
|
|
644
|
-
//
|
|
645
|
-
//
|
|
646
|
-
// the only real callers), and unifying those four into one `TTarget` would
|
|
647
|
-
// need either an unsound cast or a discriminated-union-of-detectors redesign
|
|
648
|
-
// this migration task doesn't ask for. Each entry below still gets a
|
|
649
|
-
// precisely-typed `detect` by annotating its own `target`/`ctx` parameters
|
|
650
|
-
// directly: object-literal methods (this `detect(target) {}` shorthand, not
|
|
651
|
-
// an arrow function assigned to a property) are checked bivariantly against
|
|
652
|
-
// an interface's method parameter types, so a narrower, concrete annotation
|
|
653
|
-
// here does not conflict with `Detector`'s `unknown` declaration.
|
|
553
|
+
// Entry shape for every DETECTORS item. TTarget stays `unknown` at the array level: detect's
|
|
554
|
+
// real first-arg varies by scope (DetectorStage/DetectorSqlExec/DetectorCtx/{ app }), and
|
|
555
|
+
// unifying them would need an unsound cast or a discriminated-union redesign. Each entry gets a
|
|
556
|
+
// precise detect by annotating its own params: object-literal method params are checked
|
|
557
|
+
// bivariantly, so a narrower annotation here doesn't conflict with the `unknown` declaration.
|
|
654
558
|
|
|
655
559
|
|
|
656
560
|
|
|
@@ -658,14 +562,11 @@ function diskSpilloverFinding(rdd , diskRatio , impactBan
|
|
|
658
562
|
|
|
659
563
|
|
|
660
564
|
|
|
661
|
-
|
|
662
|
-
|
|
663
|
-
|
|
664
|
-
|
|
665
|
-
|
|
565
|
+
|
|
666
566
|
|
|
667
567
|
|
|
668
568
|
|
|
569
|
+
|
|
669
570
|
|
|
670
571
|
|
|
671
572
|
|
|
@@ -678,9 +579,8 @@ function diskSpilloverFinding(rdd , diskRatio , impactBan
|
|
|
678
579
|
// *Ratio = multiplicative factor
|
|
679
580
|
// *Share/*Rate/*Util = 0–1 fraction (normalized)
|
|
680
581
|
|
|
681
|
-
//
|
|
682
|
-
//
|
|
683
|
-
// impact-band floor instead of hand-copying the literals.
|
|
582
|
+
// straggler's noise-floor thresholds (NOT SOURCED: unvalidated), exported so impact-band.ts
|
|
583
|
+
// reuses the same figures instead of hand-copying.
|
|
684
584
|
export const STRAGGLER_FLOOR_PCT_WARN = 0.005;
|
|
685
585
|
export const STRAGGLER_FLOOR_PCT_CRIT = 0.02;
|
|
686
586
|
|
|
@@ -700,9 +600,8 @@ export const DETECTORS = [
|
|
|
700
600
|
if (result === null) return null;
|
|
701
601
|
const { ratio, metric } = result;
|
|
702
602
|
if (ratio <= this.thresholds.ratioWarn) return null;
|
|
703
|
-
// Same absolute delta
|
|
704
|
-
//
|
|
705
|
-
// so the gate agrees with what's actually displayed.
|
|
603
|
+
// Same absolute delta impact-estimator.ts's 'skew' case reports as savings; clipped the
|
|
604
|
+
// same way before the floor check so the gate agrees with what's displayed.
|
|
706
605
|
const wasteMs = Math.max(0, metric === 'P95/median' ? stage.taskDurationP95 - stage.taskDurationP50 : stage.taskDurationMax - stage.taskDurationP50);
|
|
707
606
|
const appDurationMs = computeAppDurationMs(ctx);
|
|
708
607
|
const floorWasteMs = clippedWasteMs(wasteMs, stage.id, ctx);
|
|
@@ -712,6 +611,8 @@ export const DETECTORS = [
|
|
|
712
611
|
type: 'skew', stageId: stage.id,
|
|
713
612
|
impactBand: 'warning',
|
|
714
613
|
metric, value,
|
|
614
|
+
confidence: 'low',
|
|
615
|
+
validationRequired: 'The 0.5% runtime-floor percentage that gates this finding is our own noise floor, unvalidated: no external tool publishes an equivalent metric to calibrate against. Confirm against known-good/known-bad real logs before trusting the impact-band split.',
|
|
715
616
|
recommendation: `Task duration ratio (${metric}) is ${value}×: for join-driven skew, enable AQE skew-join handling (spark.sql.adaptive.skewJoin.enabled); otherwise salt the key or repartition on a better key to reduce task skew.`,
|
|
716
617
|
};
|
|
717
618
|
},
|
|
@@ -753,11 +654,9 @@ export const DETECTORS = [
|
|
|
753
654
|
});
|
|
754
655
|
}
|
|
755
656
|
}
|
|
756
|
-
// TaskStageSkew: straggler cost vs stage wall-clock. Skip near-zero duration.
|
|
757
|
-
//
|
|
758
|
-
//
|
|
759
|
-
// exactly zero on every firing (see impact-estimator.ts's costOnly branch below),
|
|
760
|
-
// so there is no wall-clock-backed impact-band tier left to gate on.
|
|
657
|
+
// TaskStageSkew: straggler cost vs stage wall-clock. Skip near-zero duration. Always info
|
|
658
|
+
// like its siblings: this trigger forces the occupancy-clipped estimate to exactly zero on
|
|
659
|
+
// every firing, so there's no wall-clock-backed tier left to gate on.
|
|
761
660
|
const stageDurationMs = (stage.completedAt ?? 0) - (stage.submittedAt ?? 0);
|
|
762
661
|
if (stageDurationMs > 0) {
|
|
763
662
|
const ratio = stage.taskDurationMax / stageDurationMs;
|
|
@@ -808,9 +707,8 @@ export const DETECTORS = [
|
|
|
808
707
|
const out = [];
|
|
809
708
|
const { shuffleReadP50: p50, shuffleReadMax: max, shuffleReadBytes: total, taskCount } = stage;
|
|
810
709
|
if (max > this.thresholds.skewRatio * p50 && max > this.thresholds.skewFloorBytes) {
|
|
811
|
-
// p50 can be 0 (
|
|
812
|
-
//
|
|
813
|
-
// back to median-free phrasing instead of dividing by p50.
|
|
710
|
+
// p50 can be 0 (over half the shuffle partitions empty): a ratio against zero renders
|
|
711
|
+
// "Infinity×", so fall back to median-free phrasing.
|
|
814
712
|
const ratioText = p50 > 0
|
|
815
713
|
? `${Math.round(max / p50 * 10) / 10}× the median (${formatBytes(p50)})`
|
|
816
714
|
: `far larger than the median (${formatBytes(p50)}, effectively empty)`;
|
|
@@ -828,6 +726,10 @@ export const DETECTORS = [
|
|
|
828
726
|
});
|
|
829
727
|
}
|
|
830
728
|
if (max >= this.thresholds.maxPartBytes) {
|
|
729
|
+
// Fixed 'critical': an OOM/crash-risk safety signal, not a time-waste one. Exempted in
|
|
730
|
+
// impact-band.ts's deriveImpactBand from the wall-clock-based overwrite every other
|
|
731
|
+
// finding here gets, so a long-running job can't demote an active crash risk to 'info'
|
|
732
|
+
// just because the modeled time savings are a small fraction of total runtime.
|
|
831
733
|
out.push({
|
|
832
734
|
type: 'partitionSizing', stageId: stage.id, impactBand: 'critical',
|
|
833
735
|
rule: 'maxPartitionTooBig', metric: 'shuffleReadMax', value: max,
|
|
@@ -864,18 +766,19 @@ export const DETECTORS = [
|
|
|
864
766
|
{
|
|
865
767
|
type: 'gc', scope: 'stage', order: 50, fixEffort: 'config', version: 1,
|
|
866
768
|
docAnchor: '#bottleneck-gc',
|
|
769
|
+
confidence: 'low',
|
|
770
|
+
validationRequired: 'The 10-second minimum-runtime floor that gates this finding is our own noise floor, unvalidated: no external tool publishes an equivalent metric to calibrate against. Confirm against known-good/known-bad real logs before trusting the impact-band split.',
|
|
867
771
|
thresholds: {
|
|
868
772
|
warnPct100: 10,
|
|
869
|
-
// Descending tier:
|
|
773
|
+
// Descending tier: ExecutorGcHeuristic, ported as-is.
|
|
870
774
|
lowInfoPct100: 5,
|
|
871
|
-
// NOT SOURCED: our own noise floor so a stage that barely ran
|
|
872
|
-
// near 0 or wildly inflated by a tiny denominator) does not flag,
|
|
873
|
-
// in either direction.
|
|
775
|
+
// NOT SOURCED: our own noise floor so a stage that barely ran doesn't flag either direction.
|
|
874
776
|
minRunTimeMs: 10000,
|
|
875
777
|
},
|
|
876
778
|
detect(
|
|
877
779
|
|
|
878
780
|
|
|
781
|
+
|
|
879
782
|
|
|
880
783
|
stage ,
|
|
881
784
|
) {
|
|
@@ -887,6 +790,7 @@ export const DETECTORS = [
|
|
|
887
790
|
type: 'gc', stageId: stage.id,
|
|
888
791
|
impactBand: 'warning',
|
|
889
792
|
metric: 'gcPct', value,
|
|
793
|
+
confidence: this.confidence, validationRequired: this.validationRequired,
|
|
890
794
|
recommendation: `GC consumed ${value}% of executor run time: reduce object creation, use primitive types, avoid UDFs, increase executor memory.`,
|
|
891
795
|
};
|
|
892
796
|
}
|
|
@@ -898,6 +802,7 @@ export const DETECTORS = [
|
|
|
898
802
|
type: 'gc', stageId: stage.id, direction: 'low',
|
|
899
803
|
impactBand: 'info',
|
|
900
804
|
metric: 'gcPct', value,
|
|
805
|
+
confidence: this.confidence, validationRequired: this.validationRequired,
|
|
901
806
|
recommendation: `GC consumed only ${value}% of executor run time: memory may be over-provisioned; consider reducing spark.executor.memory for cost savings.`,
|
|
902
807
|
};
|
|
903
808
|
}
|
|
@@ -909,12 +814,9 @@ export const DETECTORS = [
|
|
|
909
814
|
docAnchor: '#bottleneck-slow-host',
|
|
910
815
|
thresholds: {
|
|
911
816
|
minHosts: 3, minTasks: 15, ratioWarn: 2.0, minShare: 0.20, shareWarn: 0.75, taskShareWarn: 0.50, ratioTiers: [1.33, 1.78, 3.16, 10],
|
|
912
|
-
// Absolute-magnitude floors (mirrors computeSpillMagnitude's ratio+floor
|
|
913
|
-
//
|
|
914
|
-
//
|
|
915
|
-
// real slow-host problem. 1s is well above typical per-task scheduling
|
|
916
|
-
// jitter but well below the tens-of-seconds+ means genuine slow-host
|
|
917
|
-
// stages exhibit; 64MB mirrors the spill detector's disk-skew floor.
|
|
817
|
+
// Absolute-magnitude floors (mirrors computeSpillMagnitude's ratio+floor pattern): on short
|
|
818
|
+
// stages, sub-second/sub-64MB host differences produce huge noise ratios. 1s is above
|
|
819
|
+
// per-task jitter but below genuine slow-host stages; 64MB mirrors the spill disk-skew floor.
|
|
918
820
|
floorMs: 1000, floorBytes: 64 * MB,
|
|
919
821
|
},
|
|
920
822
|
detect(
|
|
@@ -946,7 +848,7 @@ export const DETECTORS = [
|
|
|
946
848
|
// `value` is a ratio; the estimator needs the absolute per-host mean.
|
|
947
849
|
hostMeanMs: h.mean,
|
|
948
850
|
host: h.host, hostTaskShare: Math.round(share * 100) / 100,
|
|
949
|
-
recommendation:
|
|
851
|
+
recommendation: `${h.host} may just hold data locality for its tasks or carry one heavy stage, not necessarily a hardware fault: check what it was running, and consider enabling spark.speculation to relaunch a lagging task automatically.`,
|
|
950
852
|
});
|
|
951
853
|
}
|
|
952
854
|
}
|
|
@@ -1012,10 +914,8 @@ export const DETECTORS = [
|
|
|
1012
914
|
type: 'stageSlowness', scope: 'stage', order: 65, fixEffort: 'code', version: 2,
|
|
1013
915
|
docAnchor: '#bottleneck-stage-slowness',
|
|
1014
916
|
thresholds: { infoMin: 15 },
|
|
1015
|
-
// Cross-detector suppression (
|
|
1016
|
-
//
|
|
1017
|
-
// Requires this entry to be declared AFTER slowHost in DETECTORS so
|
|
1018
|
-
// slowHost findings are already in `out`.
|
|
917
|
+
// Cross-detector suppression (see "Detector contract" in detector-contract.md). Requires this
|
|
918
|
+
// entry to be declared AFTER slowHost in DETECTORS so slowHost findings are already in `out`.
|
|
1019
919
|
suppressWhen(finding, out) {
|
|
1020
920
|
return out.some(o => o.type === 'slowHost' && o.stageId === finding.stageId);
|
|
1021
921
|
},
|
|
@@ -1023,9 +923,8 @@ export const DETECTORS = [
|
|
|
1023
923
|
|
|
1024
924
|
stage ,
|
|
1025
925
|
) {
|
|
1026
|
-
// Basis is real wall-clock stage duration, not per-executor average
|
|
1027
|
-
//
|
|
1028
|
-
// stageDurationMs computation.
|
|
926
|
+
// Basis is real wall-clock stage duration, not per-executor average; the impact-estimator
|
|
927
|
+
// formula reuses this exact stageDurationMs computation.
|
|
1029
928
|
const stageDurationMs = (stage.completedAt ?? 0) - (stage.submittedAt ?? 0);
|
|
1030
929
|
if (!(stageDurationMs > 0)) return null;
|
|
1031
930
|
const durationMinutes = stageDurationMs / 60000;
|
|
@@ -1036,7 +935,7 @@ export const DETECTORS = [
|
|
|
1036
935
|
return {
|
|
1037
936
|
type: 'stageSlowness', stageId: stage.id, impactBand,
|
|
1038
937
|
metric: 'stageDurationMinutes', value,
|
|
1039
|
-
recommendation: `This stage ran ${value} minutes with no more specific cause flagged:
|
|
938
|
+
recommendation: `This stage ran ${value} minutes with no more specific cause flagged: often a partition-count problem, raise parallelism via spark.sql.shuffle.partitions or spark.default.parallelism, or check for a large per-task data volume driving heavy shuffle and spill.`,
|
|
1040
939
|
};
|
|
1041
940
|
},
|
|
1042
941
|
},
|
|
@@ -1050,6 +949,9 @@ export const DETECTORS = [
|
|
|
1050
949
|
type: 'stageFailed', stageId: stage.id, impactBand: 'critical',
|
|
1051
950
|
variant: 'stageFailure',
|
|
1052
951
|
metric: 'stageFailureReason', value: stage.stageFailureReason,
|
|
952
|
+
numTasks: stage.taskCount,
|
|
953
|
+
memoryBytesSpilled: stage.memoryBytesSpilled,
|
|
954
|
+
failedTaskDetails: stage.failedTaskSamples ?? [],
|
|
1053
955
|
recommendation: `This stage attempt failed outright. Inspect the driver log for the failure reason and the job that triggered it.`,
|
|
1054
956
|
};
|
|
1055
957
|
},
|
|
@@ -1081,9 +983,8 @@ export const DETECTORS = [
|
|
|
1081
983
|
{
|
|
1082
984
|
type: 'straggler', scope: 'stage', order: 70, fixEffort: 'code', version: 1,
|
|
1083
985
|
docAnchor: '#bottleneck-straggler',
|
|
1084
|
-
// floorPctWarn/floorPctCrit are re-exported as STRAGGLER_FLOOR_PCT_WARN/CRIT
|
|
1085
|
-
//
|
|
1086
|
-
// in sync, don't hand-edit one without the other.
|
|
986
|
+
// floorPctWarn/floorPctCrit are re-exported as STRAGGLER_FLOOR_PCT_WARN/CRIT and reused as
|
|
987
|
+
// impact-band.ts's global noise floor: keep the two in sync.
|
|
1087
988
|
thresholds: { minTasks: 10, shareWarn: 0.05, warnPct: 0.10, critPct: 0.20, floorPctWarn: STRAGGLER_FLOOR_PCT_WARN, floorPctCrit: STRAGGLER_FLOOR_PCT_CRIT },
|
|
1088
989
|
detect(
|
|
1089
990
|
|
|
@@ -1097,12 +998,9 @@ export const DETECTORS = [
|
|
|
1097
998
|
if ((stage.speculativeTasks ?? 0) === 0 && stragglerShare <= this.thresholds.shareWarn) return null;
|
|
1098
999
|
const useSpeculative = (stage.speculativeTasks ?? 0) > 0;
|
|
1099
1000
|
const speculativeShare = useSpeculative ? stage.speculativeTasks / stage.taskCount : 0;
|
|
1100
|
-
// Same absolute delta
|
|
1101
|
-
//
|
|
1102
|
-
//
|
|
1103
|
-
// duration models near-zero savings, so it must not outrank 'info'.
|
|
1104
|
-
// Clipped the same way before the floor check so the gate agrees with
|
|
1105
|
-
// what's actually displayed.
|
|
1001
|
+
// Same absolute delta impact-estimator.ts's straggler/stageShape case reports as savings: a
|
|
1002
|
+
// high straggler/speculative share on a stage whose tasks barely vary models near-zero
|
|
1003
|
+
// savings, so it must not outrank 'info'. Clipped the same way before the floor check.
|
|
1106
1004
|
const wasteMs = Math.max(0, stage.taskDurationMax - stage.taskDurationP50);
|
|
1107
1005
|
const appDurationMs = computeAppDurationMs(ctx);
|
|
1108
1006
|
const floorWasteMs = clippedWasteMs(wasteMs, stage.id, ctx);
|
|
@@ -1110,21 +1008,14 @@ export const DETECTORS = [
|
|
|
1110
1008
|
const meetsCritFloor = meetsRuntimeFloor(floorWasteMs, appDurationMs, this.thresholds.floorPctCrit);
|
|
1111
1009
|
const speculativeTier = speculativeShare >= this.thresholds.critPct && meetsCritFloor ? 'critical'
|
|
1112
1010
|
: speculativeShare >= this.thresholds.warnPct && meetsWarnFloor ? 'warning' : 'info';
|
|
1113
|
-
// Straggler share has no dedicated critical tier per
|
|
1114
|
-
// docs-site/contributor-guide/architecture/detector-contract.md; it can only push to warning.
|
|
1011
|
+
// Straggler share has no dedicated critical tier per detector-contract.md; only warning.
|
|
1115
1012
|
const stragglerTier = stragglerShare > this.thresholds.shareWarn && meetsWarnFloor ? 'warning' : 'info';
|
|
1116
|
-
// Fixed fallback: overwritten by deriveImpactBand
|
|
1117
|
-
//
|
|
1118
|
-
// surfaces on the rare miss (stage excluded from the occupancy sweep).
|
|
1013
|
+
// Fixed fallback: overwritten by deriveImpactBand when this finding gets a real wallClock
|
|
1014
|
+
// estimate (the common case). Only surfaces on the rare occupancy-sweep miss.
|
|
1119
1015
|
const impactBand = 'info';
|
|
1120
|
-
// Report whichever signal actually drove the finding, not just whether
|
|
1121
|
-
// speculative
|
|
1122
|
-
//
|
|
1123
|
-
// speculativeTasks count (real-log bug: a 50%-straggler-share stage
|
|
1124
|
-
// with 1 speculative task reported an impact band of 'warning' but metric
|
|
1125
|
-
// 'speculativeTasks: 1', hiding the actual cause). Ties keep the prior
|
|
1126
|
-
// default (speculative-driven) so existing speculative-only findings
|
|
1127
|
-
// are unaffected.
|
|
1016
|
+
// Report whichever signal actually drove the finding, not just whether speculation was on:
|
|
1017
|
+
// a high stragglerShare with few speculative retries must not be reported as a low-value
|
|
1018
|
+
// speculativeTasks count. Ties keep the speculative-driven default.
|
|
1128
1019
|
const useSpeculativeMetric = useSpeculative && !(IMPACT_BAND_ORDER[stragglerTier] < IMPACT_BAND_ORDER[speculativeTier]);
|
|
1129
1020
|
const value = useSpeculativeMetric ? stage.speculativeTasks : Math.round(stragglerShare * 100);
|
|
1130
1021
|
const detail = useSpeculativeMetric
|
|
@@ -1137,7 +1028,9 @@ export const DETECTORS = [
|
|
|
1137
1028
|
unit: useSpeculativeMetric ? 'count' : 'pct',
|
|
1138
1029
|
speculativeTasks: stage.speculativeTasks ?? 0,
|
|
1139
1030
|
stragglerCount: stage.stragglerCount ?? 0,
|
|
1140
|
-
|
|
1031
|
+
confidence: 'low',
|
|
1032
|
+
validationRequired: 'The 0.5%/2% runtime-floor percentages that gate this finding are our own noise floor, unvalidated: no external tool publishes an equivalent metric to calibrate against. Confirm against known-good/known-bad real logs before trusting the impact-band split.',
|
|
1033
|
+
recommendation: `${detail}: rule out a GC pause or a slow shuffle fetch before assuming a hardware issue; if a skewed key is the real cause, that's a candidate for AQE's skew-join handling.`,
|
|
1141
1034
|
};
|
|
1142
1035
|
},
|
|
1143
1036
|
},
|
|
@@ -1179,6 +1072,9 @@ export const DETECTORS = [
|
|
|
1179
1072
|
type: 'retryWaste', stageId: stage.id,
|
|
1180
1073
|
impactBand: 'warning',
|
|
1181
1074
|
metric: 'retryWasteMs', value: wastedMs,
|
|
1075
|
+
numTasks: stage.taskCount,
|
|
1076
|
+
memoryBytesSpilled: stage.memoryBytesSpilled,
|
|
1077
|
+
retriedTaskDetails: stage.retryTaskSamples ?? [],
|
|
1182
1078
|
recommendation: `Retried task attempts wasted ${Math.round(wastedMs / 1000)}s of executor time (${wasted} attempt${wasted === 1 ? '' : 's'}) even though the stage completed: investigate executor loss or fetch failures.`,
|
|
1183
1079
|
extended: `${wasted} task attempts were superseded by a later retry, wasting ${Math.round(wastedMs / 1000)}s of executor time. Common causes: executor loss (OOM-kill, node death) or shuffle FetchFailed forcing a stage-map recompute. Check driver logs for the dominant reason (see the Failures widget) even if the final failure rate looks low; retries hide the true cost.`,
|
|
1184
1080
|
};
|
|
@@ -1206,14 +1102,13 @@ export const DETECTORS = [
|
|
|
1206
1102
|
},
|
|
1207
1103
|
},
|
|
1208
1104
|
{
|
|
1209
|
-
// No docAnchor
|
|
1210
|
-
//
|
|
1211
|
-
// section for this tool-specific "capture stopped early" signal.
|
|
1105
|
+
// No docAnchor: the upstream spark-tuning-reference docs have no section for this
|
|
1106
|
+
// tool-specific "capture stopped early" signal.
|
|
1212
1107
|
type: 'incompleteRun', scope: 'app', order: 5, fixEffort: 'code', version: 1,
|
|
1213
1108
|
thresholds: {},
|
|
1214
1109
|
recommendation: 'This event log never recorded an ApplicationEnd event: the capture stopped before the run finished (an in-flight job, a rotated-away log, or a cut-short capture). Findings and metrics elsewhere on this board reflect only what was captured up to that point, not the full run.',
|
|
1215
1110
|
detect( ctx ) {
|
|
1216
|
-
if (ctx.app.startTime == null || ctx.app.endTime != null) return null;
|
|
1111
|
+
if (!ctx.app || ctx.app.startTime == null || ctx.app.endTime != null) return null;
|
|
1217
1112
|
return {
|
|
1218
1113
|
type: 'incompleteRun', stageId: null, impactBand: 'warning',
|
|
1219
1114
|
metric: 'applicationEnd', value: 'missing',
|
|
@@ -1230,18 +1125,22 @@ export const DETECTORS = [
|
|
|
1230
1125
|
ctx ,
|
|
1231
1126
|
) {
|
|
1232
1127
|
const { app, stages } = ctx;
|
|
1233
|
-
|
|
1128
|
+
// Nullish (not falsy) check: a literal startTime:0 must not be treated as "missing".
|
|
1129
|
+
if (!app || app.startTime == null || stages.size === 0) return null;
|
|
1234
1130
|
let firstTaskLaunch = Infinity;
|
|
1235
1131
|
for (const stage of stages.values()) {
|
|
1236
1132
|
if (stage.submittedAt > 0 && stage.submittedAt < firstTaskLaunch) firstTaskLaunch = stage.submittedAt;
|
|
1237
1133
|
}
|
|
1134
|
+
// No stage ever recorded a submission timestamp: no basis to measure a startup gap against.
|
|
1135
|
+
// Exposed now that a literal app.startTime:0 no longer short-circuits this detector entirely.
|
|
1136
|
+
if (!Number.isFinite(firstTaskLaunch)) return null;
|
|
1238
1137
|
const gapSeconds = (firstTaskLaunch - app.startTime) / 1000;
|
|
1239
1138
|
if (gapSeconds <= this.thresholds.gapSeconds) return null;
|
|
1240
1139
|
const value = Math.round(gapSeconds);
|
|
1241
1140
|
return {
|
|
1242
1141
|
type: 'coldStart', stageId: null, impactBand: 'warning',
|
|
1243
1142
|
metric: 'startupGapSeconds', value,
|
|
1244
|
-
recommendation: `
|
|
1143
|
+
recommendation: `The first task waited ${value}s for executors to become available: keep a warm pool of idle executors, or if using dynamic allocation, raise the minimum/initial executor count so it doesn't scale up from zero.`,
|
|
1245
1144
|
};
|
|
1246
1145
|
},
|
|
1247
1146
|
},
|
|
@@ -1253,25 +1152,27 @@ export const DETECTORS = [
|
|
|
1253
1152
|
|
|
1254
1153
|
ctx ,
|
|
1255
1154
|
) {
|
|
1256
|
-
const { app, executorsAdded, executorsRemoved } = ctx;
|
|
1257
|
-
|
|
1155
|
+
const { app, executorsAdded, executorsRemoved, runAggregates } = ctx;
|
|
1156
|
+
// Nullish (not falsy) check: a literal startTime:0 must not be treated as "missing".
|
|
1157
|
+
if (!app || executorsAdded.length === 0 || app.startTime == null || app.endTime == null) return null;
|
|
1258
1158
|
const appDuration = app.endTime - app.startTime;
|
|
1259
1159
|
if (appDuration <= 0) return null;
|
|
1260
|
-
|
|
1261
|
-
|
|
1262
|
-
|
|
1263
|
-
|
|
1264
|
-
|
|
1265
|
-
|
|
1266
|
-
|
|
1267
|
-
|
|
1268
|
-
|
|
1269
|
-
|
|
1160
|
+
// computePeakConcurrentCores (not executorsAdded.length/computeTotalCores): real concurrent
|
|
1161
|
+
// capacity, not a cumulative sum that double-counts a churned-through executor against its
|
|
1162
|
+
// replacement's (spot preemption, dynamicAllocation replacement).
|
|
1163
|
+
const totalCores = computePeakConcurrentCores(app, executorsAdded, executorsRemoved);
|
|
1164
|
+
if (totalCores <= 0) return null;
|
|
1165
|
+
const capacityCoreMs = totalCores * appDuration;
|
|
1166
|
+
// Busy core-time (from the whole-run core-time-series, same signal memoryUtilization's
|
|
1167
|
+
// idleCores variant already uses), not executor lifetime: an executor that exists for the
|
|
1168
|
+
// whole run but sits fully idle must not score as 100% used. Missing runAggregates (older
|
|
1169
|
+
// callers, synthetic fixtures) reads as 0 busy time rather than falling back to the
|
|
1170
|
+
// lifetime-based measure this replaces.
|
|
1171
|
+
const busyCoreMs = runAggregates?.busyCoreMs ?? 0;
|
|
1172
|
+
const utilization = busyCoreMs / capacityCoreMs;
|
|
1270
1173
|
if (utilization >= this.thresholds.minUtil) return null;
|
|
1271
1174
|
|
|
1272
1175
|
// CPU-time-based utilization (sparkMeasure): metric only, no threshold.
|
|
1273
|
-
// Total cores: prefer executor-added Total Cores (real), else config cores.
|
|
1274
|
-
const totalCores = computeTotalCores(app, executorsAdded);
|
|
1275
1176
|
let cpuUtilizationPct = null;
|
|
1276
1177
|
if (totalCores > 0) {
|
|
1277
1178
|
let cpuMs = 0;
|
|
@@ -1296,10 +1197,10 @@ export const DETECTORS = [
|
|
|
1296
1197
|
type: 'memoryUtilization', scope: 'app', order: 102, fixEffort: 'config', version: 1,
|
|
1297
1198
|
docAnchor: '#bottleneck-memory-utilization',
|
|
1298
1199
|
thresholds: {
|
|
1299
|
-
idleCoreWarn: 0.50, //
|
|
1300
|
-
bandTooSmall: 0.95, //
|
|
1200
|
+
idleCoreWarn: 0.50, // WastedCoresAlertsReducer
|
|
1201
|
+
bandTooSmall: 0.95, // MemoryAlertsReducer: used/allocated
|
|
1301
1202
|
bandTooHigh: 0.70, // below this => over-provisioned (cost signal)
|
|
1302
|
-
wasteBufferMultiplier: 1.5, //
|
|
1203
|
+
wasteBufferMultiplier: 1.5, // UNVERIFIED
|
|
1303
1204
|
},
|
|
1304
1205
|
detect(
|
|
1305
1206
|
|
|
@@ -1309,19 +1210,20 @@ export const DETECTORS = [
|
|
|
1309
1210
|
|
|
1310
1211
|
ctx ,
|
|
1311
1212
|
) {
|
|
1312
|
-
const { app, executorsAdded, runAggregates, stages } = ctx;
|
|
1213
|
+
const { app, executorsAdded, executorsRemoved, runAggregates, stages } = ctx;
|
|
1313
1214
|
const out = [];
|
|
1314
|
-
// Nullish (not falsy) check:
|
|
1315
|
-
// detectors, a literal startTime:0 must not be treated as "missing".
|
|
1215
|
+
// Nullish (not falsy) check: a literal startTime:0 must not be treated as "missing".
|
|
1316
1216
|
if (app?.startTime == null || app?.endTime == null) return out;
|
|
1317
1217
|
const appDurationMs = app.endTime - app.startTime;
|
|
1318
1218
|
if (appDurationMs <= 0) return out;
|
|
1319
1219
|
|
|
1320
|
-
//
|
|
1321
|
-
|
|
1322
|
-
|
|
1323
|
-
|
|
1324
|
-
|
|
1220
|
+
// Peak concurrent executors/cores (not executorsAdded.length/computeTotalCores): a
|
|
1221
|
+
// cumulative sum or count double-counts a churned-through executor against its replacement's
|
|
1222
|
+
// (spot preemption, dynamicAllocation replacement), inflating idle-rate and waste-model figures.
|
|
1223
|
+
const peakExecutors = computePeakConcurrentExecutorCount(executorsAdded, executorsRemoved);
|
|
1224
|
+
const totalCores = computePeakConcurrentCores(app, executorsAdded, executorsRemoved);
|
|
1225
|
+
// Hoisted above 1a (also 1b/1c's input) so the idle-cores finding carries the allocated
|
|
1226
|
+
// memory its MB-seconds estimate needs.
|
|
1325
1227
|
const allocatedMB = app.resources?.executor?.memoryMB ?? null;
|
|
1326
1228
|
|
|
1327
1229
|
// ── 1a idle-cores rate ────────────────────────────────────────────────
|
|
@@ -1333,8 +1235,7 @@ export const DETECTORS = [
|
|
|
1333
1235
|
out.push({
|
|
1334
1236
|
type: 'memoryUtilization', variant: 'idleCores', stageId: null,
|
|
1335
1237
|
impactBand: 'warning', metric: 'idleCoreRate', value,
|
|
1336
|
-
// Raw (unrounded) rate plus
|
|
1337
|
-
// wasted-MB-seconds model: `value` above is a rounded percentage.
|
|
1238
|
+
// Raw (unrounded) rate plus sizing inputs for the impact estimator: `value` is rounded pct.
|
|
1338
1239
|
idleRateFraction: idleRate, allocatedMB, peakExecutors, appDurationMs,
|
|
1339
1240
|
recommendation: `${value}% of allocated core-time ran no task: reduce cluster size or enable dynamic allocation.`,
|
|
1340
1241
|
});
|
|
@@ -1362,10 +1263,8 @@ export const DETECTORS = [
|
|
|
1362
1263
|
const allocatedBytes = allocatedMB * 1024 * 1024;
|
|
1363
1264
|
for (const [execId, heap] of peakHeapByExec) {
|
|
1364
1265
|
const ratio = heap / allocatedBytes;
|
|
1365
|
-
// The two bands are opposite signals
|
|
1366
|
-
//
|
|
1367
|
-
// OOM-risk case from the over-provisioning waste case without re-deriving
|
|
1368
|
-
// the ratio against the thresholds.
|
|
1266
|
+
// The two bands are opposite signals: an explicit `rule` discriminator lets consumers
|
|
1267
|
+
// tell OOM-risk from over-provisioning without re-deriving the ratio.
|
|
1369
1268
|
if (ratio > this.thresholds.bandTooSmall) {
|
|
1370
1269
|
out.push({
|
|
1371
1270
|
type: 'memoryUtilization', variant: 'memoryBand', rule: 'heapNearCapacity',
|
|
@@ -1387,7 +1286,7 @@ export const DETECTORS = [
|
|
|
1387
1286
|
}
|
|
1388
1287
|
}
|
|
1389
1288
|
|
|
1390
|
-
// ── 1c Spark Memory Limit waste model (
|
|
1289
|
+
// ── 1c Spark Memory Limit waste model (UNVERIFIED buffer) ─
|
|
1391
1290
|
if (allocatedMB != null && peakExecutors > 0) {
|
|
1392
1291
|
const allocatedMBSeconds = peakExecutors * allocatedMB * (appDurationMs / 1000);
|
|
1393
1292
|
let usedRunTimeMs = 0;
|
|
@@ -1400,7 +1299,7 @@ export const DETECTORS = [
|
|
|
1400
1299
|
type: 'memoryUtilization', variant: 'wasteModel', stageId: null,
|
|
1401
1300
|
impactBand: 'info', metric: 'wastedMBSeconds', value,
|
|
1402
1301
|
confidence: 'low',
|
|
1403
|
-
validationRequired: 'Memory-waste estimate uses allocated-vs-used memory-time and an unverified 1.5x buffer
|
|
1302
|
+
validationRequired: 'Memory-waste estimate uses allocated-vs-used memory-time and an unverified 1.5x buffer: confirm against the Spark UI before acting.',
|
|
1404
1303
|
recommendation: `Allocated executor memory sat largely idle over the run (~${value.toLocaleString('en-US')} MB-seconds wasted): review spark.executor.memory and executor count.`,
|
|
1405
1304
|
});
|
|
1406
1305
|
}
|
|
@@ -1410,13 +1309,9 @@ export const DETECTORS = [
|
|
|
1410
1309
|
},
|
|
1411
1310
|
},
|
|
1412
1311
|
{
|
|
1413
|
-
// Per-RDD cache-utilization proxies (this repo's own design: Spark
|
|
1414
|
-
//
|
|
1415
|
-
//
|
|
1416
|
-
// #85). Two independent, per-RDD tiered checks over `ctx.app.rddInfo`
|
|
1417
|
-
// storage snapshots: partial caching (numCachedPartitions < numPartitions)
|
|
1418
|
-
// and disk spillover (diskSize share of a MEMORY_AND_DISK*-requesting
|
|
1419
|
-
// RDD's cached footprint). An RDD can produce both findings in one pass.
|
|
1312
|
+
// Per-RDD cache-utilization proxies (this repo's own design: Spark event logs carry no
|
|
1313
|
+
// block-access events, so a literal cache hit rate isn't derivable). Two per-RDD tiered
|
|
1314
|
+
// checks over rddInfo snapshots: partial caching and disk spillover. An RDD can produce both.
|
|
1420
1315
|
type: 'cacheUtilization', scope: 'app', order: 103, fixEffort: 'code', version: 1,
|
|
1421
1316
|
docAnchor: '#memory-model',
|
|
1422
1317
|
thresholds: {
|
|
@@ -1458,11 +1353,9 @@ export const DETECTORS = [
|
|
|
1458
1353
|
},
|
|
1459
1354
|
},
|
|
1460
1355
|
{
|
|
1461
|
-
// Non-local task ratio across stage.localityStats (RACK_LOCAL + ANY vs
|
|
1462
|
-
//
|
|
1463
|
-
// the
|
|
1464
|
-
// variant above. NO_PREF stays in the denominator only: it's what
|
|
1465
|
-
// shuffle-read stages legitimately report with no locality problem.
|
|
1356
|
+
// Non-local task ratio across stage.localityStats (RACK_LOCAL + ANY vs all tasks), the other
|
|
1357
|
+
// half of the "Wasted Cores Ratio" (idle-core half is memoryUtilization's idleCores).
|
|
1358
|
+
// NO_PREF stays in the denominator only: shuffle-read stages legitimately report it.
|
|
1466
1359
|
type: 'coreLocality', scope: 'app', order: 103, fixEffort: 'config', version: 1,
|
|
1467
1360
|
docAnchor: '#bottleneck-utilization',
|
|
1468
1361
|
thresholds: { minTasks: 50, warnRatio: 0.15, critRatio: 0.35 },
|
|
@@ -1472,11 +1365,8 @@ export const DETECTORS = [
|
|
|
1472
1365
|
) {
|
|
1473
1366
|
const { totalTasks, nonLocalTasks, ratio } = computeCoreLocalityRatio([...ctx.stages.values()]);
|
|
1474
1367
|
if (totalTasks == null || totalTasks < this.thresholds.minTasks) return null;
|
|
1475
|
-
// computeCoreLocalityRatio
|
|
1476
|
-
//
|
|
1477
|
-
// both at once); the `totalTasks == null` guard above already rules that
|
|
1478
|
-
// out, so `ratio` is guaranteed non-null here even though the function's
|
|
1479
|
-
// declared return type keeps the two nullable independently.
|
|
1368
|
+
// computeCoreLocalityRatio only returns ratio:null together with totalTasks:null (shared
|
|
1369
|
+
// EMPTY sentinel); the totalTasks guard above rules that out, so ratio is non-null here.
|
|
1480
1370
|
if (ratio < this.thresholds.warnRatio) return null;
|
|
1481
1371
|
|
|
1482
1372
|
const value = Math.round(ratio * 100);
|
|
@@ -1484,8 +1374,7 @@ export const DETECTORS = [
|
|
|
1484
1374
|
type: 'coreLocality', stageId: null,
|
|
1485
1375
|
impactBand: ratio >= this.thresholds.critRatio ? 'critical' : 'warning',
|
|
1486
1376
|
metric: 'nonLocalRatio', value,
|
|
1487
|
-
// Raw count behind the ratio, for the impact estimator
|
|
1488
|
-
// figure. Non-null whenever totalTasks is (both come from the same EMPTY sentinel).
|
|
1377
|
+
// Raw count behind the ratio, for the impact estimator. Non-null whenever totalTasks is.
|
|
1489
1378
|
nonLocalTaskCount: nonLocalTasks ,
|
|
1490
1379
|
confidence: 'low',
|
|
1491
1380
|
validationRequired: 'The 15%/35% non-local-ratio thresholds (and the 50-task minimum) are unvalidated design-spike values: no external tool publishes an equivalent metric to calibrate against. Confirm against known-good/known-bad real logs before trusting the impact-band split.',
|
|
@@ -1494,11 +1383,9 @@ export const DETECTORS = [
|
|
|
1494
1383
|
},
|
|
1495
1384
|
},
|
|
1496
1385
|
{
|
|
1497
|
-
// Short-lived executors:
|
|
1498
|
-
//
|
|
1499
|
-
//
|
|
1500
|
-
// `utilization` above (no new data extraction), but measures lifetime
|
|
1501
|
-
// against a threshold instead of aggregate active-time.
|
|
1386
|
+
// Short-lived executors: stood up and torn down before doing useful work (wasteful
|
|
1387
|
+
// re-provisioning, not normal scale-down). Reuses utilization's add/remove matching, but
|
|
1388
|
+
// measures lifetime against a threshold instead of aggregate active-time.
|
|
1502
1389
|
type: 'autoscalingChurn', scope: 'app', order: 103, fixEffort: 'config', version: 1,
|
|
1503
1390
|
confidence: 'low',
|
|
1504
1391
|
thresholds: { shortLivedMs: 120_000, warningPct: 0.30, criticalPct: 0.60, minExecutors: 5 },
|
|
@@ -1510,7 +1397,7 @@ export const DETECTORS = [
|
|
|
1510
1397
|
ctx ,
|
|
1511
1398
|
) {
|
|
1512
1399
|
const { app, executorsAdded, executorsRemoved } = ctx;
|
|
1513
|
-
if (executorsAdded.length === 0 || app.endTime == null) return null;
|
|
1400
|
+
if (!app || executorsAdded.length === 0 || app.endTime == null) return null;
|
|
1514
1401
|
if (executorsAdded.length < this.thresholds.minExecutors) return null;
|
|
1515
1402
|
|
|
1516
1403
|
const removedAt = new Map ();
|
|
@@ -1540,10 +1427,8 @@ export const DETECTORS = [
|
|
|
1540
1427
|
},
|
|
1541
1428
|
},
|
|
1542
1429
|
{
|
|
1543
|
-
// Cross-execution relation reuse
|
|
1544
|
-
//
|
|
1545
|
-
// SQL workloads; see docs/adr/0009-caching-opportunity-relation-reuse.md).
|
|
1546
|
-
// Flags an input relation scanned by two or more SQL executions in one run.
|
|
1430
|
+
// Cross-execution relation reuse: flags an input relation scanned by two or more SQL
|
|
1431
|
+
// executions in one run, firing on real relation names (parquet:..., jdbc:...).
|
|
1547
1432
|
type: 'cachingOpportunity', scope: 'app', order: 105, fixEffort: 'code', version: 1,
|
|
1548
1433
|
docAnchor: '#bottleneck-utilization',
|
|
1549
1434
|
thresholds: { minExecutions: 2 },
|
|
@@ -1566,12 +1451,10 @@ export const DETECTORS = [
|
|
|
1566
1451
|
|
|
1567
1452
|
|
|
1568
1453
|
|
|
1569
|
-
// relationId -> { format, relation, executionIds:Set, executionBytes: Map<execId, bytes> }
|
|
1570
1454
|
const byRelation = new Map ();
|
|
1571
1455
|
for (const exec of sql.values()) {
|
|
1572
1456
|
if (!exec.planTree) continue;
|
|
1573
|
-
// Dedupe relations within one execution (self-joins count once), summing
|
|
1574
|
-
// this execution's read bytes per relation across its scan nodes.
|
|
1457
|
+
// Dedupe relations within one execution (self-joins count once), summing read bytes per relation.
|
|
1575
1458
|
const perExec = new Map ();
|
|
1576
1459
|
walkPlanTree(exec.planTree, (node) => {
|
|
1577
1460
|
const rid = scanRelationId(node.name ?? '', node.detail ?? '');
|
|
@@ -1591,15 +1474,13 @@ export const DETECTORS = [
|
|
|
1591
1474
|
}
|
|
1592
1475
|
}
|
|
1593
1476
|
|
|
1594
|
-
// fingerprint -> { operator, exampleNode, executionIds:Set, executionBytes:Map<execId,bytes>, ancestorFingerprints:Set<fingerprint>, leafRelationRids:Set<rid> }
|
|
1595
1477
|
const byComposite = new Map ();
|
|
1596
1478
|
for (const exec of sql.values()) {
|
|
1597
1479
|
if (!exec.planTree) continue;
|
|
1598
1480
|
const candidates = findCompositeCandidates(exec.planTree);
|
|
1599
1481
|
const fingerprintByNode = new Map (candidates.map((c) => [c.node, c.fingerprint]));
|
|
1600
1482
|
|
|
1601
|
-
// Dedupe identical fingerprints within this execution (repeated
|
|
1602
|
-
// identical composite counts once, mirroring the leaf perExec dedupe).
|
|
1483
|
+
// Dedupe identical fingerprints within this execution (repeated composite counts once).
|
|
1603
1484
|
const perExecComposite = new Map ();
|
|
1604
1485
|
for (const c of candidates) {
|
|
1605
1486
|
let agg = perExecComposite.get(c.fingerprint);
|
|
@@ -1636,12 +1517,10 @@ export const DETECTORS = [
|
|
|
1636
1517
|
}
|
|
1637
1518
|
}
|
|
1638
1519
|
|
|
1639
|
-
// Qualifying =
|
|
1640
|
-
//
|
|
1641
|
-
// subsumed: fully (equal execution sets) or partially (residual).
|
|
1520
|
+
// Qualifying = enough distinct executions on its own. Nested-dedupe: a qualifying composite
|
|
1521
|
+
// with a qualifying ANCESTOR is subsumed, fully (equal sets) or partially (residual).
|
|
1642
1522
|
const isQualifying = (fp ) =>
|
|
1643
1523
|
byComposite.has(fp) && byComposite.get(fp) .executionIds.size >= this.thresholds.minExecutions;
|
|
1644
|
-
// fingerprint -> { finalExecutionIds:Set, suppressed:boolean }
|
|
1645
1524
|
const compositeResolutions = new Map ();
|
|
1646
1525
|
for (const [fingerprint, agg] of byComposite) {
|
|
1647
1526
|
if (!isQualifying(fingerprint)) { compositeResolutions.set(fingerprint, { finalExecutionIds: agg.executionIds, suppressed: true }); continue; }
|
|
@@ -1658,7 +1537,7 @@ export const DETECTORS = [
|
|
|
1658
1537
|
const compositeVerb = { join: ['joined', 'join'], union: ['unioned', 'union'] };
|
|
1659
1538
|
|
|
1660
1539
|
const out = [];
|
|
1661
|
-
// rid -> Set<execId> covered by an emitted composite
|
|
1540
|
+
// rid -> Set<execId> covered by an emitted composite, for leaf suppression.
|
|
1662
1541
|
const coveredExecutionsByRid = new Map ();
|
|
1663
1542
|
for (const [fingerprint, agg] of byComposite) {
|
|
1664
1543
|
const resolution = compositeResolutions.get(fingerprint) ;
|
|
@@ -1744,10 +1623,8 @@ export const DETECTORS = [
|
|
|
1744
1623
|
let totalTasks = 0, failedTasks = 0;
|
|
1745
1624
|
for (const s of stages.values()) { totalTasks += s.taskCount ?? 0; failedTasks += s.failedTasks ?? 0; }
|
|
1746
1625
|
const taskFailureRate = totalTasks > 0 ? failedTasks / totalTasks : 0;
|
|
1747
|
-
// Average wall-clock
|
|
1748
|
-
//
|
|
1749
|
-
// missing either timestamp are excluded rather than counted as zero-length;
|
|
1750
|
-
// with no timed failed job at all the average is 0 (never NaN).
|
|
1626
|
+
// Average wall-clock of failed jobs, for the impact estimator's cost-only figure. Jobs
|
|
1627
|
+
// missing either timestamp are excluded (not counted as zero); with none timed the average is 0.
|
|
1751
1628
|
const timedFailedJobs = failedJobList.filter(j => j.submissionTime != null && j.completionTime != null);
|
|
1752
1629
|
const avgJobDurationMs = timedFailedJobs.length > 0
|
|
1753
1630
|
? timedFailedJobs.reduce((s, j) => s + (j.completionTime - j.submissionTime ), 0) / timedFailedJobs.length
|
|
@@ -1858,12 +1735,13 @@ export const DETECTORS = [
|
|
|
1858
1735
|
const nodes = [];
|
|
1859
1736
|
for (const n of g.nodes) walkPlanTree(n, (node) => nodes.push(node));
|
|
1860
1737
|
const stageIds = unionStageIds(nodes, fallbackStageIds);
|
|
1738
|
+
// resolvePlanTree always sets id; safe downstream of it.
|
|
1739
|
+
const planNodeIds = nodes.map((n) => n.id ).filter(Boolean);
|
|
1861
1740
|
const touching = g.sampleRelation ? ` (touching ${g.sampleRelation})` : '';
|
|
1862
1741
|
return {
|
|
1863
|
-
type: 'duplicatePlanSubtree', executionId: sqlExec.id, stageIds,
|
|
1864
|
-
// Fixed fallback: overwritten by deriveImpactBand
|
|
1865
|
-
//
|
|
1866
|
-
// surfaces on the rare miss (stage excluded from the occupancy sweep).
|
|
1742
|
+
type: 'duplicatePlanSubtree', executionId: sqlExec.id, stageIds, planNodeIds,
|
|
1743
|
+
// Fixed fallback: overwritten by deriveImpactBand when this finding gets a real
|
|
1744
|
+
// wallClock estimate (the common case). Only surfaces on the rare occupancy-sweep miss.
|
|
1867
1745
|
impactBand: 'warning', metric: 'subtreeOccurrences', value: g.occurrences,
|
|
1868
1746
|
rootName: g.rootName, subtreeSize: g.subtreeSize, sampleRelation: g.sampleRelation,
|
|
1869
1747
|
groupIndex: g.groupIndex, confidence: 'medium',
|
|
@@ -1908,8 +1786,9 @@ export const DETECTORS = [
|
|
|
1908
1786
|
const fallbackStageIds = stageIdsForSqlExec(sqlExec.id, ctx.stages);
|
|
1909
1787
|
return hits.map(h => {
|
|
1910
1788
|
const stageIds = unionStageIds([h.node], fallbackStageIds);
|
|
1789
|
+
const planNodeIds = h.node.id ? [h.node.id] : [];
|
|
1911
1790
|
return {
|
|
1912
|
-
type: 'smallFiles', executionId: sqlExec.id, stageIds,
|
|
1791
|
+
type: 'smallFiles', executionId: sqlExec.id, stageIds, planNodeIds,
|
|
1913
1792
|
impactBand: 'warning',
|
|
1914
1793
|
metric: 'avgFileSizeBytes', value: Math.round(h.avgBytes),
|
|
1915
1794
|
fileCount: h.fileCount, direction: h.direction, nodeName: h.nodeName,
|
|
@@ -1921,10 +1800,9 @@ export const DETECTORS = [
|
|
|
1921
1800
|
},
|
|
1922
1801
|
},
|
|
1923
1802
|
{
|
|
1924
|
-
// Entry-level type is an identifier only; it never appears on an
|
|
1925
|
-
//
|
|
1926
|
-
//
|
|
1927
|
-
// direction rules (dataflint JoinToBroadcastAlert / BroadcastTooLargeAlert).
|
|
1803
|
+
// Entry-level type is an identifier only; it never appears on an emitted finding. Findings
|
|
1804
|
+
// carry 'underBroadcast'/'overBroadcast' since one shared plan-walk covers both
|
|
1805
|
+
// opposite-direction rules (JoinToBroadcastAlert / BroadcastTooLargeAlert).
|
|
1928
1806
|
type: 'broadcastSizing', scope: 'sql', order: 132, fixEffort: 'config', version: 2,
|
|
1929
1807
|
docAnchor: '#bottleneck-broadcast-sizing',
|
|
1930
1808
|
thresholds: {
|
|
@@ -1959,6 +1837,8 @@ export const DETECTORS = [
|
|
|
1959
1837
|
const contributors = [...boundarySizeContributors(childA), ...boundarySizeContributors(childB)];
|
|
1960
1838
|
out.push({
|
|
1961
1839
|
type: 'underBroadcast', executionId: sqlExec.id, stageIds: unionStageIds(contributors, fallbackStageIds),
|
|
1840
|
+
// resolvePlanTree always sets id; safe downstream of it.
|
|
1841
|
+
planNodeIds: contributors.map((n) => n.id ).filter(Boolean),
|
|
1962
1842
|
impactBand: 'info', metric: 'smallerSideBytes', value: smaller,
|
|
1963
1843
|
largerSideBytes: larger,
|
|
1964
1844
|
recommendation: `The smaller input to this Sort Merge Join (${formatBytes(smaller)}) is well under the broadcast threshold relative to the larger side (${formatBytes(larger)}): this could have been a broadcast join. Consider a broadcast() hint or raising spark.sql.autoBroadcastJoinThreshold.`,
|
|
@@ -1966,17 +1846,18 @@ export const DETECTORS = [
|
|
|
1966
1846
|
}
|
|
1967
1847
|
}
|
|
1968
1848
|
}
|
|
1969
|
-
if (node.name
|
|
1849
|
+
if (isBroadcastExchangeNode(node.name)) {
|
|
1970
1850
|
const m = (node.metrics ?? []).find(x => x.name === 'data size');
|
|
1971
1851
|
if (m && m.value > overBroadcastBytes) {
|
|
1972
|
-
//
|
|
1973
|
-
//
|
|
1974
|
-
//
|
|
1975
|
-
// executor-side metrics, so only the child is unioned in.
|
|
1852
|
+
// BroadcastExchange's own metrics are driver-computed and never on any TaskEnd
|
|
1853
|
+
// (node.stageIds always empty in real data); its child carries the executor-side
|
|
1854
|
+
// metrics, so only the child unions in.
|
|
1976
1855
|
const child = (node.children ?? [])[0];
|
|
1977
1856
|
out.push({
|
|
1978
1857
|
type: 'overBroadcast', executionId: sqlExec.id,
|
|
1979
1858
|
stageIds: unionStageIds(child ? [child] : [], fallbackStageIds),
|
|
1859
|
+
// resolvePlanTree always sets id; safe downstream of it.
|
|
1860
|
+
planNodeIds: [node.id ].filter(Boolean),
|
|
1980
1861
|
impactBand: 'warning', metric: 'broadcastBytes', value: m.value,
|
|
1981
1862
|
recommendation: `This broadcast (${formatBytes(m.value)}) exceeds the 1 GB threshold: check for a misapplied broadcast hint or a misconfigured spark.sql.autoBroadcastJoinThreshold.`,
|
|
1982
1863
|
});
|