sparkforensics-cli 0.1.0 → 0.2.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (361) hide show
  1. package/README.md +6 -0
  2. package/bin/sparkforensics-analyze.mjs +113 -48
  3. package/export-template/docs/404.html +25 -0
  4. package/export-template/docs/assets/app.DQTZyGL1.js +1 -0
  5. package/export-template/docs/assets/aqe-loop.IwQSATHw.svg +1 -0
  6. package/export-template/docs/assets/aqe-loop.dark.DGbaxqJE.svg +1 -0
  7. package/export-template/docs/assets/broadcast-vs-shuffle.Db4WY1XK.svg +1 -0
  8. package/export-template/docs/assets/broadcast-vs-shuffle.dark.C7Bxs0mG.svg +1 -0
  9. package/export-template/docs/assets/cache-lifecycle.dark.B-hS7AgU.svg +1 -0
  10. package/export-template/docs/assets/cache-lifecycle.rEOVYQNU.svg +1 -0
  11. package/export-template/docs/assets/chunks/@localSearchIndexroot.DppnXnDE.js +1 -0
  12. package/export-template/docs/assets/chunks/VPLocalSearchBox.BkBIPFs6.js +9 -0
  13. package/export-template/docs/assets/chunks/duplicate-plan-subtree.dark.Cdp70QhV.js +1 -0
  14. package/export-template/docs/assets/chunks/framework.DSg0KOwT.js +20 -0
  15. package/export-template/docs/assets/chunks/retry-escalation-ladder.dark.DHipdJgZ.js +1 -0
  16. package/export-template/docs/assets/chunks/theme.DP0u1AUq.js +2 -0
  17. package/export-template/docs/assets/cold-start-timeline.DxC_Sc7w.svg +1 -0
  18. package/export-template/docs/assets/cold-start-timeline.dark.CZ17YcAG.svg +1 -0
  19. package/export-template/docs/assets/columnar-layout.PghGeOEA.svg +1 -0
  20. package/export-template/docs/assets/columnar-layout.dark.BVNlz0ff.svg +1 -0
  21. package/export-template/docs/assets/container-memory.DIO0AnIm.svg +1 -0
  22. package/export-template/docs/assets/container-memory.dark.CP-5zuCl.svg +1 -0
  23. package/export-template/docs/assets/contributor-guide_architecture_board-widgets.md.CWpj01WU.js +1 -0
  24. package/export-template/docs/assets/contributor-guide_architecture_board-widgets.md.CWpj01WU.lean.js +1 -0
  25. package/export-template/docs/assets/contributor-guide_architecture_detector-contract.md.CgzUsQ6W.js +1 -0
  26. package/export-template/docs/assets/contributor-guide_architecture_detector-contract.md.CgzUsQ6W.lean.js +1 -0
  27. package/export-template/docs/assets/contributor-guide_architecture_drill-down.md.BtPdlM7r.js +1 -0
  28. package/export-template/docs/assets/contributor-guide_architecture_drill-down.md.BtPdlM7r.lean.js +1 -0
  29. package/export-template/docs/assets/contributor-guide_architecture_impact-estimation.md.CooslVJt.js +1 -0
  30. package/export-template/docs/assets/contributor-guide_architecture_impact-estimation.md.CooslVJt.lean.js +1 -0
  31. package/export-template/docs/assets/contributor-guide_architecture_index.md.3TO9ic6w.js +1 -0
  32. package/export-template/docs/assets/contributor-guide_architecture_index.md.3TO9ic6w.lean.js +1 -0
  33. package/export-template/docs/assets/contributor-guide_architecture_overview.md.CehiRmGn.js +1 -0
  34. package/export-template/docs/assets/contributor-guide_architecture_overview.md.CehiRmGn.lean.js +1 -0
  35. package/export-template/docs/assets/contributor-guide_architecture_state-and-history.md.C-xxn0q7.js +1 -0
  36. package/export-template/docs/assets/contributor-guide_architecture_state-and-history.md.C-xxn0q7.lean.js +1 -0
  37. package/export-template/docs/assets/contributor-guide_architecture_widget-rendering.md.R27gQrgY.js +1 -0
  38. package/export-template/docs/assets/contributor-guide_architecture_widget-rendering.md.R27gQrgY.lean.js +1 -0
  39. package/export-template/docs/assets/contributor-guide_architecture_worker-protocol.md.IbnfNrV3.js +6 -0
  40. package/export-template/docs/assets/contributor-guide_architecture_worker-protocol.md.IbnfNrV3.lean.js +1 -0
  41. package/export-template/docs/assets/contributor-guide_contributing.md.CvRsdr6J.js +1 -0
  42. package/export-template/docs/assets/contributor-guide_contributing.md.CvRsdr6J.lean.js +1 -0
  43. package/export-template/docs/assets/contributor-guide_development-setup.md.DvAN_9mK.js +12 -0
  44. package/export-template/docs/assets/contributor-guide_development-setup.md.DvAN_9mK.lean.js +1 -0
  45. package/export-template/docs/assets/contributor-guide_testing.md.6rIKqSyY.js +1 -0
  46. package/export-template/docs/assets/contributor-guide_testing.md.6rIKqSyY.lean.js +1 -0
  47. package/export-template/docs/assets/dag-stages.DSz_S937.svg +1 -0
  48. package/export-template/docs/assets/dag-stages.dark.F72UzxH4.svg +1 -0
  49. package/export-template/docs/assets/driver-executor.D5pQ7YN1.svg +1 -0
  50. package/export-template/docs/assets/driver-executor.dark.BmX9cPvh.svg +1 -0
  51. package/export-template/docs/assets/duplicate-plan-subtree.B4cvN6fj.svg +1 -0
  52. package/export-template/docs/assets/duplicate-plan-subtree.dark.Dw8wS0Ag.svg +1 -0
  53. package/export-template/docs/assets/index.md.CHJVslga.js +1 -0
  54. package/export-template/docs/assets/index.md.CHJVslga.lean.js +1 -0
  55. package/export-template/docs/assets/inter-italic-cyrillic-ext.r48I6akx.woff2 +0 -0
  56. package/export-template/docs/assets/inter-italic-cyrillic.By2_1cv3.woff2 +0 -0
  57. package/export-template/docs/assets/inter-italic-greek-ext.1u6EdAuj.woff2 +0 -0
  58. package/export-template/docs/assets/inter-italic-greek.DJ8dCoTZ.woff2 +0 -0
  59. package/export-template/docs/assets/inter-italic-latin-ext.CN1xVJS-.woff2 +0 -0
  60. package/export-template/docs/assets/inter-italic-latin.C2AdPX0b.woff2 +0 -0
  61. package/export-template/docs/assets/inter-italic-vietnamese.BSbpV94h.woff2 +0 -0
  62. package/export-template/docs/assets/inter-roman-cyrillic-ext.BBPuwvHQ.woff2 +0 -0
  63. package/export-template/docs/assets/inter-roman-cyrillic.C5lxZ8CY.woff2 +0 -0
  64. package/export-template/docs/assets/inter-roman-greek-ext.CqjqNYQ-.woff2 +0 -0
  65. package/export-template/docs/assets/inter-roman-greek.BBVDIX6e.woff2 +0 -0
  66. package/export-template/docs/assets/inter-roman-latin-ext.4ZJIpNVo.woff2 +0 -0
  67. package/export-template/docs/assets/inter-roman-latin.Di8DUHzh.woff2 +0 -0
  68. package/export-template/docs/assets/inter-roman-vietnamese.BjW4sHH5.woff2 +0 -0
  69. package/export-template/docs/assets/join-strategy.C_FvrCEo.svg +1 -0
  70. package/export-template/docs/assets/join-strategy.dark.ChMLnNII.svg +1 -0
  71. package/export-template/docs/assets/memory-borrowing.BqQRJg0u.svg +1 -0
  72. package/export-template/docs/assets/memory-borrowing.dark.Yhh20O9C.svg +1 -0
  73. package/export-template/docs/assets/memory-regions.XHvO7jHG.svg +1 -0
  74. package/export-template/docs/assets/memory-regions.dark.D4TP9_08.svg +1 -0
  75. package/export-template/docs/assets/repartition-vs-coalesce.BovLRrpj.svg +1 -0
  76. package/export-template/docs/assets/repartition-vs-coalesce.dark.BhAczKZQ.svg +1 -0
  77. package/export-template/docs/assets/retry-escalation-ladder.DyTKJJmZ.svg +1 -0
  78. package/export-template/docs/assets/retry-escalation-ladder.dark.BdsabtU3.svg +1 -0
  79. package/export-template/docs/assets/shuffle-map-reduce.KuOEZVmg.svg +1 -0
  80. package/export-template/docs/assets/shuffle-map-reduce.dark.BgQZnFSb.svg +1 -0
  81. package/export-template/docs/assets/spill-classification.BU2euYDO.svg +1 -0
  82. package/export-template/docs/assets/spill-classification.dark.D7i1M40d.svg +1 -0
  83. package/export-template/docs/assets/style.DSixAiZE.css +1 -0
  84. package/export-template/docs/assets/tuning-reference_anti-patterns.md.Df1YMIHu.js +1 -0
  85. package/export-template/docs/assets/tuning-reference_anti-patterns.md.Df1YMIHu.lean.js +1 -0
  86. package/export-template/docs/assets/tuning-reference_aqe.md.BIsCtLzm.js +1 -0
  87. package/export-template/docs/assets/tuning-reference_aqe.md.BIsCtLzm.lean.js +1 -0
  88. package/export-template/docs/assets/tuning-reference_bottleneck-broadcast-sizing.md.CEstB3Ia.js +1 -0
  89. package/export-template/docs/assets/tuning-reference_bottleneck-broadcast-sizing.md.CEstB3Ia.lean.js +1 -0
  90. package/export-template/docs/assets/tuning-reference_bottleneck-cold-start.md.CEuy-72y.js +7 -0
  91. package/export-template/docs/assets/tuning-reference_bottleneck-cold-start.md.CEuy-72y.lean.js +1 -0
  92. package/export-template/docs/assets/tuning-reference_bottleneck-duplicate-plan-subtree.md.CIohQDfn.js +1 -0
  93. package/export-template/docs/assets/tuning-reference_bottleneck-duplicate-plan-subtree.md.CIohQDfn.lean.js +1 -0
  94. package/export-template/docs/assets/tuning-reference_bottleneck-failures.md.4z5BXGJ2.js +6 -0
  95. package/export-template/docs/assets/tuning-reference_bottleneck-failures.md.4z5BXGJ2.lean.js +1 -0
  96. package/export-template/docs/assets/tuning-reference_bottleneck-gc.md.DSxzZRK7.js +6 -0
  97. package/export-template/docs/assets/tuning-reference_bottleneck-gc.md.DSxzZRK7.lean.js +1 -0
  98. package/export-template/docs/assets/tuning-reference_bottleneck-job-failure-rate.md.BaJl__1W.js +8 -0
  99. package/export-template/docs/assets/tuning-reference_bottleneck-job-failure-rate.md.BaJl__1W.lean.js +1 -0
  100. package/export-template/docs/assets/tuning-reference_bottleneck-memory-utilization.md.DbP-SJZc.js +7 -0
  101. package/export-template/docs/assets/tuning-reference_bottleneck-memory-utilization.md.DbP-SJZc.lean.js +1 -0
  102. package/export-template/docs/assets/tuning-reference_bottleneck-retry-waste.md.D5JMjVOt.js +1 -0
  103. package/export-template/docs/assets/tuning-reference_bottleneck-retry-waste.md.D5JMjVOt.lean.js +1 -0
  104. package/export-template/docs/assets/tuning-reference_bottleneck-shuffle.md.CM-nTmIH.js +12 -0
  105. package/export-template/docs/assets/tuning-reference_bottleneck-shuffle.md.CM-nTmIH.lean.js +1 -0
  106. package/export-template/docs/assets/tuning-reference_bottleneck-skew.md.BdUwiDhn.js +14 -0
  107. package/export-template/docs/assets/tuning-reference_bottleneck-skew.md.BdUwiDhn.lean.js +1 -0
  108. package/export-template/docs/assets/tuning-reference_bottleneck-slow-host.md.BlIo6UDW.js +7 -0
  109. package/export-template/docs/assets/tuning-reference_bottleneck-slow-host.md.BlIo6UDW.lean.js +1 -0
  110. package/export-template/docs/assets/tuning-reference_bottleneck-small-files.md.B8kloyx8.js +5 -0
  111. package/export-template/docs/assets/tuning-reference_bottleneck-small-files.md.B8kloyx8.lean.js +1 -0
  112. package/export-template/docs/assets/tuning-reference_bottleneck-spill.md.PNH7mITt.js +6 -0
  113. package/export-template/docs/assets/tuning-reference_bottleneck-spill.md.PNH7mITt.lean.js +1 -0
  114. package/export-template/docs/assets/tuning-reference_bottleneck-straggler.md.DY36fHN5.js +7 -0
  115. package/export-template/docs/assets/tuning-reference_bottleneck-straggler.md.DY36fHN5.lean.js +1 -0
  116. package/export-template/docs/assets/tuning-reference_bottleneck-tiny-tasks.md.QTV7O8kU.js +8 -0
  117. package/export-template/docs/assets/tuning-reference_bottleneck-tiny-tasks.md.QTV7O8kU.lean.js +1 -0
  118. package/export-template/docs/assets/tuning-reference_bottleneck-utilization.md.DTiueZC3.js +5 -0
  119. package/export-template/docs/assets/tuning-reference_bottleneck-utilization.md.DTiueZC3.lean.js +1 -0
  120. package/export-template/docs/assets/tuning-reference_caching.md.B7aQ8asB.js +1 -0
  121. package/export-template/docs/assets/tuning-reference_caching.md.B7aQ8asB.lean.js +1 -0
  122. package/export-template/docs/assets/tuning-reference_cluster-config.md.ZVmDGsQ3.js +1 -0
  123. package/export-template/docs/assets/tuning-reference_cluster-config.md.ZVmDGsQ3.lean.js +1 -0
  124. package/export-template/docs/assets/tuning-reference_config.md.UvveiWG3.js +1 -0
  125. package/export-template/docs/assets/tuning-reference_config.md.UvveiWG3.lean.js +1 -0
  126. package/export-template/docs/assets/tuning-reference_data-formats.md.bjCAWH3N.js +1 -0
  127. package/export-template/docs/assets/tuning-reference_data-formats.md.bjCAWH3N.lean.js +1 -0
  128. package/export-template/docs/assets/tuning-reference_index.md.BQ_NooMV.js +1 -0
  129. package/export-template/docs/assets/tuning-reference_index.md.BQ_NooMV.lean.js +1 -0
  130. package/export-template/docs/assets/tuning-reference_intro.md.CobD-lGB.js +1 -0
  131. package/export-template/docs/assets/tuning-reference_intro.md.CobD-lGB.lean.js +1 -0
  132. package/export-template/docs/assets/tuning-reference_joins.md.BtKs_CuW.js +1 -0
  133. package/export-template/docs/assets/tuning-reference_joins.md.BtKs_CuW.lean.js +1 -0
  134. package/export-template/docs/assets/tuning-reference_memory-model.md.DhT-n4y3.js +1 -0
  135. package/export-template/docs/assets/tuning-reference_memory-model.md.DhT-n4y3.lean.js +1 -0
  136. package/export-template/docs/assets/tuning-reference_metrics.md.mLOh7Apj.js +1 -0
  137. package/export-template/docs/assets/tuning-reference_metrics.md.mLOh7Apj.lean.js +1 -0
  138. package/export-template/docs/assets/tuning-reference_partitioning.md.q0zKF_8X.js +1 -0
  139. package/export-template/docs/assets/tuning-reference_partitioning.md.q0zKF_8X.lean.js +1 -0
  140. package/export-template/docs/assets/tuning-reference_pyspark.md.DDCfvN9t.js +6 -0
  141. package/export-template/docs/assets/tuning-reference_pyspark.md.DDCfvN9t.lean.js +1 -0
  142. package/export-template/docs/assets/tuning-reference_shuffle.md.BZZ7R4Ix.js +1 -0
  143. package/export-template/docs/assets/tuning-reference_shuffle.md.BZZ7R4Ix.lean.js +1 -0
  144. package/export-template/docs/assets/tuning-reference_spark-architecture.md.Dwzm5avO.js +1 -0
  145. package/export-template/docs/assets/tuning-reference_spark-architecture.md.Dwzm5avO.lean.js +1 -0
  146. package/export-template/docs/assets/tuning-reference_table-formats.md.D6wj-2dX.js +1 -0
  147. package/export-template/docs/assets/tuning-reference_table-formats.md.D6wj-2dX.lean.js +1 -0
  148. package/export-template/docs/assets/udf-execution-models.BUFDICuG.svg +1 -0
  149. package/export-template/docs/assets/udf-execution-models.dark.YTNS6GDq.svg +1 -0
  150. package/export-template/docs/assets/user-guide_alternative-log-retrieval.md.B4tPGIal.js +1 -0
  151. package/export-template/docs/assets/user-guide_alternative-log-retrieval.md.B4tPGIal.lean.js +1 -0
  152. package/export-template/docs/assets/user-guide_getting-started.md.BJvwLEIM.js +3 -0
  153. package/export-template/docs/assets/user-guide_getting-started.md.BJvwLEIM.lean.js +1 -0
  154. package/export-template/docs/assets/user-guide_mcp-tools.md.Vi3RoflJ.js +125 -0
  155. package/export-template/docs/assets/user-guide_mcp-tools.md.Vi3RoflJ.lean.js +1 -0
  156. package/export-template/docs/assets/user-guide_run-comparison.md.CQc1aoU8.js +1 -0
  157. package/export-template/docs/assets/user-guide_run-comparison.md.CQc1aoU8.lean.js +1 -0
  158. package/export-template/docs/assets/user-guide_understanding-findings.md.DL1UDhvR.js +1 -0
  159. package/export-template/docs/assets/user-guide_understanding-findings.md.DL1UDhvR.lean.js +1 -0
  160. package/export-template/docs/contributor-guide/architecture/board-widgets.html +25 -0
  161. package/export-template/docs/contributor-guide/architecture/detector-contract.html +25 -0
  162. package/export-template/docs/contributor-guide/architecture/drill-down.html +25 -0
  163. package/export-template/docs/contributor-guide/architecture/impact-estimation.html +25 -0
  164. package/export-template/docs/contributor-guide/architecture/index.html +25 -0
  165. package/export-template/docs/contributor-guide/architecture/overview.html +25 -0
  166. package/export-template/docs/contributor-guide/architecture/state-and-history.html +25 -0
  167. package/export-template/docs/contributor-guide/architecture/widget-rendering.html +25 -0
  168. package/export-template/docs/contributor-guide/architecture/worker-protocol.html +30 -0
  169. package/export-template/docs/contributor-guide/contributing.html +25 -0
  170. package/export-template/docs/contributor-guide/development-setup.html +36 -0
  171. package/export-template/docs/contributor-guide/testing.html +25 -0
  172. package/export-template/docs/favicon.svg +4 -0
  173. package/export-template/docs/hashmap.json +1 -0
  174. package/export-template/docs/index.html +25 -0
  175. package/export-template/docs/package.json +1 -0
  176. package/export-template/docs/tuning-reference/anti-patterns.html +25 -0
  177. package/export-template/docs/tuning-reference/aqe.html +25 -0
  178. package/export-template/docs/tuning-reference/bottleneck-broadcast-sizing.html +25 -0
  179. package/export-template/docs/tuning-reference/bottleneck-cold-start.html +31 -0
  180. package/export-template/docs/tuning-reference/bottleneck-duplicate-plan-subtree.html +25 -0
  181. package/export-template/docs/tuning-reference/bottleneck-failures.html +30 -0
  182. package/export-template/docs/tuning-reference/bottleneck-gc.html +30 -0
  183. package/export-template/docs/tuning-reference/bottleneck-job-failure-rate.html +32 -0
  184. package/export-template/docs/tuning-reference/bottleneck-memory-utilization.html +31 -0
  185. package/export-template/docs/tuning-reference/bottleneck-retry-waste.html +25 -0
  186. package/export-template/docs/tuning-reference/bottleneck-shuffle.html +36 -0
  187. package/export-template/docs/tuning-reference/bottleneck-skew.html +38 -0
  188. package/export-template/docs/tuning-reference/bottleneck-slow-host.html +31 -0
  189. package/export-template/docs/tuning-reference/bottleneck-small-files.html +29 -0
  190. package/export-template/docs/tuning-reference/bottleneck-spill.html +30 -0
  191. package/export-template/docs/tuning-reference/bottleneck-straggler.html +31 -0
  192. package/export-template/docs/tuning-reference/bottleneck-tiny-tasks.html +32 -0
  193. package/export-template/docs/tuning-reference/bottleneck-utilization.html +29 -0
  194. package/export-template/docs/tuning-reference/caching.html +25 -0
  195. package/export-template/docs/tuning-reference/cluster-config.html +25 -0
  196. package/export-template/docs/tuning-reference/config.html +25 -0
  197. package/export-template/docs/tuning-reference/data-formats.html +25 -0
  198. package/export-template/docs/tuning-reference/index.html +25 -0
  199. package/export-template/docs/tuning-reference/intro.html +25 -0
  200. package/export-template/docs/tuning-reference/joins.html +25 -0
  201. package/export-template/docs/tuning-reference/memory-model.html +25 -0
  202. package/export-template/docs/tuning-reference/metrics.html +25 -0
  203. package/export-template/docs/tuning-reference/partitioning.html +25 -0
  204. package/export-template/docs/tuning-reference/pyspark.html +30 -0
  205. package/export-template/docs/tuning-reference/shuffle.html +25 -0
  206. package/export-template/docs/tuning-reference/spark-architecture.html +25 -0
  207. package/export-template/docs/tuning-reference/table-formats.html +25 -0
  208. package/export-template/docs/user-guide/alternative-log-retrieval.html +25 -0
  209. package/export-template/docs/user-guide/getting-started.html +27 -0
  210. package/export-template/docs/user-guide/mcp-tools.html +149 -0
  211. package/export-template/docs/user-guide/run-comparison.html +25 -0
  212. package/export-template/docs/user-guide/understanding-findings.html +25 -0
  213. package/export-template/docs/vp-icons.css +0 -0
  214. package/export-template/favicon.svg +4 -0
  215. package/export-template/index.html +111 -0
  216. package/export-template/parser-worker-DyjiQvfP.js +112 -0
  217. package/export-template/sample-runs/sample-run.ndjson.gz +0 -0
  218. package/package.json +20 -6
  219. package/vendor-core/analyzer.js +74 -74
  220. package/vendor-core/cli/budgets.js +13 -27
  221. package/vendor-core/cli/collect-run.js +43 -19
  222. package/vendor-core/core-count.js +25 -27
  223. package/vendor-core/core-locality-ratio.js +4 -11
  224. package/vendor-core/core-time-series.js +6 -12
  225. package/vendor-core/core-usage-locality.js +3 -4
  226. package/vendor-core/detectors.js +395 -389
  227. package/vendor-core/docs-config.js +69 -21
  228. package/vendor-core/docs-content/chapters/01-intro.md +32 -0
  229. package/vendor-core/docs-content/chapters/02-spark-architecture.md +76 -0
  230. package/vendor-core/docs-content/chapters/03-memory-model.md +73 -0
  231. package/vendor-core/docs-content/chapters/04-partitioning.md +65 -0
  232. package/vendor-core/docs-content/chapters/05-joins.md +62 -0
  233. package/vendor-core/docs-content/chapters/06-shuffle.md +59 -0
  234. package/vendor-core/docs-content/chapters/07-data-formats.md +81 -0
  235. package/vendor-core/docs-content/chapters/07b-table-formats.md +56 -0
  236. package/vendor-core/docs-content/chapters/08-caching.md +58 -0
  237. package/vendor-core/docs-content/chapters/09-pyspark.md +78 -0
  238. package/vendor-core/docs-content/chapters/10-aqe.md +167 -0
  239. package/vendor-core/docs-content/chapters/11-cluster-config.md +170 -0
  240. package/vendor-core/docs-content/chapters/12-anti-patterns.md +171 -0
  241. package/vendor-core/docs-content/chapters/14-metrics.md +87 -0
  242. package/vendor-core/docs-content/chapters/15-config.md +93 -0
  243. package/vendor-core/docs-content/chapters/nav-index.json +370 -0
  244. package/vendor-core/docs-content/detection/cache.md +7 -0
  245. package/vendor-core/docs-content/detection/cfg.md +15 -0
  246. package/vendor-core/docs-content/detection/chrn.md +9 -0
  247. package/vendor-core/docs-content/detection/cold.md +4 -0
  248. package/vendor-core/docs-content/detection/cstor.md +4 -0
  249. package/vendor-core/docs-content/detection/fail.md +5 -0
  250. package/vendor-core/docs-content/detection/gc.md +4 -0
  251. package/vendor-core/docs-content/detection/host.md +5 -0
  252. package/vendor-core/docs-content/detection/incmp.md +6 -0
  253. package/vendor-core/docs-content/detection/jobs.md +4 -0
  254. package/vendor-core/docs-content/detection/local.md +6 -0
  255. package/vendor-core/docs-content/detection/mem.md +10 -0
  256. package/vendor-core/docs-content/detection/part.md +5 -0
  257. package/vendor-core/docs-content/detection/plan.md +14 -0
  258. package/vendor-core/docs-content/detection/retry.md +4 -0
  259. package/vendor-core/docs-content/detection/sfail.md +5 -0
  260. package/vendor-core/docs-content/detection/shape.md +5 -0
  261. package/vendor-core/docs-content/detection/shfl.md +4 -0
  262. package/vendor-core/docs-content/detection/skew.md +6 -0
  263. package/vendor-core/docs-content/detection/slow.md +6 -0
  264. package/vendor-core/docs-content/detection/spec.md +8 -0
  265. package/vendor-core/docs-content/detection/spill.md +7 -0
  266. package/vendor-core/docs-content/detection/strag.md +5 -0
  267. package/vendor-core/docs-content/detection/tiny.md +4 -0
  268. package/vendor-core/docs-content/detection/util.md +4 -0
  269. package/vendor-core/docs-content/diagrams/aqe-loop.dark.svg +1 -0
  270. package/vendor-core/docs-content/diagrams/aqe-loop.svg +1 -0
  271. package/vendor-core/docs-content/diagrams/broadcast-vs-shuffle.dark.svg +1 -0
  272. package/vendor-core/docs-content/diagrams/broadcast-vs-shuffle.svg +1 -0
  273. package/vendor-core/docs-content/diagrams/cache-lifecycle.dark.svg +1 -0
  274. package/vendor-core/docs-content/diagrams/cache-lifecycle.svg +1 -0
  275. package/vendor-core/docs-content/diagrams/cold-start-timeline.dark.svg +1 -0
  276. package/vendor-core/docs-content/diagrams/cold-start-timeline.svg +1 -0
  277. package/vendor-core/docs-content/diagrams/columnar-layout.dark.svg +1 -0
  278. package/vendor-core/docs-content/diagrams/columnar-layout.svg +1 -0
  279. package/vendor-core/docs-content/diagrams/container-memory.dark.svg +1 -0
  280. package/vendor-core/docs-content/diagrams/container-memory.svg +1 -0
  281. package/vendor-core/docs-content/diagrams/dag-stages.dark.svg +1 -0
  282. package/vendor-core/docs-content/diagrams/dag-stages.svg +1 -0
  283. package/vendor-core/docs-content/diagrams/driver-executor.dark.svg +1 -0
  284. package/vendor-core/docs-content/diagrams/driver-executor.svg +1 -0
  285. package/vendor-core/docs-content/diagrams/duplicate-plan-subtree.dark.svg +1 -0
  286. package/vendor-core/docs-content/diagrams/duplicate-plan-subtree.svg +1 -0
  287. package/vendor-core/docs-content/diagrams/join-strategy.dark.svg +1 -0
  288. package/vendor-core/docs-content/diagrams/join-strategy.svg +1 -0
  289. package/vendor-core/docs-content/diagrams/memory-borrowing.dark.svg +1 -0
  290. package/vendor-core/docs-content/diagrams/memory-borrowing.svg +1 -0
  291. package/vendor-core/docs-content/diagrams/memory-regions.dark.svg +1 -0
  292. package/vendor-core/docs-content/diagrams/memory-regions.svg +1 -0
  293. package/vendor-core/docs-content/diagrams/repartition-vs-coalesce.dark.svg +1 -0
  294. package/vendor-core/docs-content/diagrams/repartition-vs-coalesce.svg +1 -0
  295. package/vendor-core/docs-content/diagrams/retry-escalation-ladder.dark.svg +1 -0
  296. package/vendor-core/docs-content/diagrams/retry-escalation-ladder.svg +1 -0
  297. package/vendor-core/docs-content/diagrams/shuffle-map-reduce.dark.svg +1 -0
  298. package/vendor-core/docs-content/diagrams/shuffle-map-reduce.svg +1 -0
  299. package/vendor-core/docs-content/diagrams/spill-classification.dark.svg +1 -0
  300. package/vendor-core/docs-content/diagrams/spill-classification.svg +1 -0
  301. package/vendor-core/docs-content/diagrams/udf-execution-models.dark.svg +1 -0
  302. package/vendor-core/docs-content/diagrams/udf-execution-models.svg +1 -0
  303. package/vendor-core/docs-content/tuning/broadcast-sizing.md +78 -0
  304. package/vendor-core/docs-content/tuning/cold-start.md +81 -0
  305. package/vendor-core/docs-content/tuning/duplicate-plan-subtree.md +45 -0
  306. package/vendor-core/docs-content/tuning/failures.md +124 -0
  307. package/vendor-core/docs-content/tuning/gc.md +110 -0
  308. package/vendor-core/docs-content/tuning/job-failure-rate.md +101 -0
  309. package/vendor-core/docs-content/tuning/memory-utilization.md +58 -0
  310. package/vendor-core/docs-content/tuning/retry-waste.md +90 -0
  311. package/vendor-core/docs-content/tuning/shuffle.md +154 -0
  312. package/vendor-core/docs-content/tuning/skew.md +123 -0
  313. package/vendor-core/docs-content/tuning/slow-host.md +117 -0
  314. package/vendor-core/docs-content/tuning/small-files.md +99 -0
  315. package/vendor-core/docs-content/tuning/spill.md +114 -0
  316. package/vendor-core/docs-content/tuning/straggler.md +103 -0
  317. package/vendor-core/docs-content/tuning/tiny-tasks.md +94 -0
  318. package/vendor-core/docs-content/tuning/utilization.md +90 -0
  319. package/vendor-core/docs-site-config.js +10 -17
  320. package/vendor-core/efficiency-model.js +7 -13
  321. package/vendor-core/etl-phases.js +3 -5
  322. package/vendor-core/event-handlers.js +232 -134
  323. package/vendor-core/event-schemas.js +48 -114
  324. package/vendor-core/evidence-availability.js +5 -10
  325. package/vendor-core/evidence-report.js +73 -123
  326. package/vendor-core/export-data.js +48 -0
  327. package/vendor-core/finding-action-label.js +4 -10
  328. package/vendor-core/finding-filter-predicate.js +3 -7
  329. package/vendor-core/finding-generic-recommendation.js +112 -0
  330. package/vendor-core/finding-names.js +51 -0
  331. package/vendor-core/format-utils.js +112 -38
  332. package/vendor-core/impact-band.js +18 -24
  333. package/vendor-core/impact-estimator.js +38 -74
  334. package/vendor-core/ingest.js +7 -13
  335. package/vendor-core/job-groups.js +3 -6
  336. package/vendor-core/list-runs.js +278 -0
  337. package/vendor-core/load-vendored.js +6 -12
  338. package/vendor-core/log-header-peek.js +81 -0
  339. package/vendor-core/lz4-block.js +4 -6
  340. package/vendor-core/mcp-server-factory.js +38 -8
  341. package/vendor-core/mcp-tools.js +105 -76
  342. package/vendor-core/model-assembler.js +8 -16
  343. package/vendor-core/occupancy.js +5 -9
  344. package/vendor-core/parser-worker.js +20 -29
  345. package/vendor-core/plan-dot.js +2 -5
  346. package/vendor-core/plan-duration-attribution.js +78 -29
  347. package/vendor-core/plan-graph-model.js +126 -69
  348. package/vendor-core/plan-node-detail.js +31 -17
  349. package/vendor-core/plan-summary.js +19 -8
  350. package/vendor-core/recommendation-rollup.js +35 -39
  351. package/vendor-core/redact.js +72 -16
  352. package/vendor-core/rolling-log-reassembly.js +4 -6
  353. package/vendor-core/run-comparison.js +86 -72
  354. package/vendor-core/scaling-sim.js +5 -7
  355. package/vendor-core/session-snapshot.js +1 -1
  356. package/vendor-core/shs-fetch.js +4 -6
  357. package/vendor-core/shs-load.js +9 -13
  358. package/vendor-core/shs-request.js +1 -1
  359. package/vendor-core/stage-quantiles.js +14 -0
  360. package/vendor-core/types.js +78 -18
  361. package/vendor-core/wasted-core-hours.js +7 -12
@@ -0,0 +1 @@
1
+ import{_ as t,o,c as d,a5 as a}from"./chunks/framework.DSg0KOwT.js";const p=JSON.parse('{"title":"Detector contract","description":"","frontmatter":{},"headers":[],"relativePath":"contributor-guide/architecture/detector-contract.md","filePath":"contributor-guide/architecture/detector-contract.md"}'),c={name:"contributor-guide/architecture/detector-contract.md"};function s(i,e,r,n,l,h){return o(),d("div",null,[...e[0]||(e[0]=[a('<h1 id="detector-contract" tabindex="-1">Detector contract <a class="header-anchor" href="#detector-contract" aria-label="Permalink to &quot;Detector contract&quot;">​</a></h1><p><code>packages/core/src/detectors.ts</code> is the single source of Spark-optimization logic: one declarative <code>DETECTORS</code> entry per pattern, each carrying <code>type</code>, <code>scope</code> (<code>stage</code> / <code>app</code> / <code>config</code> / <code>sql</code>), <code>order</code>, <code>fixEffort</code>, a <code>thresholds</code> object, impactBand/copy, a <code>docAnchor</code>, and a co-located <code>detect()</code> method. Both consumers are thin loops over that array:</p><ul><li><p><code>packages/core/src/analyzer.ts</code>: <code>analyze()</code> runs every entry regardless of scope, skipping only <code>inScorecard:false</code> ones; <code>auditConfig()</code> separately runs the <code>scope:&#39;config&#39;</code> entries. The four <code>configAudit</code> entries stay out of the bottleneck catalog because each sets <code>inScorecard:false</code>, not because of <code>scope:&#39;config&#39;</code>: a future config-scope detector without that flag would run through <code>analyze()</code> too. Each finding is stamped with its entry&#39;s <code>docAnchor</code>.</p></li><li><p><code>src/view/detector-registry.tsx</code>: a <code>REGISTRY: Record&lt;findingType, {component, region}&gt;</code> replaces <code>dashboard-renderer.js</code>&#39;s <code>render:</code> bindings, one entry per emitted finding type. <code>orderedWidgets()</code> walks <code>DETECTORS</code> ascending by <code>order</code>, then sorts <code>action</code>-region components before <code>reference</code>-region ones. Every <code>finding.type</code> maps to its own component now (the 2026-09 widget/finding-type 1:1 mapping redesign split six components that used to multiplex several types each: <code>TaskSkew</code> into <code>Skew</code>/<code>StageShape</code>/<code>TinyTask</code>; <code>ShuffleIO</code> narrowed to <code>shuffle</code> only, plus a new <code>PartitionSizing</code>; <code>Failures</code> into <code>StageFailed</code>/ <code>TaskFailures</code>/<code>RetryWaste</code>; <code>ExecutorTimeline</code> into <code>SlowHost</code>/ <code>StageSlowness</code>/<code>Straggler</code>/<code>SpeculationWaste</code>/<code>ColdStart</code> (its non-finding-driven executor-count chart moved to <code>ExecutorCountChart</code>, a <code>ReferenceSection</code> tile, not a <code>REGISTRY</code> entry); <code>MemoryUtilization</code> narrowed to <code>memoryUtilization</code> only, plus a new <code>ExecutorUtilization</code> for <code>utilization</code>; <code>PlanFindings</code> into <code>DuplicatePlanSubtree</code>/<code>SmallFiles</code>/ <code>UnderBroadcast</code>/<code>OverBroadcast</code>, dropping the dead <code>broadcastSizing</code> key entirely). No two <code>REGISTRY</code> entries share a <code>component</code> value any more.</p><p>Each widget component receives the full catalog and self-gates when it has nothing to show, rendering <code>null</code> or a muted &quot;no issue&quot; card for the always-visible ones. <code>orderedWidgets()</code> itself has no empty/non-empty branching, since it iterates the static <code>DETECTORS</code> import, not the runtime <code>catalog</code>.</p></li></ul><p>Thresholds live only in each entry&#39;s <code>thresholds</code>; see <a href="#bottleneck-thresholds-spec-§4">Bottleneck thresholds</a>.</p><h2 id="confidence-disclosure" tabindex="-1">Confidence disclosure <a class="header-anchor" href="#confidence-disclosure" aria-label="Permalink to &quot;Confidence disclosure&quot;">​</a></h2><p>A <code>Detector</code> entry (or the <code>Finding</code> it returns) may carry <code>confidence: &#39;low&#39; | &#39;medium&#39; | &#39;high&#39;</code> plus a <code>validationRequired</code> string. <code>RowStatusCluster</code> (<code>src/view/RowStatusCluster.tsx</code>) is the one place that renders it, gated to Advanced density: a plain &quot;&lt;confidence&gt; confidence&quot; badge whose tooltip carries the full <code>validationRequired</code> text. A finding with no <code>confidence</code> field renders identically to a fully-validated one, so every detector whose thresholds are our own unvalidated noise floor (marked <code>NOT SOURCED</code> in a code comment) should set both fields, not just the ones that happen to already have <code>RowStatusCluster</code> wired into their widget. <code>skew</code>, <code>straggler</code>, and <code>gc</code> set <code>confidence</code> for exactly this reason: their runtime-floor thresholds carry the same kind of unvalidated-noise-floor caveat <code>coreLocality</code>, <code>autoscalingChurn</code>, and <code>memoryUtilization</code>&#39;s <code>wasteModel</code> variant already disclose. None of these hardcode a single confidence value: each scales <code>&#39;low&#39; | &#39;medium&#39; | &#39;high&#39;</code> off how far the finding sits past its own detector&#39;s threshold, via a small named helper placed just above the <code>DETECTORS</code> array (e.g. <code>skewConfidence</code>, <code>coreLocalityConfidence</code>, <code>cachingReuseConfidence</code>) rather than an inline literal.</p><h2 id="the-fixeffort-field" tabindex="-1">The <code>fixEffort</code> field <a class="header-anchor" href="#the-fixeffort-field" aria-label="Permalink to &quot;The `fixEffort` field&quot;">​</a></h2><p>Each <code>Detector</code> entry also carries <code>fixEffort: &#39;config&#39; | &#39;code&#39; | &#39;rearchitect&#39;</code>, alongside <code>order</code> and <code>thresholds</code>: a rough estimate of how much work resolving the finding takes.</p><p>No view currently reads it. The quadrant impact/effort bucketing this field was meant to feed (<code>bucketFinding</code>/<code>effortTier</code>/<code>computeImpactMagnitude</code> in a since-deleted <code>src/quadrant-bucket.ts</code>, gated behind a <code>FIX_EFFORT_MAPPING_REVIEWED</code> flag that never flipped to <code>true</code>) was removed as dead code in the recommendations-consolidation redesign: <code>FixTheseFirst</code> (<code>src/view/widgets/FixTheseFirst.tsx</code>) ranks purely by impact magnitude. See <a href="./widget-rendering.html#widget-rendering-order-fixed-spec-§5">Widget rendering order</a> for how it ranks findings today.</p><p>Two shared helpers back multiple detectors and reports. <code>packages/core/src/plan-tree-walk.ts</code>&#39;s <code>walkPlanTree(root, visit, {dedupe})</code> is the iterative pre-order plan-tree traversal used by <code>detectors.ts</code> and every <code>plan-*.ts</code> module (<code>plan-summary.ts</code>, <code>plan-duration-attribution.ts</code>, <code>plan-node-detail.ts</code>, <code>plan-dot.ts</code>). <code>packages/core/src/core-count.ts</code>&#39;s <code>computeTotalCores(app, executorsAdded)</code> is the shared core-count logic used by <code>efficiency-model.ts</code>, <code>scaling-sim.ts</code>, and <code>wasted-core-hours.ts</code>. <code>detectors.ts</code>&#39;s own <code>utilization</code> and <code>memoryUtilization</code> entries use the same file&#39;s <code>computePeakConcurrentCores</code>/<code>computePeakConcurrentExecutorCount</code> instead: <code>computeTotalCores</code> sums every <code>ExecutorAdded</code> event with no regard for overlap, so under executor churn (spot preemption, <code>dynamicAllocation</code> replacement) it double-counts a churned executor&#39;s capacity against its replacement&#39;s; the peak-concurrent sweeps don&#39;t.</p><h2 id="cross-detector-suppression" tabindex="-1">Cross-detector suppression <a class="header-anchor" href="#cross-detector-suppression" aria-label="Permalink to &quot;Cross-detector suppression&quot;">​</a></h2><p>An entry may declare an optional <code>suppressWhen(finding, out)</code> method. <code>analyzer.ts</code>&#39;s <code>push()</code>, the single choke point every finding passes through, calls it per-finding, after the null guard and before the push, and drops the finding silently when it returns <code>true</code>. <code>out</code> is the findings accumulated so far. Since <code>analyze()</code>&#39;s loop is detector-outer / stage-inner, every finding from a detector declared earlier in <code>DETECTORS</code> is already in <code>out</code> by the time a later detector runs, for every stage. That makes the pattern purely declaration-order-driven: the suppressing detector must be declared earlier in the <code>DETECTORS</code> array than the suppressed one.</p><p><code>stageSlowness</code> uses this to defer to <code>slowHost</code>. It is spliced immediately after the <code>slowHost</code> entry regardless of its <code>order</code> field (<code>order</code> only controls render sequencing, not evaluation order), and <code>tests/analyzer.test.js</code>&#39;s &quot;detector contract&quot; suite asserts the array-index ordering so a future reorder can&#39;t silently break the suppression. The mechanism is deliberately minimal: a same-array, predicate-in-<code>push()</code> filter, not a general dependency graph. <code>auditConfig()</code>&#39;s own <code>push()</code> call is unaffected, since <code>scope:&#39;config&#39;</code> entries declare no <code>suppressWhen</code>.</p><h2 id="per-operator-duration-attribution" tabindex="-1">Per-operator duration attribution <a class="header-anchor" href="#per-operator-duration-attribution" aria-label="Permalink to &quot;Per-operator duration attribution&quot;">​</a></h2><p><code>packages/core/src/plan-duration-attribution.ts</code> (entry <code>attributeStageDurationToPlan(planTree, stagesById, sqlExec)</code>) approximates how a SQL execution&#39;s stage wall-time splits across plan operators, returning a <code>Map&lt;planNode, milliseconds&gt;</code>. It cuts the plan tree at Exchange boundaries into connected components: since the Exchange write/read split (<code>resolvePlanTree</code> in <code>event-handlers.ts</code> always synthesizes a <code>read</code> node wrapping a <code>write</code> node for every raw <code>Exchange</code>/<code>BroadcastExchange</code>), the cut is keyed off <code>PlanNode.exchangeRole === &#39;read&#39;</code> on the parent, not a name regex: the write half starts the new component, the read half stays in its parent&#39;s. A node with no <code>exchangeRole</code> at all (for example <code>ReusedExchange</code>, which is never split) never starts a new component on its own, unlike the old name-based regex, which matched any Exchange-family name regardless of split state. Each component receives a stable pre-order identity and separate parent/depth/traversal metadata; the identity itself does not encode its count of Exchange ancestors. It zips components deepest-first by that explicit depth against submission-ordered stage IDs, then apportions each matched stage&#39;s wall-time across that component&#39;s nodes by timing-metric weight, falling back to an even split when no node carries a timing metric.</p><p>This is best-effort inference, not measurement. Spark&#39;s event model exposes no ground truth for per-operator time within a stage; the Exchange-boundary segmentation and deepest-component-to-earliest-stage zip are heuristics. Treat the per-operator numbers as directional hints, never as authoritative timings, and do not build hard thresholds or findings on top of them.</p><h2 id="stage-id-attribution-for-plan-advisor-findings" tabindex="-1">Stage-ID attribution for Plan Advisor findings <a class="header-anchor" href="#stage-id-attribution-for-plan-advisor-findings" aria-label="Permalink to &quot;Stage-ID attribution for Plan Advisor findings&quot;">​</a></h2><p>The Plan Advisor detectors (<code>duplicatePlanSubtree</code>, <code>smallFiles</code>, <code>broadcastSizing</code> in <code>packages/core/src/detectors.ts</code>) each attribute their finding to a narrowed <code>stageIds</code> set rather than the whole SQL execution: <code>PlanNode.stageIds</code> is resolved once per plan tree at parse time by unioning, per node, every metric&#39;s accumulator ID against a <code>taskAccumStages: Map&lt;accumulatorId, Set&lt;stageId&gt;&gt;</code> built while parsing <code>TaskEnd</code> events, then clipping the result to the execution&#39;s own stage set. An accumulator ID occasionally points to a <em>different</em> execution&#39;s stages, e.g. a <code>ReusedSubquery</code> computed once and reused verbatim, and the clip prevents misattributing that other execution&#39;s work. Each detector unions its implicated node(s)&#39; <code>stageIds</code> and falls back to the execution-wide set only when no implicated node has any coverage; a finding never partially blends a narrowed set with the execution-wide one. When an execution has no stage universe at all (no jobs ever recorded against it, which is true for 42% of real-log SQL executions with a plan tree, typically job-less/driver-only executions), the clip drops every candidate stage ID instead of passing them through: every node in that execution&#39;s tree ends up with no <code>stageIds</code> anywhere, same &quot;coverage is partial&quot; framing as below. (An earlier version of this clip treated &quot;no stage universe&quot; as &quot;no clip,&quot; which let a foreign accumulator ID collision, e.g. the <code>ReusedSubquery</code> case above, leak another execution&#39;s stages into a job-less execution&#39;s nodes; the clip is now unconditional on <code>executionStageIds</code> being present.)</p><p>Coverage is partial by Spark&#39;s own design: whole-stage-codegen wrapper nodes (<code>InputAdapter</code>, and other purely structural passthrough markers) carry no accumulators at all, and <code>BroadcastExchangeExec</code>&#39;s own metrics are computed entirely on the driver and never appear on any <code>TaskEnd</code> (real Spark behavior). Since the Exchange write/read split, those driver-computed metrics live specifically on the synthesized <em>write</em> half (<code>exchangeRole: &#39;write&#39;</code>); the <em>read</em> half always carries <code>metrics: []</code>. The write half&#39;s immediate child, which does carry executor-side metrics, is unioned in instead, see <code>overBroadcast</code>&#39;s wiring. A <code>TaskEnd</code> arriving after its stage has already been finalized is also silently excluded from <code>taskAccumStages</code>, consistent with the parser&#39;s existing out-of-order tolerance elsewhere.</p><p><code>planTree</code> itself is kept current against Spark&#39;s adaptive query execution (AQE) re-plans: <code>SparkListenerSQLAdaptiveExecutionUpdate</code> events overwrite the execution&#39;s <code>sparkPlanInfo</code>/<code>physicalPlanDescription</code> last-write-wins, so accumulator-ID evidence is matched against the plan that actually ran rather than a stale pre-AQE snapshot.</p><p>No eviction/pruning is added to <code>taskAccumStages</code>, a deliberate choice, not an oversight: measured on real logs, it holds roughly 1,050 keys per compressed MB (9,850 keys on an 11.6 MB fixture, about 29,000 keys on a 28.1 MB fixture). Extrapolated to a 240MB+ log, the scale this tool targets (see <code>CLAUDE.md</code>), that is roughly 250,000 keys, around 45 MB of heap for an equivalent synthetic <code>Map&lt;number, Set&lt;number&gt;&gt;</code>. This heap estimate is still small relative to this tool&#39;s other in-memory state. It is higher, though, than the fixture-only measurements taken when this mechanism was built suggested.</p><p>Per-execution pruning (e.g. dropping a <code>taskAccumStages</code> entry once its stage finalizes or its owning SQL execution resolves, mirroring how <code>accumState</code> is cleared in <code>endSqlExecution</code>) is deliberately not done either: unlike <code>accumState</code>, <code>taskAccumStages</code> is one global, un-scoped map read by every execution&#39;s <code>resolvePlanTree</code> call, and the <code>ReusedSubquery</code> case above depends on a stage recorded under one execution still being visible when a later execution resolves. Pruning on any single execution&#39;s lifecycle would break that cross-execution lookup. What is bounded is the growth from a single pathological event: <code>TaskEndEventSchema</code>&#39;s <code>Accumulables</code> array is capped at <code>MAX_ACCUMULABLES_PER_TASK</code> (10,000, <code>event-schemas.ts</code>), well above any real plan&#39;s per-task metric count, so a single crafted <code>TaskEnd</code> can&#39;t grow the map past that per-event bound; a <code>TaskEnd</code> exceeding it fails schema validation and the line is skipped (counted in <code>skippedLines</code>) like any other malformed event.</p><h2 id="bottleneck-thresholds-spec-§4" tabindex="-1">Bottleneck thresholds (spec §4) <a class="header-anchor" href="#bottleneck-thresholds-spec-§4" aria-label="Permalink to &quot;Bottleneck thresholds (spec §4)&quot;">​</a></h2><p>Every change to <code>packages/core/src/detectors.ts</code> should reference this table.</p><p>Every finding&#39;s <code>impactBand</code> comes from one of two places. For any finding whose <code>impactEstimate</code> carries a <code>wallClock</code> estimate (the common case for most rules below), <code>analyzer.ts</code> calls <code>deriveImpactBand()</code> (<code>packages/core/src/impact-band.ts</code>) immediately after <code>estimateImpact()</code>, which sets <code>.impactBand</code> purely from <code>wallClock.high</code> as a fraction of the app&#39;s total duration (<code>&gt;= 2%</code> critical, <code>&gt;= 0.5%</code> warning, else info: the same <code>floorPctWarn</code>/<code>floorPctCrit</code> values <code>skew</code>/<code>straggler</code> use for their own thresholds below). For those rules, the table below documents their firing gate plus their fixed fallback constant, which surfaces only when this run&#39;s finding of that type didn&#39;t get a wallClock estimate (a stage excluded from the occupancy sweep). For rules whose finding type never gets a wallClock estimate (<code>resourceOnly</code>/<code>informational</code> basis, e.g. <code>configAudit</code>, or a rule that keeps its own ratio-tiered classification per the design&#39;s Decision 2, e.g. <code>failures</code>), the full threshold table below is the real, displayed classification: <code>detectors.ts</code> sets <code>impactBand</code> directly and nothing overwrites it. <code>partitionSizing</code>&#39;s <code>maxPartitionTooBig</code> rule is a third case: it does carry a <code>wallClock</code> estimate but is explicitly exempted in <code>deriveImpactBand()</code> because it&#39;s a hardcoded-critical OOM/crash-risk safety signal, not a time-recovery one, so <code>detectors.ts</code>&#39;s own classification stands regardless of how small that estimate is relative to the run.</p><h3 id="fixed-fallback-only-usually-wallclock-derived-instead" tabindex="-1">Fixed fallback only (usually wallClock-derived instead) <a class="header-anchor" href="#fixed-fallback-only-usually-wallclock-derived-instead" aria-label="Permalink to &quot;Fixed fallback only (usually wallClock-derived instead)&quot;">​</a></h3><p>These rules&#39; <em>band</em> tiers were deleted from <code>packages/core/src/detectors.ts</code> (they were always overwritten by <code>deriveImpactBand</code> whenever a wallClock estimate was available); the constant in the last column is only a floor-case fallback. Their <em>firing</em> gate is untouched and still lives in each entry&#39;s <code>thresholds</code> object: it decides whether the rule reports anything, so it stays documented here in full.</p><table tabindex="0"><thead><tr><th>Rule</th><th>Fires when</th><th>Fallback</th></tr></thead><tbody><tr><td>Task skew</td><td><code>taskDurationP95 / taskDurationP50 &gt; 3×</code> (<code>taskDurationMax / P50</code> for stages under <code>minTasksForP95</code> = 20 tasks), <strong>and</strong> the occupancy-clipped P95−P50 (or max−P50) delta is ≥ <code>floorPctWarn</code> = 0.5% of app runtime</td><td><code>warning</code></td></tr><tr><td>Shuffle read</td><td><code>shuffleReadBytes &gt; minBytes</code> = 50 MiB</td><td><code>info</code></td></tr><tr><td>Partition sizing: skew</td><td><code>shuffleReadMax &gt; 5×</code> <code>shuffleReadP50</code> <strong>and</strong> <code>shuffleReadMax &gt; 256 MiB</code></td><td><code>warning</code></td></tr><tr><td>Partition sizing: low parallelism</td><td><code>shuffleReadBytes ≥ 1 GiB</code> <strong>and</strong> <code>taskCount ≤ 7</code></td><td><code>warning</code></td></tr><tr><td>Partition sizing: oversized partition</td><td><code>shuffleReadMax ≥ 5 GiB</code></td><td><code>critical</code></td></tr><tr><td>GC</td><td><code>executorRunTime ≥ minRunTimeMs</code> = 10 s <strong>and</strong> <code>gcPct &gt; 10%</code></td><td><code>warning</code></td></tr><tr><td>GC (low / cost)</td><td><code>executorRunTime ≥ 10 s</code> <strong>and</strong> <code>gcPct &lt; lowInfoPct100</code> = 5% (checked only when the GC row above did not fire)</td><td><code>info</code></td></tr><tr><td>Spill (magnitude v2)</td><td>any non-zero <code>memoryBytesSpilled</code>. The magnitude sub-table below classifies <em>how much</em>, but does not gate firing</td><td><code>warning</code></td></tr><tr><td>Cold start</td><td><code>firstStageSubmittedAt − app.startTime &gt; gapSeconds</code> = 30 s</td><td><code>warning</code></td></tr><tr><td>Slow host: mean-duration ratio</td><td>stage has ≥ <code>minHosts</code> = 3 hosts (or executors) and ≥ <code>minTasks</code> = 15 tasks; then per host: mean task duration / overall median ≥ <code>ratioWarn</code> = 2.0× <strong>and</strong> host task-share ≥ <code>minShare</code> = 20% <strong>and</strong> host mean ≥ <code>floorMs</code> = 1000 ms (absolute-magnitude floor, rules out sub-second noise)</td><td><code>warning</code></td></tr><tr><td>Slow host: duration-share</td><td>same stage gate as the row above; then per host: ≥ <code>shareWarn</code> = 75% of the stage&#39;s total task-duration <strong>and</strong> ≥ <code>taskShareWarn</code> = 50% of its task count</td><td><code>warning</code></td></tr><tr><td>Stage slowness: absolute fallback, suppressed when <code>slowHost</code> already fired</td><td>stage wall-clock duration ≥ <code>infoMin</code> = 15 min</td><td><code>info</code></td></tr><tr><td>Straggler / speculative-execution</td><td><code>taskCount ≥ minTasks</code> = 10, <strong>and</strong> either any speculative task ran <strong>or</strong> straggler share &gt; <code>shareWarn</code> = 5%. <code>warnPct</code>/<code>critPct</code> (10%/20% speculative share) and <code>floorPctWarn</code>/<code>floorPctCrit</code> (0.5%/2% of app runtime) no longer set the band; they rank the straggler-vs-speculative tiers that pick which <em>metric</em> the finding reports</td><td><code>info</code></td></tr><tr><td>Speculation waste (new)</td><td><code>speculationWastedAttempts ≥ minWasted</code> = 5 <strong>and</strong> <code>speculationWasteMs ≥ minWasteMs</code> = 60 s</td><td><code>warning</code></td></tr><tr><td>Retry waste</td><td><code>wastedAttempts ≥ minWasted</code> = 3 <strong>and</strong> <code>retryWasteMs ≥ minWasteMs</code> = 30 s, on a stage that still completed</td><td><code>warning</code></td></tr><tr><td>Tiny tasks</td><td><code>taskCount ≥ minTasks</code> = 100 <strong>and</strong> <code>taskDurationP50 ≤ maxP50</code> = 500 ms <strong>and</strong> <code>taskDurationP95 ≤ maxP95</code> = 1000 ms</td><td><code>info</code></td></tr><tr><td>Duplicate plan subtree</td><td>a subtree of ≥ <code>minSubtreeSize</code> = 3 nodes whose shape fingerprint repeats ≥ <code>minOccurrences</code> = 2× in the plan</td><td><code>warning</code></td></tr><tr><td>Small files read/write</td><td>per read/write side: file count &gt; <code>minFiles</code> = 100 <strong>and</strong> average file size &lt; <code>maxAvgFileSizeMB</code> = 3 MiB</td><td><code>warning</code></td></tr><tr><td>Broadcast sizing: missed</td><td>a 2-child <code>SortMergeJoin</code> whose smaller side is &lt; 10 MiB (unconditional), or &lt; 100 MiB with the larger side &gt; 10 GiB, or &lt; 1 GiB with larger &gt; 300 GiB, or &lt; 5 GiB with larger &gt; 1 TiB (<code>broadcastTiers</code> × <code>comparisonTiers</code>)</td><td><code>info</code></td></tr><tr><td>Broadcast sizing: oversized</td><td>a <code>BroadcastExchange</code> node whose <code>data size</code> metric &gt; <code>overBroadcastBytes</code> = 1 GiB</td><td><code>warning</code></td></tr></tbody></table><h4 id="spill-magnitude-tiers" tabindex="-1">Spill magnitude tiers <a class="header-anchor" href="#spill-magnitude-tiers" aria-label="Permalink to &quot;Spill magnitude tiers&quot;">​</a></h4><p>The spill row&#39;s band is a fixed <code>warning</code> fallback, but <code>computeSpillMagnitude</code> (<code>packages/core/src/detectors.ts</code>) still runs on every spill finding and sets its <code>spillMagnitude</code> field, which the Spill widget displays. Its tiers, in evaluation order (first match wins, <code>null</code> when nothing matches):</p><table tabindex="0"><thead><tr><th>Condition</th><th>Magnitude</th></tr></thead><tbody><tr><td>Single-task stage: <code>spillDiskMax ≥ singleTaskDiskGiB</code> = 1 GiB <strong>or</strong> <code>spillMemMax ≥ singleTaskMemGiB</code> = 4 GiB</td><td><code>severe</code></td></tr><tr><td>Multi-task stage: <code>spillDiskMax ≥ highDiskGiB</code> = 1 GiB, <strong>or</strong> <code>spillDiskMax / taskCount ≥ highTaskDiskMB</code> = 512 MiB (per-task proxy), <strong>or</strong> <code>spillMemMax ≥ highMemGiB</code> = 4 GiB</td><td><code>high</code></td></tr><tr><td>Multi-task stage: <code>spillDiskMax ≥ medDiskMB</code> = 256 MiB <strong>or</strong> <code>spillMemMax ≥ medMemGiB</code> = 1 GiB</td><td><code>medium</code></td></tr><tr><td>Skew (<code>taskCount ≥ skewMinTasks</code> = 10): <code>spillDiskMax / spillDiskP50 &gt; skewRatio</code> = 5× <strong>and</strong> <code>spillDiskMax ≥ skewDiskFloorMB</code> = 128 MiB</td><td><code>high</code></td></tr><tr><td>Skew (<code>taskCount ≥ 10</code>): <code>spillMemMax / spillMemP50 &gt; 5×</code> <strong>and</strong> <code>spillMemMax ≥ skewMemFloorMB</code> = 256 MiB</td><td><code>medium</code></td></tr></tbody></table><p>Disk spill is weighted worse than memory spill by design: the memory thresholds sit well above their disk counterparts at every tier.</p><h3 id="full-threshold-table-never-wallclock-derived" tabindex="-1">Full threshold table (never wallClock-derived) <a class="header-anchor" href="#full-threshold-table-never-wallclock-derived" aria-label="Permalink to &quot;Full threshold table (never wallClock-derived)&quot;">​</a></h3><table tabindex="0"><thead><tr><th>Rule</th><th>Warning</th><th>Critical</th></tr></thead><tbody><tr><td>Stage shape: PRatio</td><td><code>taskCount / totalCores &lt; 0.5</code> (info, under-parallelized)</td><td>none</td></tr><tr><td>Stage shape: OIRatio</td><td><code>outputBytes / inputBytes &gt; 10×</code> (info, data explosion)</td><td>none</td></tr><tr><td>Stage shape: TaskStageSkew</td><td><code>taskDurationMax / stageDuration &gt; 3×</code> (info)</td><td>none</td></tr><tr><td>Failed tasks</td><td>failure rate &gt; 5% (min 10 tasks)</td><td>&gt; 20%</td></tr><tr><td>Stage failed outright</td><td>none</td><td>any <code>stageFailureReason</code> present</td></tr><tr><td>Slow host: multi-dimensional</td><td>max/median ratio across taskTime/inputBytes/shuffleBytes/storageMemory ≥ 1.33× (info); each dimension&#39;s sample must also clear an absolute floor (1000 ms for taskTime, 64 MiB for the byte dimensions)</td><td>≥ 3.16× warning, ≥ 10× critical</td></tr><tr><td>Utilization</td><td>avg active executors / peak &lt; 60% (info)</td><td>none</td></tr><tr><td>Autoscaling churn: short-lived executors (design spike, unvalidated thresholds)</td><td>&gt; 30% of executors alive under 2 min (min 5 executors)</td><td>&gt; 60%</td></tr><tr><td>Job failure rate</td><td>≥ 30% (≥ 10% info)</td><td>≥ 50%</td></tr><tr><td>Idle cores</td><td>busy-core-time / (peak cores × wall-clock) idle &gt; 50% (warning)</td><td>none</td></tr><tr><td>Memory band</td><td>peak heap / allocated &gt; 95% too-small (warning); &lt; 70% over-provisioned (info)</td><td>none</td></tr><tr><td>Caching opportunity</td><td>RDD read across ≥3 stages without <code>.persist()</code></td><td>none (single tier, info)</td></tr><tr><td>Cache utilization: partial caching (this repo)</td><td><code>numCachedPartitions / numPartitions &lt; 0.90</code> (info)</td><td><code>&lt; 0.50</code> (warning)</td></tr><tr><td>Cache utilization: disk spillover (this repo)</td><td><code>diskSize / (memorySize + diskSize) &gt; 0.15</code> (info), <code>MEMORY_AND_DISK*</code> only</td><td><code>&gt; 0.40</code> (warning)</td></tr></tbody></table><p>Spill classification: ≥80% tasks with zero spill → <code>skew</code>; &lt;20% zero → <code>volume</code>; else <code>unclassified</code>. The classification badge is always shown in both compact and expanded spill widget states. This is independent of the magnitude tiers above: classification says <em>what kind</em> of spill, magnitude says <em>how much</em>.</p><h4 id="evidence-fields-stagefailed-retrywaste" tabindex="-1">Evidence fields (<code>stageFailed</code> / <code>retryWaste</code>) <a class="header-anchor" href="#evidence-fields-stagefailed-retrywaste" aria-label="Permalink to &quot;Evidence fields (`stageFailed` / `retryWaste`)&quot;">​</a></h4><p>Neither entry&#39;s <code>detect()</code> used to put anything beyond a scalar <code>metric</code>/ <code>value</code> on its finding. Both now also attach:</p><ul><li><code>numTasks</code>: <code>stage.taskCount</code> at detection time.</li><li><code>memoryBytesSpilled</code>: <code>stage.memoryBytesSpilled</code> at detection time.</li><li><code>stageFailed</code> only: <code>failedTaskDetails</code>: up to 20 <code>FailedTaskSample</code> records (<code>taskId</code>, <code>attemptNumber</code>, <code>host</code>, <code>executorId</code>, <code>reason</code>, <code>peakExecMem</code>, <code>memSpilled</code>, <code>shuffleWrite</code>) for tasks still marked failed when the stage was finalized (<code>finalizeStage</code>, <code>stage-quantiles.ts</code>).</li><li><code>retryWaste</code> only: <code>retriedTaskDetails</code>: up to 20 <code>FailedTaskSample</code> records for attempts discarded by the retry-dedup logic in <code>accumulateTask</code> (<code>event-handlers.ts</code>): captured at the moment they&#39;d otherwise be thrown away, since by finalize time only the winning attempt survives.</li></ul><p>Both sample arrays are capped at 20 entries, filled in first-encountered order (finalize order for <code>failedTaskDetails</code>, discard order for <code>retriedTaskDetails</code>), not spread across distinct hosts/executors: a stage with failures clustered on one bad host could fill the cap before a more informative failure elsewhere in the stage is ever sampled. This is the first evidence-shape documentation in this file; no other finding type has one yet.</p>',39)])])}const g=t(c,[["render",s]]);export{p as __pageData,g as default};
@@ -0,0 +1 @@
1
+ import{_ as t,o,c as d,a5 as a}from"./chunks/framework.DSg0KOwT.js";const p=JSON.parse('{"title":"Detector contract","description":"","frontmatter":{},"headers":[],"relativePath":"contributor-guide/architecture/detector-contract.md","filePath":"contributor-guide/architecture/detector-contract.md"}'),c={name:"contributor-guide/architecture/detector-contract.md"};function s(i,e,r,n,l,h){return o(),d("div",null,[...e[0]||(e[0]=[a("",39)])])}const g=t(c,[["render",s]]);export{p as __pageData,g as default};
@@ -0,0 +1 @@
1
+ import{_ as o,o as t,c as a,a5 as d}from"./chunks/framework.DSg0KOwT.js";const g=JSON.parse('{"title":"Drill-down","description":"","frontmatter":{},"headers":[],"relativePath":"contributor-guide/architecture/drill-down.md","filePath":"contributor-guide/architecture/drill-down.md"}'),n={name:"contributor-guide/architecture/drill-down.md"};function s(c,e,i,r,l,h){return t(),a("div",null,[...e[0]||(e[0]=[d('<h1 id="drill-down" tabindex="-1">Drill-down <a class="header-anchor" href="#drill-down" aria-label="Permalink to &quot;Drill-down&quot;">​</a></h1><p><code>StageDetailProvider</code> (<code>src/view/StageDetailContext.tsx</code>) replaces the legacy <code>openStageDetail</code> <code>window</code> <code>CustomEvent</code> with React context. Any component calls <code>useStageDetail().openStage(stageId)</code> to open <code>StageDetailDialog.tsx</code>, a Radix/shadcn <code>Dialog</code>, for that stage. The callers are StagePill, Timeline, StageTable, and (indirectly, via an embedded <code>StageHeader</code>/<code>StagePill</code>) Skew&#39;s per-stage rows.</p><p>The legacy <code>drillDownToStage</code> force-expand-and-<code>scrollIntoView</code> event has no replacement. No code ever dispatched it, so it was dormant even before the migration.</p><h2 id="plan-dot-serialization" tabindex="-1">Plan DOT serialization <a class="header-anchor" href="#plan-dot-serialization" aria-label="Permalink to &quot;Plan DOT serialization&quot;">​</a></h2><p><code>packages/core/src/plan-dot.ts</code> (entry <code>planTreeToDot(planTree, { title })</code>) serializes a resolved <code>planTree</code> to a Graphviz DOT string. It is dependency-free string building. A first pass walks the tree assigning stable node ids and <code>label</code> (name, plus <code>detail</code> on a second line when it differs). A second pass emits the parent→child edges once ids are known. The graph is laid out <code>rankdir=BT</code>, leaves at the bottom, matching Spark&#39;s own plan orientation.</p><p>It carries no metric annotation: pure structure. Adding metric annotation is a deferred roadmap item. A null plan returns an empty string. There is no download/export UI for this output anymore (removed); <code>PlanView.tsx</code> calls it only to decide whether a stage&#39;s plan tree can render as a graph at all, and a non-empty result gates the &quot;View plan graph&quot; button.</p><h2 id="plan-graph-view" tabindex="-1">Plan graph view <a class="header-anchor" href="#plan-graph-view" aria-label="Permalink to &quot;Plan graph view&quot;">​</a></h2><p><code>buildPlanGraphModel(planTree, opts)</code> (<code>packages/core/src/plan-graph-model.ts</code>) flattens a resolved <code>planTree</code> into a <code>{ nodes, edges, segmentIndex, segmentCount, scope, segmentStageIds }</code> graph shape. <code>src/view/PlanGraphRoute.tsx</code> renders it with <code>@xyflow/react</code> (React Flow, pan/zoom/viewport/MiniMap) and <code>@dagrejs/dagre</code> (node layout, <code>src/view/plan-graph/dagre-layout.ts</code>, <code>rankdir: &#39;RL&#39;</code>).</p><p>Every <code>PlanGraphEdge</code> points <code>source: parentId, target: id</code> (parent/consumer → child/producer), and dagre places an edge&#39;s source at the higher-rank end. So <code>RL</code>, rather than the more intuitive-looking <code>LR</code>, is what lands reads/scans (targets, computed first) on the left and the final write/root (source) on the right, matching the left-to-right reading order of the plan&#39;s data flow. Each node&#39;s React Flow <code>Handle</code>s follow the same horizontal routing: <code>type=&quot;target&quot;</code> on <code>Position.Right</code> (its parent sits to the right) and <code>type=&quot;source&quot;</code> on <code>Position.Left</code> (its children sit to the left), rather than the top/bottom anchors a vertical <code>TB</code>/<code>BT</code> layout would use.</p><p><code>PlanGraphNode.tsx</code> renders every node at a fixed <code>NODE_WIDTH × NODE_HEIGHT</code> (220×90, also what dagre lays the graph out around) with <code>truncate</code>/<code>title</code> on every text field. An operator with an unusually long label/detail/metric string can&#39;t inflate the box and overlap neighbors; it ellipsizes instead, full text on hover.</p><p>Every raw <code>Exchange</code>/<code>BroadcastExchange</code> plan node is split by <code>resolvePlanTree</code> (<code>packages/core/src/event-handlers.ts</code>) into paired write/read halves sharing a <code>sourceNodeId</code> once flattened into the graph. <code>ReusedExchange</code> is classified as an exchange for display purposes but is never split: it carries no <code>exchangeRole</code> and stays a single node. The write half is grouped with its children&#39;s producer component; the read half is grouped with its parent&#39;s consumer component. Both halves get their own entry in <code>buildDurationMap</code>&#39;s duration share now (the write half no longer hard-codes to null): <code>PlanGraphNode.tsx</code> only suppresses the displayed value for the read half, since the read half occupies the exact tree position the original unsplit node used to.</p><p>Default scope is <code>segment</code>: one Exchange-bounded slice of the plan, resolved via <code>computeSegments</code>/<code>zipSegmentsToStages</code> (<code>packages/core/src/plan-duration-attribution.ts</code>, shared with <code>attributeStageDurationToPlan</code>). <code>computeSegments</code> gives every connected component a stable pre-order identity plus separate parent/depth/traversal metadata. The numeric identity does not encode Exchange-ancestor depth: <code>zipSegmentsToStages</code> orders by the explicit depth metadata (deepest first, plan traversal order for ties), and the display-only fallback uses component-tree edge distance to find the nearest strictly paired component.</p><p>The view has an opt-in &quot;Expand to full plan&quot; toggle gated by a 300-node confirmation dialog (<code>ExpandConfirmDialog.tsx</code>). A segment-lookup failure that would otherwise render an unguarded full plan is routed through the same guardrail rather than bypassing it. That same failure can also resolve to a full-scope model <em>below</em> the guardrail threshold, rendering immediately with no dialog and no <code>requestedScope</code> change. Since it&#39;s a permanent property of that stage&#39;s plan (<code>buildPlanGraphModel</code> is deterministic per <code>(planTree, stageId, appModel)</code>), there is no segment view left for that stage to switch back to. So the toggle, still correctly labeled &quot;Back to segment view&quot; per <code>model?.scope</code>, renders <code>disabled</code> rather than silently no-opping on click. Only the explicit-expand path (<code>requestedScope === &#39;full&#39;</code>) leaves it enabled.</p><p>The four Plan Advisor detectors (<code>duplicatePlanSubtree</code>, <code>smallFiles</code>, <code>overBroadcast</code>, <code>underBroadcast</code>, in <code>packages/core/src/detectors.ts</code>) set <code>Finding.planNodeIds</code>, an unambiguous pointer to the specific plan-tree node(s) each finding is about (a whole subtree&#39;s root(s) for <code>duplicatePlanSubtree</code>, the flagged scan/write node for <code>smallFiles</code>, the join/broadcast node(s) for the broadcast pair). <code>buildPlanGraphModel</code> indexes <code>findings</code> by <code>planNodeIds</code> and attaches each node&#39;s matches to its <code>PlanGraphNodeData.findings</code>, scoped to the current SQL execution (see below), and <code>PlanGraphNode.tsx</code> renders a corner badge from that per-node list. The badge follows the same dot+tag problem-flagging vocabulary as the rest of the dashboard: an <code>ImpactDot</code> plus the ALL-CAPS <code>typeTag</code> (<code>PLAN</code> for every current plan-node finding), colored by the worst band across the node&#39;s findings (critical &gt; warning &gt; info, via the local <code>worstFinding</code> helper), plus a count when the node carries more than one finding. Hovering or focusing the badge opens a tooltip listing each finding by its <code>findingActionLabel</code> (e.g. &quot;Dedupe repeated subtree&quot;, &quot;Compact small files&quot;), so the node discloses which findings hit it without leaving the graph. Every other finding type still has no plan-node pointer and renders at the stage level only, via <code>Finding.stageId</code>/<code>Finding.stageIds</code>.</p><p>Plan-node ids are only unique within one SQL execution&#39;s tree, since <code>resolvePlanTree</code> resets its <code>n0, n1, ...</code> id counter on every call (once per SQL execution), so <code>buildPlanGraphModel</code> filters <code>findings</code> to <code>finding.executionId === sqlExecutionId</code> (the execution the stage being graphed belongs to) before indexing by node id. Without that filter, two unrelated executions&#39; trees can both contain a node named e.g. <code>n1</code>, and a finding from one would badge onto the other&#39;s same-named node.</p><p>Two other per-node UI features render independently of findings:</p><ul><li>A traffic-light <strong>duration heat bar</strong> on each node (<code>plan-graph-heat.ts</code>&#39;s <code>heatBand</code>, consumed by <code>PlanGraphNode.tsx</code>) bands the node&#39;s duration share into critical/warning/info, reusing the impact-band color tokens (see the Plan Violet exception noted in <code>DESIGN.md</code>).</li><li>A <strong>duration-attribution mode toggle</strong> (<code>PlanGraphDurationModeControl.tsx</code>) switches every node&#39;s displayed share between &quot;Node only&quot; (exclusive) and &quot;Node + descendants&quot; (inclusive), threaded into the model-build/cache key as <code>durationMode</code>.</li></ul><p>One more identity subtlety: a split <code>Exchange</code>/<code>BroadcastExchange</code> pair (the write/read halves <code>resolvePlanTree</code> synthesizes, see above) counts as <strong>one</strong> node for <code>duplicatePlanSubtree</code>&#39;s subtree-size and occurrence counting. <code>computePlanShapes</code> in <code>detectors.ts</code> walks straight through the read half to the write half&#39;s real children, so every real Exchange in a matched subtree is counted once instead of twice.</p><p>The segment-level group box (<code>PlanGraphSegmentGroupNode.tsx</code>) always renders, even in the default single-stage view where it&#39;s the only box on screen. It is headered with its stage id (via <code>segmentStageIds</code>) and a duration chip, or an em-dash placeholder when the segment has no attributed duration. It also carries one finding chip (<code>PlanGraphFindingChip</code>) per finding on this stage, but only when there&#39;s no outer stage box to carry them instead (i.e. only in the default single-stage view). Each chip is a <code>TagBadge</code> (the ALL-CAPS tag) followed by the finding&#39;s compact magnitude and recoverable-time detail from <code>formatFindingChipDetail</code> (e.g. &quot;SPILL 4.2 GB · ~38s&quot;: the finding&#39;s own <code>value</code>/<code>metric</code> and its <code>impactEstimate.wallClock</code>), or the bare tag when the finding carries neither.</p><p>The expanded full-plan view draws two nested layers of background group boxes (Dagre <code>compound: true</code> for spacing, <code>computeGroupBounds</code> for both): the same inner segment box described above, plus an outer solid tinted container per distinct stage id (<code>PlanGraphStageGroupNode.tsx</code>, layered under the inner segment box, which is in turn under the plan nodes; all three sit above the edge layer so a routed edge can&#39;t paint over a box&#39;s finding chips) merging every segment zipped to that stage, so a stage split across several Exchange-bounded segments still reads as one unit of work. In this expanded view the outer stage box, not the nested segment box, carries the stage&#39;s finding chips, since findings are stage-scoped rather than segment-scoped. Each stage box shows only findings matched to its own stage through <code>Finding.stageId</code> or <code>Finding.stageIds</code>; the default segment scope uses the same matching for its displayed stage.</p><p>The outer layer&#39;s own corner tag is positioned opposite the segment box&#39;s top-left header, because a stage with just one segment (the common case) would otherwise show the same &quot;Stage N&quot; header twice, stacked. <code>STAGE_GROUP_PADDING_Y</code> (64, no reserved header height) is sized to strictly exceed the segment layer&#39;s own reserved offset (<code>GROUP_PADDING + GROUP_HEADER_HEIGHT</code> = 56 at the top), while <code>STAGE_GROUP_PADDING_X</code> (32) is kept close to <code>GROUP_PADDING</code>&#39;s own 24px floor so the outer box doesn&#39;t visibly balloon sideways. Both still strictly contain the outer box around its nested segment box(es).</p><p>Every global control lives on one <strong>vertical control rail</strong> down the left edge (<code>PlanGraphControlRail.tsx</code>), grouped View / Navigate / Display, so nothing floats in its own corner. View is zoom in/out and fit (replacing React Flow&#39;s <code>Controls</code>); Navigate is &quot;Next worst duration&quot; and &quot;Next problem&quot; (cycling to the worst-share node and the worst-finding stage, via <code>setCenter</code>); Display is the node-filter/duration Settings popover (<code>PlanGraphSettingsControl</code> with its <code>iconOnly</code> rail variant, moved off the topbar), plus the legend and minimap toggles. The rail renders inside <code>ReactFlowProvider</code> alongside <code>&lt;ReactFlow&gt;</code>, so its <code>useReactFlow</code> zoom/center calls drive the same instance.</p><p>Clicking a node opens the <strong>detail inspector</strong> (<code>PlanGraphNodeDetail.tsx</code>), a right-docked panel (not a floating card) that reflows the graph rather than covering it. Each node box is a fixed size and truncates every field to one line, showing a single <code>primaryMetric</code>; the inspector is where the whole operator is legible: category and segment, duration share, the node&#39;s findings (dot + tag + <code>findingActionLabel</code>), the operator&#39;s <strong>full metric set</strong> (<code>PlanGraphNodeData.metrics</code>, every metric formatted through <code>formatPlanMetricValue</code>), and the <strong>complete plan text</strong> (<code>PlanGraphNodeData.detailText</code>, the untruncated <code>PlanNode.detail</code>). For a split Exchange half the inspector adds an <strong>Exchange section</strong>: which half is in view, the shuffle volume across the boundary (<code>PlanGraphNodeData.exchangeShuffleBytes</code>, the same producing-stage bytes the pairing edge is weighted by, mirrored onto both halves so it shows whichever half is open, or &quot;Broadcast, no shuffle&quot;), and a <strong>jump to the paired half</strong> (<code>PlanGraphNodeData.pairedNodeId</code>). Because the two halves always sit in different segments, the jump selects the partner directly when it is already on screen (full plan) and otherwise expands to the full plan first, then selects it (<code>handleJumpToPaired</code> in <code>PlanGraphRoute.tsx</code>, carried past the selection-reset effect by a pending-jump state); either way the canvas recenters on it via a <code>centerRequest</code> token. The selected node id lives in <code>PlanGraphRoute.tsx</code>, not the canvas, so the route&#39;s one Escape handler closes the inspector first and the whole route only on a second press. Nodes are <code>draggable: false</code> (a read-only, auto-laid-out graph); they click to inspect but don&#39;t move.</p><p>The remaining legibility aids sit on the canvas itself:</p><ul><li>The <strong>MiniMap</strong> (bottom-right, toggled from the rail) colors each node by the worst finding band on it (<code>planGraphMiniMapNodeColor</code>, <code>plan-graph-minimap.ts</code>), so the overview shows where the problems are; a node with no finding keeps the neutral plan color and the large group boxes recede into a muted fill.</li><li>The rail&#39;s legend toggle reveals a <strong>legend</strong> panel (<code>PlanGraphLegend.tsx</code>) keying the operator icons, the heat-bar colors, the shuffle-weighted edge thickness, and the segment-vs-stage box layers.</li><li>When the category filter hides every operator in view, a <strong>status hint</strong> (top-center) names the count hidden and points at Settings, instead of leaving only empty group boxes on screen.</li></ul><p>The view is reached via a &quot;View plan graph&quot; button in <code>PlanView.tsx</code>&#39;s toolbar, gated by the same <code>if (dot)</code> check described in &quot;Plan DOT serialization&quot; above. It opens a <code>planGraph: { active, stageId }</code> Zustand slice (<code>openPlanGraph</code>/<code>closePlanGraph</code>, <code>src/store/store.ts</code>) driving a top-level <code>AppRoutes</code> branch in <code>src/App.tsx</code>, modeled directly on the existing <code>comparison</code>/<code>RunComparisonRoute</code> full-takeover pattern.</p><p><code>buildPlanGraphModel</code>&#39;s output is memoized per <code>(activeFileId, stageId, scope)</code> in <code>PlanGraphRoute.tsx</code>, since <code>stageId</code> alone isn&#39;t unique across loaded runs and <code>applySnapshot</code> mutates <code>appModel</code> in place rather than replacing it (see <a href="./state-and-history.html#state-model">State model</a>). The memo cache is a module-level <code>Map</code>, so it survives across route open/close: re-opening the same stage in the same run reuses the cached model instead of rebuilding it. It must still be evicted on a fresh parse or reload. <code>resetModel()</code> (<code>store.ts</code>) bumps a <code>modelResetCount</code> counter for exactly this purpose, and <code>PlanGraphRoute.tsx</code> subscribes to it to clear the cache. A counter rather than a direct call, since <code>store.ts</code> has no view-layer imports anywhere else and importing a <code>.tsx</code> module there would invert that dependency direction.</p><h2 id="disclosure-hierarchy" tabindex="-1">Disclosure hierarchy <a class="header-anchor" href="#disclosure-hierarchy" aria-label="Permalink to &quot;Disclosure hierarchy&quot;">​</a></h2><p>The Summary/Context/Details 3-tier framing (collapsed lead metric → expanded widget body → per-stage <code>StageDetailDialog</code>) does not apply uniformly across the 28 registry widgets (<code>src/view/detector-registry.tsx</code>). 14 have a stage-anchored Details tier reachable via <code>StagePill</code>/<code>StagePillGroup</code>: Skew, StageShape, TinyTask (all split from TaskSkew), ShuffleIO, PartitionSizing (split from ShuffleIO), Spill, GcPressure, StageFailed, TaskFailures, RetryWaste (split from Failures), SlowHost, StageSlowness, Straggler, and SpeculationWaste (split from ExecutorTimeline). The other 14 are app/sql-scope with no stage to drill into, by design, so they stop at Summary/Context: MemoryUtilization, ExecutorUtilization (split from the same widget as MemoryUtilization; <code>utilization</code> is an app-wide average, no stage), JobFailures, ConfigAudit, CacheUtilization, CoreUsageArea, AutoscalingChurn, CachingOpportunity (<code>scope: &#39;app&#39;</code>, <code>stageId: null</code> on both its finding constructions, so it has no stage to anchor to despite reading like a per-stage widget), IncompleteRun, DuplicatePlanSubtree, SmallFiles, UnderBroadcast, OverBroadcast (all split from PlanFindings, sql-scope, spanning multiple stages via <code>stageIds</code> rather than one <code>stageId</code>), and ColdStart (split from ExecutorTimeline, but unlike its four siblings above, app-scoped with no <code>stageId</code>).</p><h2 id="reference-panel" tabindex="-1">Reference panel <a class="header-anchor" href="#reference-panel" aria-label="Permalink to &quot;Reference panel&quot;">​</a></h2><p>The topbar&#39;s &quot;Reference&quot; button and <code>DocsLink</code> (<code>src/view/DocsContext.tsx</code>) open a shadcn <code>Sheet</code> (<code>src/view/DocsSheet.tsx</code>, mounted once inside <code>DocsProvider</code>/<code>Dashboard.tsx</code>) that iframes a docs-site (VitePress) page, built from the tuning reference committed under <code>packages/core/src/docs-content/</code> and published as static HTML at <code>docs/tuning-reference/&lt;page&gt;.html</code> (<code>docs-config.ts</code>&#39;s <code>docsUrl()</code> resolves an anchor to that path plus a <code>#&lt;anchor&gt;</code> fragment). <code>useDocs().open(anchor)</code> sets React state (<code>isOpen</code>, <code>target</code>); Radix/Base UI&#39;s <code>Sheet</code> owns the slide-in animation, focus trap, and outside-click/Escape dismissal. There is a single <code>DocsTarget</code> shape (<code>{ kind: &#39;site&#39;, path }</code>): no vendor HTML and no <code>&#39;vendor&#39;</code> target kind, so <code>DocsSheet</code> always drives the iframe the same way, reassigning <code>src</code> on any path or theme change.</p><p>Most anchors the app links to are the page of the same name; a handful are in-page fragments on another page instead (config-audit sub-findings and the metric glossary live on the <code>config</code>/<code>metrics</code> pages; the two bottleneck &quot;stage-*&quot; sub-anchors live on the page of the bottleneck that owns them): <code>docs-config.ts</code>&#39;s <code>pageForAnchor()</code> is the one place that resolves an anchor to its owning page. <code>npm run docs:build</code> (run automatically by <code>npm run build</code>) renders <code>packages/core/src/docs-content/</code> into <code>docs-site/.vitepress/dist</code>, and <code>vite.config.ts</code>&#39;s <code>copyDocsSite</code> plugin copies that output to <code>dist/docs</code>; Vite&#39;s relative asset base keeps the app and docs usable when <code>dist/</code> is deployed under a URL subpath. <code>tests/doc-anchor-coverage.test.js</code> intersects <code>packages/core/src/docs-content/chapters/nav-index.json</code> against every detector&#39;s <code>docAnchor</code> and warns (never fails) on dead links (detector points at an anchor the nav index doesn&#39;t have) or orphaned Detector Catalog anchors (no detector points at them); see <code>scripts/doc-anchor-coverage.js</code>. The ported landing page (<code>docs-site/tuning-reference/index.md</code>, the symptom-picker entry page) is hand-authored, committed markdown like the rest of the corpus, not a build-time copy.</p><p>A docs-site page has no channel back to this app: it&#39;s a plain static page with no <code>postMessage</code> listener. <code>DocsSheet.tsx</code> reassigns the iframe&#39;s <code>src</code> outright on any change to the resolved path or the theme, forcing a full reload. Since re-assigning the exact same <code>src</code> string wouldn&#39;t make the browser reload it, a <code>t=&lt;theme&gt;</code> marker is threaded into the query string ahead of the <code>#anchor</code> hash purely to change the string and force a real reload; the page never reads that param itself: it reads its light/dark preference once, from the <code>vitepress-theme-appearance</code> localStorage key <code>ThemeProvider</code> keeps current, the moment it boots.</p>',33)])])}const u=o(n,[["render",s]]);export{g as __pageData,u as default};
@@ -0,0 +1 @@
1
+ import{_ as o,o as t,c as a,a5 as d}from"./chunks/framework.DSg0KOwT.js";const g=JSON.parse('{"title":"Drill-down","description":"","frontmatter":{},"headers":[],"relativePath":"contributor-guide/architecture/drill-down.md","filePath":"contributor-guide/architecture/drill-down.md"}'),n={name:"contributor-guide/architecture/drill-down.md"};function s(c,e,i,r,l,h){return t(),a("div",null,[...e[0]||(e[0]=[d("",33)])])}const u=o(n,[["render",s]]);export{g as __pageData,u as default};
@@ -0,0 +1 @@
1
+ import{_ as t,o,c as d,a5 as a}from"./chunks/framework.DSg0KOwT.js";const p=JSON.parse('{"title":"Impact estimation","description":"","frontmatter":{},"headers":[],"relativePath":"contributor-guide/architecture/impact-estimation.md","filePath":"contributor-guide/architecture/impact-estimation.md"}'),c={name:"contributor-guide/architecture/impact-estimation.md"};function s(i,e,n,r,l,u){return o(),d("div",null,[...e[0]||(e[0]=[a('<h1 id="impact-estimation" tabindex="-1">Impact estimation <a class="header-anchor" href="#impact-estimation" aria-label="Permalink to &quot;Impact estimation&quot;">​</a></h1><p>Every finding covered by this section carries an optional <code>impactEstimate: {basis, wallClock, estimateMethod, rawWaste?}</code> (<code>src/types.ts</code>), attached by <code>src/impact-estimator.ts</code> as a post-pass after <code>DETECTORS</code> finishes (<code>src/analyzer.ts</code>). <code>estimateMethod</code> (&#39;measured&#39; | &#39;modeled&#39; | &#39;none&#39;) is a distinct axis from the per-finding <code>confidence</code> field (<a href="./board-widgets.html#confidence-metadata">Confidence metadata</a>): <code>confidence</code> says how much to trust the finding itself, <code>estimateMethod</code> says how its impact number was derived. <code>&#39;none&#39;</code> marks a purely informational finding with no waste model at all (<code>configAudit</code>, <code>stageFailed</code>, <code>failures</code>, <code>incompleteRun</code>, <code>slowHost</code>&#39;s byte-dimension multiDim shapes): it&#39;s distinct from <code>&#39;measured&#39;</code>/<code>&#39;modeled&#39;</code>, which both attach a real (if approximate) formula. <code>basis</code> is one of:</p><ul><li><code>&#39;serial&#39;</code>: the tied stage ran (effectively) alone; <code>wallClock</code> is a near-point estimate, <code>low === high</code>.</li><li><code>&#39;contended&#39;</code>: the tied stage shared wall-clock time with others; <code>wallClock</code> is an honest range, <code>high</code> optimistic (assumes the fix could still fully land), <code>low</code> the guaranteed floor.</li><li><code>&#39;resourceOnly&#39;</code>: no wall-clock claim is defensible (not stage-tied by nature, or the stage was excluded from the occupancy sweep), <code>wallClock: null</code>, but the formula&#39;s real signal survives in <code>rawWaste</code>.</li><li><code>&#39;informational&#39;</code>: no quantifiable magnitude at all, <code>wallClock: null</code>, no <code>rawWaste</code>.</li></ul><p><code>{basis: &#39;resourceOnly&#39;|&#39;informational&#39;, wallClock: null}</code> replaced the earlier design&#39;s <code>{low: 0, high: 0}</code>: that single value used to mean two incompatible things (&quot;provably no wall-clock cost&quot; and &quot;the model gave up&quot;), and 76-97% of stage-tied findings on real logs were the second case wearing the first case&#39;s clothing (2026-08-30 N1 redesign; see <code>docs/superpowers/specs/2026-08-30-critical-path-occupancy-redesign.md</code>).</p><p><code>rawWaste</code> is not exclusive to <code>resourceOnly</code>/<code>informational</code> findings: every <code>serial</code>/ <code>contended</code> finding carries it too (the only exceptions are <code>coldStart</code>, whose <code>wallClock</code> figure is already unclipped, and <code>estimateMethod: &#39;none&#39;</code> findings, which have no formula at all), holding the formula&#39;s pre-clip magnitude in the formula&#39;s own natural unit (ms, bytes or core-ms). <code>wallClock</code> is what the occupancy model says is recoverable, which clips against the stage&#39;s own physical floor; <code>rawWaste</code> is what the stage really wasted either way. The two answer different questions.</p><h2 id="occupancy-weighted-attribution" tabindex="-1">Occupancy-weighted attribution <a class="header-anchor" href="#occupancy-weighted-attribution" aria-label="Permalink to &quot;Occupancy-weighted attribution&quot;">​</a></h2><p><code>src/occupancy.ts</code> sweeps every stage&#39;s observed <code>[submittedAt, completedAt)</code> window and splits each instant&#39;s wall-clock among concurrently-active stages proportional to <code>coreWeight(S) = stage.executorRunTime / stageDurationMs</code> (an average-concurrency proxy, held constant across the stage&#39;s whole window: this codebase has no per-task timestamps outside the parser worker to do better). Summing a stage&#39;s share across its own window gives its <code>occupancy(S)</code>; <code>gate(S) = occupancy(S) / duration(S) ∈ [0, 1]</code> is the single number that replaces the old CPM model&#39;s <code>onCriticalPath</code>/<code>slackMs</code>/<code>isUniquelyCritical</code>. 1.0 means the stage ran completely alone; 0 means the stage had zero <code>executorRunTime</code> while overlapping other, positive-weight stages, so it got no share of the shared window. Stages with <code>duration(S) &lt;= 0</code> (Spark-skipped stages, or a malformed <code>submittedAt === completedAt</code>) are excluded from the sweep entirely.</p><p>This mechanism replaced a CPM (critical-path-method) graph over <code>parentIds</code> that produced near-zero on-critical-path membership on real logs (0.1-10.6% of stages, max graph depth 2 on 5 of 6 real logs measured): <code>parentIds</code> alone is too sparse a precedence signal for a meaningful longest-path computation. The same degeneracy fed <code>efficiency-model.ts</code>&#39;s <code>floorInfiniteMs</code> (&quot;floor with infinite executors&quot;); rather than leave a second, unreconciled critical-path number in the codebase for a future UI to display next to the occupancy-based figures above, <code>criticalPathMs()</code>/<code>CriticalPathStage</code> (embedded in <code>efficiency-model.ts</code>, never a standalone module) were removed outright (no occupancy-based replacement: occupancy apportions observed concurrent time, it doesn&#39;t compute a dependency-graph longest path, so there&#39;s no drop-in equivalent). <code>efficiency-model.ts</code> now reports only <code>floorZeroSkewMs</code> (total task time / total cores) as its theoretical floor.</p><p><code>ceiling(S) = max(stage.taskDurationMax, stage.executorRunTime / totalCores)</code> is a physical floor on a stage&#39;s own duration: bounded below by its single longest task (unsplittable no matter how much parallelism exists) or by its core-work spread across every core in the cluster, whichever is larger. Every waste formula&#39;s raw claim is clipped against it before gate-weighting: <code>wasteMs_clipped(S) = min(wasteMs_claimed, max(0, duration(S) - ceiling(S)))</code>, so a finding can never claim to save more than the portion of the stage&#39;s observed duration that sits above its own unbeatable floor. This is what fixes historical overclaim bugs (a <code>tinyTask</code> finding claiming 407.5s on a 13.1s stage capped to 8.9s; a <code>shuffle</code> finding claiming 1939.9s on a 991.3s/1688.3s stage capped to 610.3s/250.1s).</p><p><code>analyzer.ts</code> feeds this <code>totalCores</code> from <code>src/core-count.ts</code>&#39;s <code>computePeakConcurrentCores(app, executorsAdded, executorsRemoved)</code>, not the shared <code>computeTotalCores</code> helper. <code>computeTotalCores</code> sums every <code>ExecutorAdded</code> event&#39;s cores regardless of overlap, so under dynamic allocation or executor replacement it can far exceed the cores ever actually concurrent, which understates <code>ceiling(S)</code> and lets churn inflate a finding&#39;s claimed wall-clock. <code>computePeakConcurrentCores</code> instead sweeps add/remove events by timestamp and tracks the running total&#39;s peak, so a churned-through executor&#39;s cores are never double-counted against its replacement&#39;s. Same-timestamp events tie-break by delta ascending, so a removal applies before a same-instant replacement&#39;s addition (otherwise a same-instant swap would momentarily double-count both as concurrent). If every <code>executorsAdded</code> entry lacks <code>totalCores</code> the cores sweep peaks at zero and tells us nothing; the function then falls back to sweeping peak <em>executor count</em> instead (still concurrency-aware, just cores-blind) and multiplies by the configured per-executor core count, rather than falling back to <code>executorsAdded.length × cores</code>, which would reintroduce the exact cumulative-overcount-under-churn bug this function exists to avoid.</p><p>Per-finding estimate, using <code>wasteMs_clipped(S)</code>:</p><ul><li><code>gate(S) &gt;= 0.999</code>: <code>basis: &#39;serial&#39;</code>, <code>low = high = wasteMs_clipped(S)</code>.</li><li><code>gate(S) &lt; 0.999</code>: <code>basis: &#39;contended&#39;</code>, <code>high = wasteMs_clipped(S)</code>, <code>low = wasteMs_clipped(S) * gate(S)</code>.</li></ul><p>A finding spanning multiple stages (<code>stageIds</code>, plural) sums each stage&#39;s own estimate and caps the joint total at the union of just that finding&#39;s own stage windows (via <code>mergeIntervals</code>, <code>src/wall-clock.ts</code>): <code>high = min(Σ high_i, unionMs(stageIds))</code>, <code>low = min(Σ low_i, unionMs(stageIds))</code>. This is what prevents overclaiming when two or more of a finding&#39;s stages overlap in wall-clock time: a plain sum-and-cap, no CPM re-simulation. The union cap can force <code>low === high</code> numerically even when the constituent stages were individually contended (e.g. two fully-overlapping stages each at <code>gate</code> 0.5), so <code>basis</code> isn&#39;t derived from that numeric equality: a multi-stage finding gets <code>basis: &#39;serial&#39;</code> only when every one of its per-stage estimates was itself <code>&#39;serial&#39;</code>; otherwise <code>&#39;contended&#39;</code>.</p><p>Regression guard: a finding whose stage (or, for a multi-stage finding, every one of its stages) ran alone (<code>gate &gt;= 0.95</code>, deliberately looser than the <code>0.999</code> &quot;serial&quot; cutoff above: this guard exists to catch egregious false zeros, not to gate which <code>basis</code> a finding gets) with a real underlying magnitude (<code>rawWaste.value &gt; 0</code>) must never report <code>wallClock.high === 0</code> or <code>basis: &#39;informational&#39;/&#39;resourceOnly&#39;</code>. Covered by <code>tests/impact-estimator-real-log.test.js</code> against a real fixture; relies on <code>rawWaste</code> being attached to every serial/contended-capable formula (see above), so it&#39;s blind only to <code>coldStart</code> and the purely informational (<code>estimateMethod: &#39;none&#39;</code>) finding types.</p><p>Real-log spot-check (2026-08-30, <code>grupo-semanal-beauty-application_1785266278671_91660.zstd</code>): median <code>gate</code> across stages was <code>≈0.34</code> (0.3428060791718594 exactly); <code>collectRun</code> plus the occupancy sweep together took <code>≈6,589</code>ms on the largest fixture measured (<code>run-compare-calimax-candidate-application_1784568768686_119096.zstd</code>, <code>1169</code> stages): parse-dominated, the sweep alone was not isolated by this measurement, but not a magnitude that suggests a regression either. <code>54</code> previously-<code>{0,0}</code> stage-tied findings on stages that ran effectively alone now report a real <code>wallClock</code> range instead.</p><h2 id="cross-finding-rollup-computestageunionms" tabindex="-1">Cross-finding rollup: <code>computeStageUnionMs</code> <a class="header-anchor" href="#cross-finding-rollup-computestageunionms" aria-label="Permalink to &quot;Cross-finding rollup: `computeStageUnionMs`&quot;">​</a></h2><p>The Findings tab&#39;s recommendation rollup (<code>FixTheseFirst.tsx</code>, built from <code>buildRecommendationRollup</code>, <code>src/recommendation-rollup.ts</code>) groups the filtered catalog by detector <code>type</code>, then needs its own cap for a group of several findings of that type, not just one finding&#39;s own <code>stageIds</code>. Summing each finding&#39;s already-clipped <code>wallClock.high</code> naively double-counts any stage two of those findings both touch. <code>computeStageUnionMs(stageIds, stages)</code> covers this: collect every stage touched by any finding in the group, merge their <code>[submittedAt, completedAt)</code> intervals, and sum the merged intervals&#39; durations, so the group&#39;s wall-clock union holds regardless of how many findings&#39; <code>stageIds</code> overlap. Stages missing either bound are skipped rather than defaulted to <code>0</code> (the same filter <code>computeWallClock</code> applies), so a truncated log (the case <code>incompleteRun</code> flags) can&#39;t contribute a negative interval and a negative recoverable-time figure.</p><p>It reuses the same <code>mergeIntervals</code> primitive (<code>src/wall-clock.ts</code>) that backs <code>src/occupancy.ts</code>&#39;s per-finding <code>estimateMultiStage</code>/its internal <code>unionMs</code> sum, but is not an extension of that function: <code>estimateMultiStage</code> caps one finding&#39;s own multi-stage claim during the impact-estimation pass, before a <code>Finding</code> object even exists; <code>computeStageUnionMs</code> runs later, in the view layer, capping a naive sum <em>across</em> several already-estimated findings that happen to share a detector <code>type</code>. <code>buildTimeGroup</code> (same module) takes the smaller of the naive per-finding sum and this union figure as the group&#39;s <code>recoverableMsHigh</code>, falling back to the naive sum untouched when the group&#39;s findings carry no stage IDs at all (nothing to union against).</p><p>Within one <code>type</code> group, <code>buildRecommendationRollup</code> splits findings into up to three tiers, always rendered in this fixed order. <code>time</code> covers findings with a real <code>impactEstimate.wallClock</code> (<code>buildTimeGroup</code>, the union-capped figure above). <code>resource</code> covers findings with no <code>wallClock</code> but a <code>rawWaste</code> figure (<code>buildResourceGroup</code>), grouped again by <code>rawWaste.unit</code> so a <code>bytes</code> total never gets summed against a <code>coreHours</code> total under one type. <code>count</code> covers findings with neither (<code>buildCountGroup</code>), a plain per-impact-band tally with no magnitude claim at all. Each <code>RollupGroup</code> also carries its own <code>findings: Finding[]</code> (the exact members that fed the aggregate), which <code>FixTheseFirst.tsx</code> reads directly to pick a group&#39;s highest-impact member and to render its expanded, paginated list.</p><p>A type only contributes a tier when it has at least one finding of that kind; most types produce exactly one tier, but a type whose formula varies by <code>variant</code>/<code>rule</code> (e.g. <code>memoryUtilization</code>, see the coverage table below) can produce more than one.</p><p><code>cachingOpportunity</code> and <code>cacheUtilization</code> are both <code>cost-only</code>: <code>basis: &#39;resourceOnly&#39;</code>, <code>wallClock: null</code>, but <code>rawWaste.unit</code> is <code>&#39;ms&#39;</code>, the same unit a real <code>wallClock</code> figure would use, because their formula&#39;s natural output happens to be time (a re-read cost), not because either finding makes a wall-clock claim. Left unlabeled, a <code>resource</code>-tier &quot;ms&quot; total sitting next to a <code>time</code>-tier &quot;recoverable time&quot; total would read as directly comparable when it isn&#39;t: the resource figure was never gate-clipped against any stage&#39;s occupancy, so it can exceed what the stage actually spent. <code>FixTheseFirst.tsx</code> calls this out via its trailing-stat copy: a <code>resource</code>- kind group (any unit, including <code>ms</code>) always reads &quot;resource-cost projection&quot;, never &quot;recoverable&quot;, so the two ms-shaped numbers are never mistaken for the same kind of claim.</p><h2 id="per-formula-spot-checks" tabindex="-1">Per-formula spot-checks <a class="header-anchor" href="#per-formula-spot-checks" aria-label="Permalink to &quot;Per-formula spot-checks&quot;">​</a></h2><table tabindex="0"><thead><tr><th>Detector</th><th>Formula basis</th><th>Spot-check</th></tr></thead><tbody><tr><td>gc</td><td><code>jvmGCTime / (executorRunTime / stageDurationMs)</code></td><td><code>grupo-semanal-beauty-application_1785266278671_91660.zstd</code>, stage 507: <code>jvmGCTime</code>=1080ms, <code>executorRunTime</code>=27509ms, <code>stageDurationMs</code>=56279ms → <code>wasteMs</code> = 1080 / (27509/56279) ≈ 2209.5ms. That&#39;s ≈3.9% of the stage&#39;s 56.3s wall-clock duration, matching the finding&#39;s own reported <code>gcPct</code> (3.9%) exactly, as the formula guarantees by construction. Under the occupancy model this stage&#39;s <code>gate</code> is <code>0.041</code> (0.04145044590332269 exactly): <code>basis: &#39;contended&#39;</code>, <code>wallClock: {low: 91.6, high: 2209.5}</code> (91.5850382272182 / 2209.506706895925 exactly, per the Step 1 script&#39;s per-stage output).</td></tr><tr><td>shuffle</td><td><code>shuffleReadBytes / SHUFFLE_THROUGHPUT_BPS</code> (fallback tier; real-metrics parser not yet built)</td><td><code>ventas-mensual-multi-big-application_1785266278671_91510.zstd</code>, stage 99 (<code>SHFL</code> finding): <code>shuffleReadBytes</code>=204,172,518,504 → <code>wasteMs</code> = 204172518504 / 125,000,000 × 1000 ≈ 1,633,380ms, matching <code>rawWaste.value</code> exactly. Eyeball against the timeline: the stage&#39;s actual wall-clock duration is only 763,776ms, i.e. this run moved shuffle data at ≈267MB/s, roughly 2x the assumed 125MB/s (1Gbps) constant: expected for the fallback tier&#39;s deliberately conservative assumption, but worth knowing the modeled figure runs high on fast-network clusters. Under the occupancy model this stage&#39;s <code>gate</code> is <code>1</code>, <code>ceiling</code> is <code>≈434,568.1</code>ms (434568.1041666667 exactly): <code>wallClock</code> capped to <code>≈329,207.9</code>ms (329207.8958333333 exactly), since the raw 1,633,380ms claim vastly exceeds the stage&#39;s own 763,776ms duration.</td></tr><tr><td>spill</td><td><code>diskBytesSpilled / SPILL_IO_THROUGHPUT_BPS</code></td><td>Same run and stage (99): <code>diskBytesSpilled</code>=145,978,433,675 (note: the <code>SPILL</code> finding&#39;s own <code>value</code>/<code>metric</code> report <code>memoryBytesSpilled</code>=913,686,966,448, ~6x larger; the formula correctly uses the smaller disk figure, not that one) → <code>wasteMs</code> = 145978433675 / 200,000,000 × 1000 ≈ 729,892ms, matching <code>rawWaste.value</code> exactly. Eyeball: that&#39;s ≈191MB/s of implied disk throughput against the stage&#39;s 763,776ms actual duration, close to the assumed 200MB/s constant. Same ceiling-clip caveat as the shuffle row above applies here too, on the same stage.</td></tr></tbody></table><h2 id="overlap-caveat-skew-straggler" tabindex="-1">Overlap caveat: skew / straggler <a class="header-anchor" href="#overlap-caveat-skew-straggler" aria-label="Permalink to &quot;Overlap caveat: skew / straggler&quot;">​</a></h2><p><code>skew</code> (small-stage max-P50 fallback branch) and <code>straggler</code> can both fire on the same stage from the same single dominant outlier task, and each is clipped independently. This phase does not dedupe or suppress either: each keeps its own independently-computed <code>wallClock</code>. Do not sum <code>wallClock.high</code> across multiple findings on the same stage: if both fire together, they describe the same underlying waste, not two separate wastes. This overlap caveat is orthogonal to (and compounds with) the ceiling clip above: a stage with one dominant outlier task trips both detectors <em>and</em> has a small <code>ceiling</code>-derived recoverable room, since <code>ceiling</code> is itself <code>&gt;= taskDurationMax</code>, the very quantity these two detectors are reacting to.</p><p><code>analyzer.ts</code>&#39;s <code>flagSkewStragglerOverlap</code> (run after <code>deriveImpactBand</code>, once per <code>analyze()</code> call) surfaces this caveat to the reader instead of leaving it as an internal-only comment: whenever <code>skew</code>&#39;s <code>max/median</code> branch and <code>straggler</code> both fire on the same <code>stageId</code>, it appends a &quot;this overlaps with the X finding on this stage&quot; sentence to both findings&#39; <code>validationRequired</code> text (rather than suppressing either, so neither finding&#39;s own diagnostic value is lost). <code>skew</code>&#39;s <code>P95/median</code> branch samples a different task from <code>straggler</code>&#39;s own <code>taskDurationMax - taskDurationP50</code> delta, so it&#39;s excluded from the flag. The note rides the same confidence-caveat UI (<code>RowStatusCluster</code>) a reader already sees before trusting either finding&#39;s magnitude, since both detectors also carry a <code>confidence</code> field that scales <code>low</code>/<code>medium</code>/<code>high</code> off how far the finding sits past its own runtime-floor threshold (still unvalidated; see the confidence-disclosure note in detector-contract.md).</p><p><code>stageShape</code>&#39;s <code>taskStageSkew</code> rule no longer participates in this caveat: it reports a <code>resourceOnly</code> idle-core-ms figure (see the coverage table below) instead of a wall-clock claim, so there&#39;s nothing left to double-count against <code>skew</code>/<code>straggler</code>. Its trigger condition (<code>taskDurationMax / stageDurationMs &gt; skewWarn</code>) mathematically forces the occupancy-clipped wall-clock estimate to exactly zero on every firing (see <code>src/detectors.ts</code>&#39;s <code>taskStageSkew</code> comment), which is why it was moved off the wall-clock path entirely rather than reconciled against the same ceiling clip as its two siblings above.</p><h2 id="per-finding-type-coverage" tabindex="-1">Per-finding-type coverage <a class="header-anchor" href="#per-finding-type-coverage" aria-label="Permalink to &quot;Per-finding-type coverage&quot;">​</a></h2><p>One row per distinct <code>type</code> string <code>src/detectors.ts</code> actually emits (cross-checked against <code>computeEstimateForFinding</code>&#39;s <code>case</code> labels in <code>src/impact-estimator.ts</code>, not assumed from the prose here): every row below has a case, so the table itself is the coverage count, not a number restated here. <code>broadcastSizing</code> is a <code>DETECTORS</code> entry label only, and the plan-walk it drives emits <code>overBroadcast</code>/<code>underBroadcast</code> findings instead, so those two are the rows that appear, not <code>broadcastSizing</code> itself. Tag meanings: <code>measured</code> and <code>modeled</code> both produce a real, gate-clipped, non-<code>{0,0}</code><code>wallClock</code> (the difference is whether the formula&#39;s inputs are recorded per-stage fields or an assumed constant like a throughput figure); <code>cost-only</code> always reports <code>basis: &#39;resourceOnly&#39;</code>, <code>wallClock: null</code> but carries its real signal in <code>rawWaste</code>; <code>informational-only</code> reports <code>basis: &#39;informational&#39;</code>, <code>wallClock: null</code> with no <code>rawWaste</code> at all, since there&#39;s nothing quantifiable. A type with more than one tag fires a different formula per <code>variant</code>/<code>rule</code> on the same finding type; the basis column says which.</p><table tabindex="0"><thead><tr><th>Finding type</th><th>Scope</th><th>Tag</th><th>Basis</th></tr></thead><tbody><tr><td><code>retryWaste</code></td><td>stage</td><td>measured</td><td><code>retryWasteMs</code>, gate-clipped; pre-clip figure kept as <code>rawWaste</code> in <code>ms</code></td></tr><tr><td><code>speculationWaste</code></td><td>stage</td><td>measured</td><td><code>speculationWasteMs</code>, gate-clipped; pre-clip figure kept as <code>rawWaste</code> in <code>ms</code></td></tr><tr><td><code>coldStart</code></td><td>app</td><td>measured</td><td><code>gapSeconds × 1000</code>, unclipped, <code>basis: &#39;serial&#39;</code> unconditionally (a pre-first-task gap can&#39;t overlap any stage)</td></tr><tr><td><code>gc</code></td><td>stage</td><td>modeled</td><td><code>jvmGCTime / (executorRunTime / stageDurationMs)</code>, gate-clipped: the concurrency division is an approximation, not a reconstruction, hence <code>modeled</code>; <code>rawWaste</code> in <code>coreMs</code> is the raw <code>jvmGCTime</code> sum before that conversion</td></tr><tr><td><code>skew</code></td><td>stage</td><td>measured</td><td><code>taskDurationP95</code> or <code>Max</code> minus <code>P50</code> (per <code>metric</code>), gate-clipped; pre-clip figure kept as <code>rawWaste</code> in <code>ms</code></td></tr><tr><td><code>straggler</code></td><td>stage</td><td>measured</td><td><code>taskDurationMax − taskDurationP50</code>, gate-clipped</td></tr><tr><td><code>stageShape</code></td><td>stage</td><td>cost-only</td><td>all three rules are <code>estimateMethod: &#39;measured&#39;</code>, real per-stage fields, no assumed constant: <code>&#39;lowParallelism&#39;</code> → <code>rawWaste</code> in <code>coreMs</code> (idle cores × stage duration); <code>&#39;dataExplosion&#39;</code> → <code>rawWaste</code> in <code>bytes</code> (<code>outputBytes − inputBytes</code>); <code>&#39;taskStageSkew&#39;</code> → <code>rawWaste</code> in <code>coreMs</code> (<code>max(0, min(totalCores, taskCount) − 1) × (taskDurationMax − taskDurationP50)</code>, the cores idle during the straggler&#39;s tail at achieved concurrency)</td></tr><tr><td><code>slowHost</code></td><td>stage</td><td>measured / informational-only</td><td>duration-based variants (<code>hostMeanRatio</code>, <code>durationShare</code>, <code>multiDim</code>+<code>taskTime</code>): <code>value − taskDurationP50</code>, gate-clipped; byte-based <code>multiDim</code> dimensions: no formula yet</td></tr><tr><td><code>duplicatePlanSubtree</code></td><td>sql</td><td>measured</td><td>each contributing stage&#39;s real wall-clock duration × the redundant fraction <code>(occurrences − 1) / occurrences</code>, summed and capped at the finding&#39;s own <code>stageIds</code> union. <code>stageIds</code> is narrowed to the stages that actually ran the duplicated subtree&#39;s matched node instances (accumulator-ID evidence resolved onto each <code>PlanNode</code> at parse time, see <a href="./detector-contract.html#stage-id-attribution-for-plan-advisor-findings">Stage-ID attribution for Plan Advisor findings</a>), falling back to the whole execution&#39;s stages only when no matched instance has any accumulator coverage</td></tr><tr><td><code>shuffle</code></td><td>stage</td><td>modeled</td><td><code>shuffleReadBytes / SHUFFLE_THROUGHPUT_BPS</code> (assumed ~125MB/s), gate-clipped; <code>rawWaste</code> in <code>bytes</code> is the measured <code>shuffleReadBytes</code> behind it</td></tr><tr><td><code>spill</code></td><td>stage</td><td>modeled</td><td><code>diskBytesSpilled / SPILL_IO_THROUGHPUT_BPS</code> (assumed ~200MB/s), gate-clipped; <code>rawWaste</code> in <code>bytes</code> is <code>diskBytesSpilled</code>, which is the number the formula uses and not the <code>memoryBytesSpilled</code> the finding&#39;s own <code>metric</code> displays</td></tr><tr><td><code>stageSlowness</code></td><td>stage</td><td>modeled</td><td>stage duration minus the <code>stageSlowness</code> detector&#39;s own <code>infoMin</code> threshold, gate-clipped</td></tr><tr><td><code>partitionSizing</code></td><td>stage</td><td>modeled</td><td><code>maxPartitionTooBig</code>/<code>shufflePartitionSkew</code>: shuffle-throughput formulas, gate-clipped. <code>lowShuffleParallelism</code>: stage duration scaled down by the shortfall between actual and ideal-partition-count task counts (<code>stageDurationMs × (1 − taskCount / targetTaskCount)</code>), i.e. the serialized work more partitions would let run concurrently, not the scheduling cost of the tasks you&#39;d add to fix it</td></tr><tr><td><code>tinyTask</code></td><td>stage</td><td>modeled</td><td>excess task count over 10% of the stage&#39;s actual count, × assumed per-task scheduling overhead, gate-clipped; pre-clip figure kept as <code>rawWaste</code> in <code>ms</code></td></tr><tr><td><code>smallFiles</code></td><td>sql</td><td>modeled / cost-only</td><td><code>excessFileCount × FILE_OPEN_OVERHEAD_MS</code>, summed and capped over <code>stageIds</code>&#39;s union; with no <code>stageIds</code> to map to, cost-only with that same figure as <code>rawWaste</code> in <code>ms</code>. <code>stageIds</code> is narrowed the same way (see <a href="./detector-contract.html#stage-id-attribution-for-plan-advisor-findings">Stage-ID attribution for Plan Advisor findings</a>); falls back to the whole execution&#39;s stages when the flagged node(s) have no accumulator coverage.</td></tr><tr><td><code>overBroadcast</code></td><td>sql</td><td>modeled / cost-only</td><td><code>broadcastBytes / BROADCAST_BANDWIDTH_BPS</code>, summed and capped over <code>stageIds</code>&#39;s union; cost-only with <code>rawWaste</code> in <code>ms</code> when not stage-mappable. <code>stageIds</code> is narrowed the same way; falls back to the whole execution&#39;s stages when the flagged node(s) have no accumulator coverage.</td></tr><tr><td><code>underBroadcast</code></td><td>sql</td><td>modeled / cost-only</td><td><code>smallerSideBytes / BROADCAST_BANDWIDTH_BPS</code>, summed and capped over <code>stageIds</code>&#39;s union; cost-only with <code>rawWaste</code> in <code>ms</code> when not stage-mappable. <code>stageIds</code> is narrowed the same way; falls back to the whole execution&#39;s stages when the flagged node(s) have no accumulator coverage.</td></tr><tr><td><code>memoryUtilization</code></td><td>app</td><td>cost-only / informational-only</td><td>Three of the four variants report <code>rawWaste</code> in <code>mbSeconds</code>: <code>variant: &#39;wasteModel&#39;</code> passes through its own <code>wastedMBSeconds</code>; <code>&#39;idleCores&#39;</code> uses <code>idleRateFraction × allocatedMB × peakExecutors × appDurationSeconds</code>; <code>&#39;memoryBand&#39;</code> with <code>rule: &#39;heapOverProvisioned&#39;</code> uses <code>(allocatedBytes − heap) in MB × appDurationSeconds</code>. <code>&#39;memoryBand&#39;</code> with <code>rule: &#39;heapNearCapacity&#39;</code> is an OOM-risk signal rather than a waste, and the <code>dataUnavailable</code> shape has no inputs at all: both informational-only</td></tr><tr><td><code>utilization</code></td><td>app</td><td>cost-only</td><td><code>rawWaste</code> in <code>coreHours</code>: <code>(1 − utilizationFraction) × appDurationMs × totalCores / 3.6e6</code></td></tr><tr><td><code>coreLocality</code></td><td>app</td><td>cost-only</td><td><code>rawWaste</code> in <code>coreMs</code>: <code>nonLocalTaskCount × NETWORK_FETCH_PENALTY_MS</code></td></tr><tr><td><code>autoscalingChurn</code></td><td>app</td><td>cost-only</td><td><code>rawWaste</code> in <code>coreHours</code>: <code>shortLivedExecutorCount × EXECUTOR_STARTUP_OVERHEAD_MS / 3.6e6</code></td></tr><tr><td><code>configAudit</code></td><td>config</td><td>informational-only</td><td>a config-drift check standing alone; no waste formula</td></tr><tr><td><code>jobFailureRate</code></td><td>app</td><td>cost-only</td><td><code>rawWaste</code> in <code>coreHours</code>: <code>failedJobCount × avgJobDurationMs / 3.6e6</code></td></tr><tr><td><code>cachingOpportunity</code></td><td>app</td><td>cost-only</td><td><code>rawWaste</code> in <code>ms</code>: <code>totalReadBytes / RE_READ_THROUGHPUT_BPS</code></td></tr><tr><td><code>cacheUtilization</code></td><td>app</td><td>cost-only</td><td><code>rawWaste</code> in <code>ms</code>: uncached-or-spilled bytes <code>/ RE_READ_THROUGHPUT_BPS</code>, where the never-cached partitions&#39; bytes are extrapolated from the cached partitions&#39; own average size (<code>memorySize + diskSize</code>, over <code>numCachedPartitions</code>), plus <code>diskSize</code> again for the already-cached-but-on-disk partitions&#39; own re-read cost</td></tr><tr><td><code>stageFailed</code></td><td>stage</td><td>informational-only</td><td>no waste formula</td></tr><tr><td><code>failures</code></td><td>stage</td><td>informational-only</td><td>no waste formula</td></tr><tr><td><code>incompleteRun</code></td><td>app</td><td>informational-only</td><td>no waste formula</td></tr></tbody></table>',30)])])}const g=t(c,[["render",s]]);export{p as __pageData,g as default};
@@ -0,0 +1 @@
1
+ import{_ as t,o,c as d,a5 as a}from"./chunks/framework.DSg0KOwT.js";const p=JSON.parse('{"title":"Impact estimation","description":"","frontmatter":{},"headers":[],"relativePath":"contributor-guide/architecture/impact-estimation.md","filePath":"contributor-guide/architecture/impact-estimation.md"}'),c={name:"contributor-guide/architecture/impact-estimation.md"};function s(i,e,n,r,l,u){return o(),d("div",null,[...e[0]||(e[0]=[a("",30)])])}const g=t(c,[["render",s]]);export{p as __pageData,g as default};
@@ -0,0 +1 @@
1
+ import{_ as t,o as r,c as a,a5 as i}from"./chunks/framework.DSg0KOwT.js";const m=JSON.parse('{"title":"Architecture","description":"","frontmatter":{},"headers":[],"relativePath":"contributor-guide/architecture/index.md","filePath":"contributor-guide/architecture/index.md"}'),c={name:"contributor-guide/architecture/index.md"};function o(n,e,s,h,d,l){return r(),a("div",null,[...e[0]||(e[0]=[i('<h1 id="architecture" tabindex="-1">Architecture <a class="header-anchor" href="#architecture" aria-label="Permalink to &quot;Architecture&quot;">​</a></h1><p>These pages cover one concern each: the worker protocol, the state model, the detector contract, and the fixed widget rendering order.</p><p>Where to start:</p><ul><li><a href="./overview.html#two-actors">Two actors</a>: the parser worker and the view, and why the split exists.</li><li><a href="./detector-contract.html#detector-contract">Detector contract</a>: the interface every bottleneck detector implements. Read this before adding a finding.</li><li><a href="./impact-estimation.html#impact-estimation">Impact estimation</a>: how findings get a wall-clock/resource waste estimate attached.</li><li><a href="./../testing.html#testing-layout">Testing layout</a>: where tests live and what each layer covers.</li><li><a href="./../contributing.html">Contributing</a>: how architectural decisions get recorded as ADRs.</li></ul>',4)])])}const p=t(c,[["render",o]]);export{m as __pageData,p as default};
@@ -0,0 +1 @@
1
+ import{_ as t,o as r,c as a,a5 as i}from"./chunks/framework.DSg0KOwT.js";const m=JSON.parse('{"title":"Architecture","description":"","frontmatter":{},"headers":[],"relativePath":"contributor-guide/architecture/index.md","filePath":"contributor-guide/architecture/index.md"}'),c={name:"contributor-guide/architecture/index.md"};function o(n,e,s,h,d,l){return r(),a("div",null,[...e[0]||(e[0]=[i("",4)])])}const p=t(c,[["render",o]]);export{m as __pageData,p as default};
@@ -0,0 +1 @@
1
+ import{_ as o,o as r,c as t,a5 as c}from"./chunks/framework.DSg0KOwT.js";const m=JSON.parse('{"title":"Architecture overview","description":"","frontmatter":{},"headers":[],"relativePath":"contributor-guide/architecture/overview.md","filePath":"contributor-guide/architecture/overview.md"}'),a={name:"contributor-guide/architecture/overview.md"};function d(s,e,i,n,l,p){return r(),t("div",null,[...e[0]||(e[0]=[c('<h1 id="architecture-overview" tabindex="-1">Architecture overview <a class="header-anchor" href="#architecture-overview" aria-label="Permalink to &quot;Architecture overview&quot;">​</a></h1><h2 id="core-invariant" tabindex="-1">Core invariant <a class="header-anchor" href="#core-invariant" aria-label="Permalink to &quot;Core invariant&quot;">​</a></h2><p>Task data must never live on the main thread. A Spark job can emit millions of <code>SparkListenerTaskEnd</code> events (240 MB+ of metrics); parsing them on the UI thread freezes Chrome. The design below enforces that in both deploy modes: the static zero-backend app, and the optional local-server mode in <code>server/</code>, which also proxies Spark History Server fetches to sidestep CORS.</p><h2 id="two-actors" tabindex="-1">Two actors <a class="header-anchor" href="#two-actors" aria-label="Permalink to &quot;Two actors&quot;">​</a></h2><h3 id="main-thread" tabindex="-1">Main thread <a class="header-anchor" href="#main-thread" aria-label="Permalink to &quot;Main thread&quot;">​</a></h3><p>Owns the DOM (React) and a summary <code>AppModel</code> (jobs, stages, sql, executors, no raw tasks), held in a Zustand store (<code>src/store/store.ts</code>).</p><p><code>src/store/useIngest.ts</code> is the composition root. It owns the <code>IngestClient</code> lifecycle and wires <code>model-assembler.ts</code>&#39;s <code>createModelCallbacks</code> (worker-message → <code>AppModel</code>, unchanged from before the view migration) into store updates: <code>onProgress</code> → <code>parse</code>; <code>onDone</code> → derives <code>evidenceAvailability</code> from the final normalized model plus <code>done.skippedLines</code>, runs <code>analyzer.ts</code>, sets <code>catalog</code>, and prefetches flagged-stage task data.</p><p><code>src/App.tsx</code> routes on store <code>status</code>: idle/error → <code>DropZone</code>, parsing → a live progress readout, ready → <code>Dashboard</code>. <code>src/view/Dashboard.tsx</code> and <code>src/view/detector-registry.tsx</code> replace <code>dashboard-renderer.js</code>&#39;s widget board (see <a href="./widget-rendering.html">Widget rendering</a>).</p><h3 id="worker-web-worker-core-js" tabindex="-1">Worker (Web Worker, core JS) <a class="header-anchor" href="#worker-web-worker-core-js" aria-label="Permalink to &quot;Worker (Web Worker, core JS)&quot;">​</a></h3><p>Owns the file and <code>taskStore: Map&lt;stageId, Float64Array&gt;</code>, keyed by 8-field stride <code>[duration, gcTime, memSpilled, diskSpilled, shuffleRead, shuffleWrite, launchTime, finishTime]</code>. Four modules:</p><ul><li><code>src/stage-quantiles.ts</code>, a pure leaf, exporting <code>finalizeStage</code>, <code>computeFieldQuantiles</code>, <code>computeDurationQuantiles</code>, <code>classifySpill</code>, <code>FIELDS</code>, and <code>TASK_FIELD_NAMES</code> for the task-field packed-array layout.</li><li><code>src/event-handlers.ts</code>, holding <code>processEvent</code> and the per-event handler functions, <code>accumulateTask</code>, <code>createState</code>, <code>dispatchLine</code>, <code>buildChunkDecoder</code>, <code>emitParseCompletion</code>, <code>collectStageExecutorMetrics</code>.</li><li><code>src/shs-fetch.ts</code>, the SHS zip-fetch/decompress path: <code>runParseFromUrl</code>, <code>naturalCompare</code>, <code>reassembleRollingEntries</code>, <code>sniffCodec</code>.</li><li><code>src/parser-worker.ts</code>, a thin entrypoint (<code>streamFile</code>, <code>runParse</code>, <code>runParseFiles</code>, the <code>isWorker</code>/<code>self.onmessage</code> bus) that barrel-re-exports the other three. It is the only piece needing the File/Blob streaming API.</li></ul><h2 id="streaming" tabindex="-1">Streaming <a class="header-anchor" href="#streaming" aria-label="Permalink to &quot;Streaming&quot;">​</a></h2><p>Worker reads the <code>File</code> in 4 MB chunks via <code>file.slice(...).arrayBuffer()</code>, <code>TextDecoder({ stream: true })</code>, splits on <code>\\n</code>, JSON-parses, dispatches per event type. Quantiles (P50/P95/max) and spill classification (<code>skew | volume | unclassified</code>) are computed at <code>SparkListenerStageCompleted</code> time, before posting <code>StageAggregate</code> to main.</p><p>When <code>spark.eventLog.logStageExecutorMetrics=true</code> (default <code>false</code>), <code>SparkListenerStageExecutorMetrics</code> events populate <code>stage.executorMetrics: Map&lt;execId, {...}&gt;</code> with the 23 raw peak-memory/GC fields verbatim (camelCased), consumed by the <code>memoryUtilization</code> detector&#39;s per-executor memory bands (see <a href="./board-widgets.html">Memory Utilization</a>).</p><p>These events can arrive <em>after</em> <code>SparkListenerStageCompleted</code> for the same stage, so the per-stage <code>StageAggregate</code> message posted at completion time never carries them. Instead, <code>collectStageExecutorMetrics(state)</code> walks every stage once more just before <code>done</code>, and the worker re-posts a single <code>stageExecutorMetrics</code> message (<code>Map&lt;stageId, Map&lt;execId, metrics&gt;&gt;</code>, empty when the log has no such data). Main-thread consumers get every stage&#39;s metrics without a per-stage race.</p><p>Compressed logs are inflated inline. <code>sniffCodec</code> reads the leading magic bytes: gzip (<code>1f 8b</code>), Zstandard (<code>28 b5 2f fd</code>, Spark&#39;s <code>spark.io.compression.codec=zstd</code>), Spark&#39;s custom <code>LZ4Block</code> framing, or Spark&#39;s Snappy framing (<code>org.xerial.snappy</code>&#39;s <code>\\x82SNAPPY\\0</code> header, <code>spark.io.compression.codec=snappy</code>), falling back to the filename suffix. Both the dropped-file path and the SHS-fetch path stream block-by-block (fflate <code>Gunzip</code> / fzstd <code>Decompress</code> / the LZ4Block decoder / the Snappy block decoder) to keep one decompressed chunk live at a time.</p><p>The SHS-fetch path does buffer the downloaded zip whole for <code>unzipSync</code>. But one-shotting the decompression of a single entry (fzstd&#39;s <code>decompress()</code>, fflate&#39;s <code>gunzipSync()</code>) would allocate that entry&#39;s full declared content size as one ArrayBuffer, which fails outright for a multi-GB event log. So <code>decodeEntry(name, raw, onChunk)</code> picks the codec&#39;s streaming decoder per entry and invokes <code>onChunk</code> per decompressed piece. Decompressors are vendored under <code>src/vendor/</code> (<code>fflate.js</code>, <code>fzstd.js</code>) plus <code>src/lz4-block.ts</code> and <code>src/snappy-block.ts</code>.</p><p>Rolling <code>eventlog_v2_*</code> directories (Spark&#39;s multi-file event-log format, <code>events_&lt;index&gt;_&lt;appId&gt;(.codec)</code> files plus a zero-byte <code>appstatus_*</code> completion marker and, periodically, one <code>*.compact</code> merge file) are reassembled into parse order by <code>reassembleRollingEntries(names)</code>: drop the marker, drop every non-compact <code>events_*</code> file at or below the most recent <code>.compact</code> file&#39;s index, sort the rest numerically. Both the SHS-zip fetch path (<code>runParseFromUrl</code>) and the local folder-drop path (<code>runParseFiles</code>) call this one function.</p><p><code>runParseFiles</code> streams each file in the resulting order through the same chunked codec dispatch as <code>runParse</code>, with a fresh decompressor per file, since codecs never span a roll boundary. The NDJSON line-decoder is the exception: it stays alive across all files, because nothing guarantees a roll boundary lands on a line boundary.</p>',19)])])}const u=o(a,[["render",d]]);export{m as __pageData,u as default};
@@ -0,0 +1 @@
1
+ import{_ as o,o as r,c as t,a5 as c}from"./chunks/framework.DSg0KOwT.js";const m=JSON.parse('{"title":"Architecture overview","description":"","frontmatter":{},"headers":[],"relativePath":"contributor-guide/architecture/overview.md","filePath":"contributor-guide/architecture/overview.md"}'),a={name:"contributor-guide/architecture/overview.md"};function d(s,e,i,n,l,p){return r(),t("div",null,[...e[0]||(e[0]=[c("",19)])])}const u=o(a,[["render",d]]);export{m as __pageData,u as default};
@@ -0,0 +1 @@
1
+ import{_ as o,o as t,c as a,a5 as s}from"./chunks/framework.DSg0KOwT.js";const u=JSON.parse('{"title":"State and History Server intake","description":"","frontmatter":{},"headers":[],"relativePath":"contributor-guide/architecture/state-and-history.md","filePath":"contributor-guide/architecture/state-and-history.md"}'),d={name:"contributor-guide/architecture/state-and-history.md"};function r(c,e,n,i,l,h){return t(),a("div",null,[...e[0]||(e[0]=[s('<h1 id="state-and-history-server-intake" tabindex="-1">State and History Server intake <a class="header-anchor" href="#state-and-history-server-intake" aria-label="Permalink to &quot;State and History Server intake&quot;">​</a></h1><h2 id="state-model" tabindex="-1">State model <a class="header-anchor" href="#state-model" aria-label="Permalink to &quot;State model&quot;">​</a></h2><p><code>src/store/store.ts</code> is a single Zustand store (<code>createStore</code> from <code>zustand/vanilla</code>, wrapped by a <code>useStore</code> hook). It holds all shared state; no component keeps a <code>useState</code> of its own for anything shared: <code>appModel: AppModel</code>, <code>catalog: Finding[]</code>, <code>activeFileId</code>, <code>sessionCache</code> (in-memory snapshot cache for instant file-switching, see <code>session-snapshot.ts</code>), <code>taskDataCache</code>, <code>parse: {pct, lines, etaMs}</code>, <code>status: &#39;idle&#39;|&#39;parsing&#39;|&#39;ready&#39;|&#39;error&#39;</code>, <code>errorMessage</code>, <code>theme</code>, <code>skippedLines</code> (malformed-JSON-line count from the parser&#39;s <code>done</code> payload).</p><p><code>src/store/useIngest.ts</code> is the only writer during a parse. It builds <code>createModelCallbacks</code>&#39; <code>onProgress</code>/<code>onDone</code>/<code>onError</code> handlers to call the store&#39;s setters directly (<code>setParse</code>, <code>setCatalog</code>, <code>setStatus</code>, <code>setSkippedLines</code>, ...). Components never talk to the worker: they use <code>useIngest()</code>&#39;s returned actions (<code>startLoad</code>/<code>startLoadFolder</code>/<code>startLoadFromUrl</code>/<code>pickRecent</code>/<code>getTaskData</code>/ <code>resetToDropZone</code>) and the store&#39;s read state.</p><p><code>resetModel()</code> empties <code>appModel</code>/<code>catalog</code>/<code>taskDataCache</code>/<code>skippedLines</code> on every new parse or reset-to-drop-zone, and bumps <code>modelResetCount</code>. That counter has no setter of its own; only <code>PlanGraphRoute.tsx</code>&#39;s <code>store.subscribe</code> reads it, to evict the plan-graph model memo cache (see <a href="./drill-down.html#plan-graph-view">Plan graph view</a>).</p><h3 id="finding-filter-state" tabindex="-1">Finding filter state <a class="header-anchor" href="#finding-filter-state" aria-label="Permalink to &quot;Finding filter state&quot;">​</a></h3><p>The board-wide finding filter (impact band, raw <code>finding.type</code>, stage) lives outside the Zustand store, in <code>FindingFilterContext</code> (<code>src/view/FindingFilterContext.tsx</code>): a <code>createContext</code>+<code>useState</code><code>FilterSelection</code> (<code>src/view/finding-filter.ts</code>, three <code>Set</code>s) that every widget filters <code>catalog</code> through via <code>filterFindings</code>. It is seeded from the URL&#39;s <code>impact</code>/<code>type</code>/<code>stage</code> query params on mount and nowhere else, so a reload with no params gives the unfiltered board. Every change writes those params back with <code>history.replaceState</code>, never <code>push</code>, so filtering doesn&#39;t grow Back history. A <code>popstate</code> listener re-seeds the selection from the URL, so Back and forward re-apply filters.</p><p>Switching files resets the selection to empty, keyed on the stable file id rather than <code>catalog</code>, so a same-file catalog refresh keeps the active filter. A page reload re-reads the address bar, so deep-linked filtered URLs still restore. Filters are seeded only from the URL and never persisted anywhere else (no localStorage, no restore-as-default): a filtered view silently becoming the default on reload would risk hiding findings from a user who didn&#39;t realize a filter was still active, so a plain reload with no filter params is always the unfiltered board.</p><h3 id="recent-files-vs-session-cache" tabindex="-1">Recent files vs. session cache <a class="header-anchor" href="#recent-files-vs-session-cache" aria-label="Permalink to &quot;Recent files vs. session cache&quot;">​</a></h3><p>Two mechanisms cover reopening a file, at different lifetimes.</p><p><code>sessionCache</code> (in the Zustand store, see <a href="#state-model">State model</a>) is in-memory and per-session: it makes switching between files already loaded in the current tab instant, and it is gone on reload.</p><p>Recent files (<code>src/recent-files.ts</code>, consumed by <code>src/view/useRecentFiles.ts</code>) is IndexedDB-backed and cross-session. It persists each file&#39;s <code>FileSystemFileHandle</code> plus light metadata (name, size, <code>lastModified</code>, app name, issue count, <code>lastOpenedAt</code>), capped at 10 entries with the oldest evicted past the cap, so a file can be reopened after a full browser restart, pending the browser re-granting permission on the handle. Picking a recent entry re-parses from the handle; no parsed model is ever persisted.</p><h2 id="run-comparison" tabindex="-1">Run comparison <a class="header-anchor" href="#run-comparison" aria-label="Permalink to &quot;Run comparison&quot;">​</a></h2><p><code>src/run-comparison.ts</code> is the whole A/B engine. The entry point <code>compareRuns(baseline, candidate)</code> takes two <code>{ label, snapshot }</code> run records (each <code>snapshot</code> a normalized model: <code>app</code>, <code>stages</code>, <code>sql</code>, <code>catalog</code>, <code>executors</code>) and returns one plain object the view renders. It runs on already-parsed snapshots, with no worker involved.</p><ul><li><code>stageIdentity(stage, snapshot)</code> is a run-independent key: <code>normalizeStageName</code> (lowercased, digit-runs and long hex ids collapsed to <code>#</code>) joined with the stage&#39;s SQL-execution plan identity. That identity is scoped to only the plan nodes this stage&#39;s tasks were attributed to (<code>node.stageIds</code>), not the whole tree, so two stages sharing one SQL execution (e.g. a self-join&#39;s two Exchange stages) don&#39;t collapse onto one identity; it falls back to a bottom-up structural fingerprint of the whole resolved <code>planTree</code> (<code>planTreeIdentity</code>, <code>normalizeDetail</code>-normalized: the same normalizer <code>cachingOpportunity</code> uses in <code>src/detectors.ts</code>) when a stage has no such attribution. <code>matchStages(baseSnap, candSnap)</code> indexes each run by that identity and pairs identities that map to exactly one stage on both sides. An identity colliding equally on both sides (same count) is also paired, positionally by sorted stage id: exact when comparing a run against itself (every stage matches itself), a best-effort guess otherwise (two unrelated same-named stages with no SQL/attribution could get cross-paired). Collisions are still recorded in <code>collisionIdentities</code> even when resolved this way; a differing count leaves them there unpaired. <code>coverage</code> reports the matched fraction.</li><li><code>metricDeltas</code> computes whole-run aggregate deltas (wall-clock, spill, task skew p95, failed-task rate, GC, I/O bytes, executor count, ...) as plain sums over all stages, deliberately not gated on stage matching, since matching is unreliable on real logs. Each metric carries a <code>direction</code> (improvement/regression/unchanged) and an <code>unavailableReason</code> when a side lacks the field.</li><li><code>findingsDelta</code> tallies each run&#39;s <code>catalog</code> by <code>(rule × impact band)</code> and reports <code>introduced</code> vs <code>resolved</code> categories: a count diff, also matching-free.</li><li>Only the per-stage skew deltas (<code>stageSkewDeltas</code>) and the pinned-stage panel consume the <code>matchStages</code> pairs, so low match coverage degrades those two surfaces without invalidating the aggregate deltas.</li></ul><p><code>compareRuns</code> also flags <code>confidence: &#39;low&#39;</code> when the two app names differ, or independently when matched stage coverage falls below 0.5 (<code>LOW_COVERAGE_THRESHOLD</code>; either condition alone is enough, both are weak signals, not a hard gate). The view lives in <code>src/view/RunComparison.tsx</code> (the comparison page), <code>CompareLanding.tsx</code> (the two-slot Run A / Run B intake off the landing), and <code>PinnedStageDeltas.tsx</code> (the manual per-stage pinning panel fed by <code>baseStages</code>/<code>candStages</code>).</p><h2 id="history-server-intake-and-recovery" tabindex="-1">History Server intake and recovery <a class="header-anchor" href="#history-server-intake-and-recovery" aria-label="Permalink to &quot;History Server intake and recovery&quot;">​</a></h2><p><code>DropZone</code> keeps the History Server disclosure, Base URL, Application ID, optional Attempt ID, validation/touched state, recoverable SHS error, and local-server reachability in mounted React state rather than Zustand. The local browser-first path is still the default: <strong>Choose file</strong> loads a single event log, while <strong>Choose rolling-log folder</strong> accepts only an <code>eventlog_v2_*</code> directory and directs a rejected folder back to the file picker.</p><p>The three text fields also mirror to <code>window.localStorage</code> (<code>shuffle-works-shs-base-url</code>/<code>-app-id</code>/<code>-attempt-id</code>), read back as each <code>useState</code>&#39;s initializer, so a returning visitor&#39;s values survive a reload; storage access is wrapped in try/catch and silently ignored when unavailable, matching <code>store.ts</code>&#39;s <code>initialTheme</code>/<code>initialWidgetDensity</code> pattern. Both this disclosure&#39;s toggle and the <strong>Other sources</strong> toggle show a chevron (<code>ChevronDownIcon</code>/<code>ChevronUpIcon</code>) that flips with <code>aria-expanded</code>, so the open/closed state has a visual signal beyond the attribute.</p><p>On mount (skipped in <code>compact</code> mode), <code>DropZone</code> probes reachability with an empty, short-timeout <code>fetch(&#39;/shs-proxy&#39;)</code>: a 400 means <code>validateShsRequest</code> (<code>packages/core/src/proxy.js</code>) rejected the empty request synchronously, which only happens when a local server is actually routing that path, so it flips <code>shsReachable</code> to <code>true</code>. A network error, a 404 (static deploy, no such route), or a probe still in flight all leave <code>shsReachable</code> at its default <code>false</code>, so nothing changes on screen after paint unless the server is confirmed present. When <code>shsReachable</code> is <code>true</code>, the landing page shows a neutral callout above the <strong>Other sources</strong> disclosure pointing the user at it; the disclosure itself doesn&#39;t move or auto-expand.</p><p>The collapsed <strong>Fetch from Spark History Server</strong> disclosure requires local-server mode, a reachable History Server, and a supported base application ID: <code>application_&lt;timestamp&gt;_&lt;id&gt;</code>, <code>local-&lt;timestamp&gt;</code>, or <code>app-&lt;identifier&gt;</code>. <code>server/lib/shs-request.js</code> trims and validates the three request fields, accepts only absolute credential-free <code>http:</code>/<code>https:</code> base URLs without a query or fragment, preserves a reverse-proxy path prefix, and canonicalizes the base URL to one trailing slash. The optional attempt is a separate path-safe identifier; neither identifier can contain a path separator. The shared helper builds the encoded <code>/shs-proxy</code> request and the encoded <code>api/v1/applications/&lt;app&gt;[/&lt;attempt&gt;]/logs</code> upstream path from that normalized object only.</p><p>The optional Node server is loopback-only: a narrow CORS proxy, not a general or hosted proxy. It validates the same request contract, sends no credentials, follows no upstream redirects, and returns only stable safe error codes. It never forwards upstream response text, status details, locations, or credentials to the browser.</p><p>Routing preserves the recovery boundary: local file and folder failures use the existing page-level error route. A typed SHS failure instead resets the model to idle and returns to the still-mounted, expanded History Server disclosure, which retains its values and shows safe recovery guidance along with local file intake. During an SHS parse, the mounted intake shows progress in place; successful completion follows the normal dashboard route.</p>',23)])])}const g=o(d,[["render",r]]);export{u as __pageData,g as default};
@@ -0,0 +1 @@
1
+ import{_ as o,o as t,c as a,a5 as s}from"./chunks/framework.DSg0KOwT.js";const u=JSON.parse('{"title":"State and History Server intake","description":"","frontmatter":{},"headers":[],"relativePath":"contributor-guide/architecture/state-and-history.md","filePath":"contributor-guide/architecture/state-and-history.md"}'),d={name:"contributor-guide/architecture/state-and-history.md"};function r(c,e,n,i,l,h){return t(),a("div",null,[...e[0]||(e[0]=[s("",23)])])}const g=o(d,[["render",r]]);export{u as __pageData,g as default};
@@ -0,0 +1 @@
1
+ import{_ as o,o as t,c as d,a5 as a}from"./chunks/framework.DSg0KOwT.js";const p=JSON.parse('{"title":"Widget rendering","description":"","frontmatter":{},"headers":[],"relativePath":"contributor-guide/architecture/widget-rendering.md","filePath":"contributor-guide/architecture/widget-rendering.md"}'),i={name:"contributor-guide/architecture/widget-rendering.md"};function n(c,e,r,s,l,g){return t(),d("div",null,[...e[0]||(e[0]=[a('<h1 id="widget-rendering" tabindex="-1">Widget rendering <a class="header-anchor" href="#widget-rendering" aria-label="Permalink to &quot;Widget rendering&quot;">​</a></h1><h2 id="widget-rendering-order-fixed-spec-§5" tabindex="-1">Render order (fixed, spec §5) <a class="header-anchor" href="#widget-rendering-order-fixed-spec-§5" aria-label="Permalink to &quot;Render order (fixed, spec §5) {#widget-rendering-order-fixed-spec-§5}&quot;">​</a></h2><p><code>src/view/Dashboard.tsx</code>&#39;s <code>FilteredBoard</code> renders inside a <code>&lt;main&gt;</code> that opens with <code>FindingFilterBar</code> and (only when an active filter empties both finding streams) <code>NoMatchBanner</code>, then a single <code>Scorecard</code> strip, then (when the active filter doesn&#39;t empty the board) a two-tab <code>Tabs</code> (<code>src/components/ui/tabs.tsx</code>, a base-ui primitive): <strong>Findings</strong> and <strong>Full app report</strong> (2026-09-03 tabbed-impact-band-board redesign, replacing the prior three stacked sections: All recommendations, Suggested Improvements, Full app report, with a merged, impact-grouped Findings tab and an always-reachable Full app report tab). Base-ui <code>Tabs</code> fully unmount the inactive <code>TabsContent</code> panel rather than hiding it: a widget mounted only in Findings (every routeable <code>REGISTRY</code> widget; see &quot;First investigation routing&quot; below) is not in the DOM at all while Full app report is active, and remounts fresh, with its own state reset, when the user switches back.</p><p><code>region</code> on <code>RegistryEntry</code> (<code>src/view/detector-registry.tsx</code>) is read again, but only to decide whether a widget always mounts: <code>isAlwaysMountedType()</code> flags the one <code>reference</code>-region type still carved out as an always-mounted exception, <code>coreLocality</code> → <code>CoreUsageArea</code>. The other <code>reference</code>-region types stay ordinary finding-gated instead: <code>cacheUtilization</code>, <code>memoryUtilization</code>, and <code>utilization</code> are a product decision: a clean run on any of them isn&#39;t evidence worth surfacing unconditionally, so each collapses to a plain clean-check line like any other detector. The Findings tab (<code>src/view/widgets/ImpactBoard.tsx</code>) groups its content into three tiers, same as the retired Suggested Improvements section did, just re-sliced by impact band instead of living as one flat active grid. The three tiers: impact-banded rows and cards for every <code>REGISTRY</code> component and every recommendation row with at least one finding; a small always-visible grid holding just the one exception above (mounted unconditionally from <code>appModel</code> regardless of finding state, below the impact bands); and a collapsed &quot;Clean checks&quot; disclosure covering every remaining type with zero findings, built per detector <em>type</em> (<code>Object.keys(REGISTRY)</code>). Full app report stays structural-only (see &quot;ReferenceSection&quot; below).</p><p>Tags carry their own docs links; there is no separate legend widget. <code>TagBadge</code> (<code>src/view/ImpactBadge.tsx</code>) resolves its own tooltip. When its type resolves a single known documentation anchor (<code>docAnchorForType</code>, <code>src/view/finding-tag-help.ts</code>), the pill itself links into the docs panel, and (density <code>advanced</code> only) a second icon link opens that tag&#39;s entry in <code>docs-site/user-guide/understanding-findings.md</code> (<code>findingGuideUrl</code>, <code>packages/core/src/docs-site-config.ts</code>) as a plain new-tab navigation. When a type has no vendor-doc anchor (e.g. <code>incompleteRun</code>), the pill links straight to that same guide entry instead of rendering as inert text, and the second icon link is skipped as redundant. Either way, a linked pill gets a visible dotted underline, not just a hover tooltip. <code>TagBadge</code>&#39;s <code>plainBadge</code> prop suppresses both links together; a Findings-tab recommendation row is the one place that never sets it (see below), since its badge cell isn&#39;t nested inside a button.</p><h3 id="findings-tab" tabindex="-1">Findings tab <a class="header-anchor" href="#findings-tab" aria-label="Permalink to &quot;Findings tab&quot;">​</a></h3><p>Rendered by <code>FilteredBoard</code>&#39;s <code>TabsContent value=&quot;findings&quot;</code> (<code>src/view/widgets/ImpactBoard.tsx</code>). Merges what used to be two stacked sections, All recommendations and Suggested Improvements, into one impact-ranked board. The recommendation-rollup logic lives in <code>src/view/widgets/FixTheseFirst.tsx</code> (kept as its own file and its own directly-testable exports, but no longer rendered as a standalone page section by <code>Dashboard.tsx</code>) and the active-widget logic lives in <code>src/view/widgets/Alerts.tsx</code> (same: kept, no longer rendered standalone). <code>ImpactBoard</code> calls <code>FixTheseFirst.tsx</code>&#39;s exported <code>useFixTheseFirstData</code> (eligible findings, rollup groups, the top triage target) and <code>Alerts.tsx</code>&#39;s exported <code>computeActiveWidgets</code> (the ranked active-<code>REGISTRY</code> list) to get the exact same data these two retired sections used to render independently, then regroups both by impact band via <code>groupImpactBand</code> (<code>FixTheseFirst.tsx</code>: a rollup group&#39;s representative member&#39;s impact band, the same finding whose impact band its own badge already shows) and each active widget&#39;s own <code>worstImpactBand</code>.</p><p>Eligible findings for the rollup are <code>catalog</code> ∪ <code>configFindings</code>, filtered to a <code>REGISTRY</code>-mapped type, with <code>incompleteRun</code> (a pipeline-completeness caveat, not an addressable fix; see the spec&#39;s <code>fixEffort</code> table) and <code>memoryUtilization</code>&#39;s <code>memoryBand</code>/<code>dataUnavailable</code> variant (a missing-evidence caveat already covered by Evidence availability&#39;s own <code>executorMetrics</code> entry, <code>packages/core/src/evidence-availability.ts</code>) both explicitly excluded. Findings are grouped strictly by <code>finding.type</code> via <code>buildRecommendationRollup</code> (<code>packages/core/src/recommendation-rollup.ts</code>, further split within a type by impact kind and, for <code>resource</code>, unit: never merged across types); each resulting group becomes one row: a type with exactly one finding renders that finding directly, a type with more than one collapses into a summary row. Within an impact band, group order comes from <code>buildRecommendationRollup</code>&#39;s own sort: <code>time</code> groups (a real <code>wallClock</code> claim) first, ranked among themselves by their union-capped <code>recoverableMsHigh</code> descending; then <code>resource</code> groups (<code>rawWaste</code> but no <code>wallClock</code>); then <code>count</code> groups (neither), both of the latter two ranked by worst impact band, never by their incomparable raw magnitudes. A summary row&#39;s tag and dot, plus its text, all come from the group&#39;s own highest-impact member (via the same three-tier comparator, computed locally in <code>FixTheseFirst.tsx</code>); the trailing stat depends on the group&#39;s kind (<code>×N · &lt;time&gt; recoverable</code>, <code>×N · resource-cost projection</code>, or a plain per-impact-band tally). Clicking it expands straight to the group&#39;s full, impact-ranked list, with no intermediate &quot;worst-K&quot; step, paginated at 10 rows per page (<code>data-testid=&quot;fix-these-first-group-row&quot;</code>; no pager renders for a group of 10 or fewer findings; it appears once a group exceeds 10). A band&#39;s rollup rows render as a headerless three-column <code>Table</code> (<code>src/components/ui/table.tsx</code>): every individual row, whether shown directly or inside an expanded group, is a <code>TableRow</code> (<code>data-testid=&quot;fix-these-first-row&quot;</code>, <code>data-finding-type</code>) with three <code>TableCell</code>s: the impact dot + ALL-CAPS tag as a real <code>TagBadge</code> (not <code>plainBadge</code>: nothing wraps it, so its own docs links stay real <code>&lt;a&gt;</code>s, same as everywhere else on the board); a text block inside its own nested <code>&lt;button&gt;</code> (a short imperative action label, e.g. &quot;Reduce shuffle size&quot;, from <code>findingActionLabel</code> (<code>src/view/finding-action-label.ts</code>), over the finding&#39;s own full <code>recommendation</code> sentence in smaller muted text, both wrapping rather than truncating); and a right-aligned monospace stage reference + impact figure (e.g. <code>St.49 · 20.1s</code>, via <code>ImpactEstimate.tsx</code>&#39;s shared <code>formatWallClockRange</code>/<code>formatRawWaste</code>). That inner button, not the row, is the click target: it routes via <code>selectTriageTargetForFinding</code> (<code>src/view/triage-target.ts</code>), the same per-finding resolver Stage Summary Table&#39;s own control uses (see &quot;First investigation routing&quot; below); a <code>TypeGroupRow</code>&#39;s own inner button toggles its expand state instead (<code>aria-expanded</code>) and its expanded findings render as further <code>TableRow</code>s indented one badge-cell notch to read as the group&#39;s sub-list.</p><p>Each impact band (<code>ImpactBoard.tsx</code>&#39;s own <code>ImpactGroup</code>, one call per entry of <code>IMPACT_BAND_ORDER_LIST = [&#39;critical&#39;, &#39;warning&#39;, &#39;info&#39;]</code>) is a <code>&lt;section aria-label=&quot;Critical&quot; | &quot;Warning&quot; | &quot;Info&quot;&gt;</code> with an <code>&lt;h3&gt;</code> heading, and renders nothing (not even the heading) when it has neither a rollup row nor an active widget: a run with no critical findings has no &quot;Critical&quot; heading or section at all. Inside a band, rollup rows render first as the headerless <code>Table</code> described above, followed by that band&#39;s active <code>REGISTRY</code> widget cards (<code>computeActiveWidgets</code>&#39;s ranked list, filtered to this impact band) in their own <code>WidgetGrid</code>: every one of <code>orderedWidgets()</code>&#39;s deduped <code>REGISTRY</code> components <em>except</em> the one always-mounted one below, with at least one finding in <code>catalog</code> ∪ <code>configFindings</code>. Within a band, active widgets keep <code>orderedWidgets()</code>&#39;s own <code>action</code>-region-first, ascending-<code>DETECTORS</code>-order tiebreak. <code>cacheUtilization</code>, <code>memoryUtilization</code>, and <code>utilization</code> are <code>reference</code>-region types but aren&#39;t always-mounted exceptions, so a Cache Storage, Memory Utilization, or Executor Utilization card with an active finding surfaces in its own impact band like any other active widget.</p><p>Below the impact bands, <code>Alerts.tsx</code>&#39;s exported <code>AlwaysVisibleAndCleanChecks</code> (shared verbatim with the retired standalone <code>Alerts</code> component) renders the same two tiers it always did, now living outside the impact-band grouping entirely rather than as this section&#39;s second and third tier: a small always-visible grid holding just Core Usage by Locality (<code>coreLocality</code>, resolving to <code>CoreUsageArea</code>), mounted unconditionally from <code>appModel</code> regardless of finding state (<code>alwaysMountedWidgets()</code>/<code>isAlwaysMountedType()</code>), carrying its own impact-band indicator when a finding is active instead of collapsing to a clean-check line on a clean run; and a collapsed &quot;Clean checks&quot; disclosure of <code>CleanCheckRow</code> lines (<code>src/view/widgets/CleanCheckRow.tsx</code>: label, the threshold it was measured against via <code>getThresholdSummary</code>, and &quot;No fix needed.&quot;) built per detector <em>type</em> (every <code>REGISTRY</code> key except that one always-mounted key): a clean run lands <code>cacheUtilization</code>, <code>memoryUtilization</code>, and <code>utilization</code> here too, same as any ordinary action-region type. Caching Opportunities, Config Audit, and the four split Plan Advisor widgets (Redundant Plan Subtree, Excessive Small Files, Missed Broadcast Join, Oversized Broadcast Join) render through the ordinary active/clean paths above (see <a href="./board-widgets.html#board-widgets-beyond-the-fixed-six">Board widgets beyond the fixed six</a>). One consequence of this always-visible grid sitting below every impact band: a <code>critical</code>-band <code>coreLocality</code> finding still renders in that lower grid, below the <code>info</code>-band widgets above it, a deliberate tradeoff the spec accepted in exchange for never losing the widget on a clean run, not a ranking bug.</p><h3 id="per-widget-list-sort-mode" tabindex="-1">Per-widget list sort mode <a class="header-anchor" href="#per-widget-list-sort-mode" aria-label="Permalink to &quot;Per-widget list sort mode&quot;">​</a></h3><p><code>src/view/impact-sort.ts</code> (<code>sumWallClockLow</code>, <code>hasSortableImpact</code>, <code>byImpactDesc</code>, <code>stageIdOf</code>, <code>minStageId</code>, <code>byStageAsc</code>) and <code>src/view/SortModeToggle.tsx</code> are a shared, opt-in pair a Findings-tab active widget can use to let its own expanded, multi-row list default to potential-savings order (<code>wallClock.low</code> descending, the guaranteed-floor bound, not the optimistic <code>high</code>) instead of stage number. Each of the widgets below holds local <code>sortMode</code> state (<code>SortMode</code>, <code>&#39;impact&#39; | &#39;stage&#39;</code>, default <code>&#39;impact&#39;</code>) and renders a <code>SortModeToggle</code> in the <code>WidgetCard</code> <code>badges</code> slot, so it sits on the title row itself rather than floating in the body: a button alongside the widget&#39;s tag badges (a <code>WidgetCard</code> header renders <code>badges</code> beside the title heading, never inside another control). Every one of the former combined widgets&#39; split single-type widgets carries the same <code>canToggleSort</code> check its parent did, so the toggle only actually renders for a type whose finding can carry a wall-clock estimate: <code>Skew.tsx</code>, <code>TinyTask.tsx</code> (<code>stageShape</code>&#39;s own rules never produce one, so <code>StageShape.tsx</code> carries the same check but it never fires), <code>ShuffleIO.tsx</code> (narrowed to <code>shuffle</code>), <code>PartitionSizing.tsx</code>, <code>Spill.tsx</code>, <code>GcPressure.tsx</code> (both its high-GC and low-GC sections, one shared toggle), <code>RetryWaste.tsx</code> (its siblings <code>StageFailed.tsx</code>/<code>TaskFailures.tsx</code> carry the same check, but <code>stageFailed</code>/<code>failures</code> are <code>estimateMethod: &#39;none&#39;</code>, so it never fires there either), <code>SlowHost.tsx</code>, <code>StageSlowness.tsx</code>, <code>Straggler.tsx</code>, <code>SpeculationWaste.tsx</code>, <code>ColdStart.tsx</code> (five widgets now, one per type, each sorting only its own flat issue list; the old cross-type &quot;coldStart sinks to the bottom under By stage&quot; note no longer applies now that each type has its own single-type list), and <code>DuplicatePlanSubtree.tsx</code>/ <code>SmallFiles.tsx</code>/<code>UnderBroadcast.tsx</code>/<code>OverBroadcast.tsx</code> (each reorders its own list by <code>stageIdOf</code>, the lowest stage id its finding touches; the old cross-type <code>minStageId</code> group-ordering tiebreak no longer applies, since each type is its own widget now). Its <code>alternateOrderLabel</code> prop is required and every caller passes &quot;By stage&quot;: stage number is the one axis every one of these widgets&#39; items can be compared on, unlike the impact/raw-metric order this pattern replaced (dropped: comparing findings by impact band ranks them by how bad they are, not by how much fixing them would save, which is a worse default now that a real potential-savings figure exists to sort by instead). A per-stage detector&#39;s own <code>finding.stageId</code> is the sort key directly; a sql-scope finding that spans several stages (<code>duplicatePlanSubtree</code>, <code>smallFiles</code>, <code>underBroadcast</code>, <code>overBroadcast</code>, via <code>stageIds</code>) sorts by the lowest stage id it touches (<code>stageIdOf</code>). The toggle only renders when <code>hasSortableImpact</code> finds at least one wall-clock claim in the list; re-sorting a list with none would be a silent no-op. Both comparators return <code>0</code> when neither side has a comparable value, so those items keep their prior relative order (<code>Array.prototype.sort</code>&#39;s stability) rather than being shuffled. <code>FixTheseFirst</code>&#39;s own expanded group list (see &quot;Findings tab&quot; above) already sorted by impact before this pattern existed and does not use it; it has no stage-order toggle.</p><h3 id="gold-standard-row-expand-contract" tabindex="-1">Gold Standard row/expand contract <a class="header-anchor" href="#gold-standard-row-expand-contract" aria-label="Permalink to &quot;Gold Standard row/expand contract&quot;">​</a></h3><p>Every <code>REGISTRY</code> widget renders its content as N≥1 rows, using <code>Skew.tsx</code> as the reference implementation. A widget&#39;s data shape (chart, table, single app-scoped scalar) is never by itself a reason to skip this: the three named exceptions below are the only ones. Any further exception must get its own entry here, justified in the same PR that introduces it.</p><ul><li>Tier A (universal): <code>WidgetCard</code> chrome, an impact-band left-border + dot, ALL-CAPS <code>TagBadge</code>s, a <code>SortModeToggle</code> when <code>hasSortableImpact</code> is true, a <code>finding-anchor</code> ref on any focusable/deep-linkable unit, and an early <code>null</code> return on an empty catalog filter (except Core Usage by Locality, the one always-mounted widget that renders unconditionally from <code>appModel</code> via <code>alwaysMountedWidgets()</code>/<code>isAlwaysMountedType()</code> rather than early-returning on an empty catalog filter).</li><li>Tier B (the row/expand pattern): a row&#39;s collapsed state shows its core metric(s), <code>ImpactEstimate</code>, its recommendation text, any config-hint code snippet/list, and docs links, all unconditionally. Confidence and evidence are unconditional too, since the 2026-09 redesign: <code>RowStatusCluster</code> (<code>src/view/RowStatusCluster.tsx</code>) is a single, fixed-position pill combining both, replacing the old per-row <code>ExpandToggleButton</code> + <code>ConfidenceMarker</code> + <code>EvidenceLink</code> trio entirely. Every widget that carries one wraps it in <code>AdvancedOnly</code> (<code>src/view/AdvancedOnly.tsx</code>), so it only renders at the Advanced density tier: this is a content-visibility gate (Basic vs. Advanced), not a per-row collapse, and it&#39;s applied consistently across every adopting widget. There is nothing left to gate behind a per-row click for confidence or evidence in any widget. A widget-header <code>RowStatusCluster</code> (passed as <code>WidgetCard</code>&#39;s <code>statusBadge</code> prop) carries one further, unrelated gate on top of <code>AdvancedOnly</code>: <code>open &amp;&amp; statusBadge</code> (<code>WidgetCard.tsx:138,141</code>) keeps it out of the collapsed header, matching the comment on <code>statusBadge</code> itself (&quot;kept separate so tag/impact badges stay visible in the collapsed summary while a <code>RowStatusCluster</code>-style ... marker doesn&#39;t&quot;). This is the card&#39;s own expand/collapse, not a separate reveal-on-click built for the cluster, and it composes with density as an AND: a header cluster needs both Advanced tier and an expanded card. A row with neither renders no cluster at all (<code>RowStatusCluster</code> returns <code>null</code>). Confidence renders as plain, non-interactive text (&quot;<code>{confidence} confidence</code>&quot;, no further suffix) with a <code>title</code> + <code>aria-describedby</code> tooltip carrying the full <code>validationRequired</code> detail: never itself a click target. Evidence renders as the cluster&#39;s one real click target: a button whose visible text is just the evidence label (e.g. &quot;SQL plan&quot;) but whose accessible name always carries the &quot;Evidence: ...&quot; prefix via an explicit <code>aria-label</code>, regardless of what else sits nearby (the old <code>bare</code> prop/prefix distinction is gone: there&#39;s only one form now). Clicking it calls the same <code>revealEvidence</code> navigation the retired <code>EvidenceLink</code> used. <code>RowStatusCluster</code>&#39;s fixed slot is the row&#39;s own stage-pill/title line (<code>justify-between</code>, cluster right-aligned), the same position in every adopting widget, not floating with the recommendation text below. Lists longer than <code>VISIBLE_LIMIT</code> (6, <code>packages/core/src/format-utils.ts</code>) still get page-based navigation (Previous/Next, <code>usePagedRows</code>/<code>RowPagination</code>, <code>src/view/usePagedRows.ts</code>/<code>src/view/RowPagination.tsx</code>) instead of rendering unconditionally: the one sanctioned cap mechanism across every Tier B widget. A Tier B widget whose rows are also triage-routable (wires <code>useFindingAnchor</code>) must pass its <code>routeIndex</code> (the routed finding&#39;s position in the paginated list) into <code>usePagedRows</code>, or a deep-linked route to a finding beyond the first page silently degrades to focusing the widget&#39;s disclosure title instead of the row.</li></ul><p>The following widgets carry a <code>RowStatusCluster</code>, all Advanced-only: <code>Spill.tsx</code> (confidence only), <code>ConfigAudit.tsx</code>&#39;s widget-header control (evidence only, a widget-wide constant rather than per-finding data, since the evidence key doesn&#39;t vary per row; also the control shown in the widget&#39;s empty-findings/no-data states), <code>MemoryUtilization.tsx</code>&#39;s rows (confidence + evidence together; <code>ExecutorUtilization.tsx</code>&#39;s rows, split out of the same former combined widget, carry neither), each of the four split Plan Advisor widgets&#39; (<code>DuplicatePlanSubtree.tsx</code>, <code>SmallFiles.tsx</code>, <code>UnderBroadcast.tsx</code>, <code>OverBroadcast.tsx</code>) per-row control (evidence only) plus their own widget-header marker (confidence only, one marker per widget now that each is its own finding-type, replacing the old <code>PlanFindings.tsx</code> per-group heading), <code>CoreUsageArea.tsx</code>, <code>EfficiencyModel.tsx</code>, <code>PlanView.tsx</code> (three call sites), <code>ScalingSim.tsx</code> (two call sites), and <code>WastedCoreHours.tsx</code> (all confidence-only, standalone rather than per-finding-row: a card-header badge, a widget-level single marker, a plan-tree node/summary-row marker, or an &quot;unavailable data&quot; message: the same component, same visual language, regardless of where it sits); and, since <code>skew</code>/<code>straggler</code>/<code>gc</code> started disclosing their own unvalidated noise-floor thresholds (confidence only, per-row, no <code>evidenceKey</code> passed), <code>GcPressure.tsx</code>&#39;s rows, <code>Straggler.tsx</code>&#39;s rows, <code>CachingOpportunity.tsx</code>&#39;s rows (the <code>cachingOpportunity</code> detector scales <code>confidence</code> per finding via <code>cachingReuseConfidence</code>, so the badge sits next to each row&#39;s recommendation rather than as a single caveat below the table), and the shared <code>StageFindingGroup.tsx</code> row (adopted by <code>Skew.tsx</code>/<code>StageShape.tsx</code>/<code>TinyTask.tsx</code>, though today only <code>skew</code> findings actually carry a <code>confidence</code> field).</p><p>One named exception:</p><ul><li><code>Skew.tsx</code>/<code>StageShape.tsx</code>/<code>TinyTask.tsx</code>: their per-row histogram toggle (<code>ExpandToggleButton</code> + <code>useExpandableRow</code>) gates a lazily-fetched duration histogram, unrelated to confidence/evidence, so it was untouched by the 2026-09 redesign and untouched again when <code>RowStatusCluster</code> was later added alongside it in the shared <code>StageFindingGroup.tsx</code> row. The two controls coexist per row. <code>ExpandToggleButton</code> is now single-purpose (always the task-detail toggle) since these three (split out of the former combined <code>TaskSkew.tsx</code>) are its only remaining adopters.</li></ul><p>No toggle at all, unconditional recommendation/content, same as before this redesign (unaffected either way, since these widgets never carried confidence/evidence display in the first place): <code>IncompleteRun.tsx</code>, <code>ShuffleIO.tsx</code>, <code>PartitionSizing.tsx</code>, <code>StageFailed.tsx</code>, <code>TaskFailures.tsx</code>, <code>RetryWaste.tsx</code>, <code>ColdStart.tsx</code>, <code>SlowHost.tsx</code>, <code>StageSlowness.tsx</code>, <code>SpeculationWaste.tsx</code>, <code>ExecutorCountChart.tsx</code>, <code>ExecutorUtilization.tsx</code>, <code>JobFailures.tsx</code>, <code>CacheUtilization.tsx</code>, <code>AutoscalingChurn.tsx</code>, and the four split Plan Advisor widgets (<code>DuplicatePlanSubtree.tsx</code>, <code>SmallFiles.tsx</code>, <code>UnderBroadcast.tsx</code>, <code>OverBroadcast.tsx</code>), split out of, respectively, the former combined <code>ExecutorTimeline.tsx</code>, <code>Failures.tsx</code>, and <code>PlanFindings.tsx</code>, none of which carried a per-row toggle either.</p><p>Three named exceptions to specific pieces of the contract, not to the row wrapper or pagination, which still apply to all three:</p><ul><li><code>CoreUsageArea.tsx</code>&#39;s non-local-stage list is a derived stat breakdown (<code>computeCoreLocalityRatio</code>&#39;s <code>topStages</code>), not a <code>Finding[]</code>: there is no per-row finding or recommendation to reveal. Paginated like every other Tier B list, but with no per-row expand.</li><li><code>CacheUtilization.tsx</code>&#39;s RDD <code>&lt;Table&gt;</code> holds reference columns (name, storage level, partitions, memory, disk bytes) with no recommendation attached to any row. Paginated like every other Tier B list, but with no per-row expand. Its separate recommendation list below the table is full Tier B (unconditional, no toggle, per the paragraph above).</li><li><code>IncompleteRun.tsx</code> leads with no metric of its own: this app-scoped, single-finding widget has no natural lead figure distinct from its title/badge, unlike every other Gold Standard widget, which leads with a percentage, duration, or count.</li></ul><h3 id="referencesection" tabindex="-1">ReferenceSection <a class="header-anchor" href="#referencesection" aria-label="Permalink to &quot;ReferenceSection&quot;">​</a></h3><p>Rendered by <code>FilteredBoard</code>&#39;s <code>TabsContent value=&quot;full-report&quot;</code>, tab label &quot;Full app report&quot;. A plain <code>&lt;div&gt;</code>, structural-only, reading none of <code>REGISTRY</code>/<code>orderedWidgets()</code> at all. Its heading is a visually-hidden (<code>sr-only</code>) <code>&lt;h2&gt;Full app report&lt;/h2&gt;</code>, matching the tab&#39;s own label for the accessibility-tree heading outline without visually duplicating the tab text. No longer an <code>Accordion</code>: base-ui <code>Tabs</code> unmount this panel entirely while Findings is active (see &quot;Render order&quot; above), so there is no &quot;collapsed but present&quot; state left to model, and this content is simply not in the DOM until the user clicks the Full app report tab. Its exact order is WallClock → Timeline → Executor Count Over Time (<code>ExecutorCountChart.tsx</code>, the executor add/remove count chart extracted out of the former combined <code>ExecutorTimeline.tsx</code>; not driven by any finding, so it isn&#39;t a <code>REGISTRY</code> entry) → StageTable → a <code>WidgetGrid</code> holding Evidence availability → ETL Phase Attribution → What-If Executor Scaling → Compute Efficiency → Wasted Core-Hours → Core-Usage Distribution. Scorecard used to lead this section; it now renders once, above the tabs themselves, in <code>FilteredBoard</code> (<code>src/view/Dashboard.tsx</code>), so it stays visible regardless of which tab is active rather than living inside either one (a three-tile run-info row: Wall-clock, Efficiency, Wastage; see <a href="./board-widgets.html#board-widgets-beyond-the-fixed-six">Board widgets beyond the fixed six</a>). WallClock, Timeline, StageTable and every tile in the grid beside them (Evidence availability included) all render immediately and fully expanded. No detector-driven <code>REGISTRY</code> card renders in this section any more: Memory Utilization, Executor Utilization, Core Usage by Locality, and Cache Storage all moved to the Findings tab above (Cache Storage, Memory Utilization, and Executor Utilization only surface there when they have an active finding; Core Usage by Locality alone is always-mounted).</p><p>The Evidence availability card is the persistent, non-impact-band ledger <a href="./worker-protocol.html#evidence-availability-contract-v1">defined in the worker protocol</a>, not an alert or detector widget. An <code>Evidence: …</code> control appears only where a conclusion or unavailable report lens declares a relevant ledger dependency. <code>revealEvidence</code> (<code>src/view/EvidenceAvailabilityContext.tsx</code>) sets <code>referenceOpen</code> true, which <code>DashboardContent</code> (<code>src/view/Dashboard.tsx</code>) watches in a <code>useEffect</code> and translates into <code>setActiveTab(&#39;full-report&#39;)</code>, replacing what used to be &quot;expand the Reference accordion&quot;: mouse and keyboard activation switches to the Full app report tab, opens the ledger card, then focuses the referenced stable entry id (<code>evidence-availability-&lt;key&gt;</code>). The control explains evidence availability; it does not promise an unavailable signal would have produced a finding.</p><p><code>WidgetGrid</code> (both inside a Findings impact band and inside <code>ReferenceSection</code>): collapsed cards occupy one responsive column; opening a card expands it across the full row. Nested <code>WidgetCard</code>s are isolated from the parent grid state. A Findings-tab <code>FixTheseFirst</code> rollup row is a table row, not a <code>WidgetGrid</code> card, and carries no collapse/expand state of its own (a <code>TypeGroupRow</code> summary row does own its own local expand/pagination state). Full app report&#39;s non-<code>REGISTRY</code> tiles (ETL Phase Attribution, What-If Executor Scaling, Compute Efficiency, Wasted Core-Hours, Core-Usage Distribution) sit in their own <code>WidgetGrid</code> alongside Evidence availability, but that grid has no <code>widgetId</code> wired to any card (see &quot;First investigation routing&quot; below), so nothing there is ever a route destination.</p><p>Firm constraint: <code>orderedWidgets()</code> (in <code>src/view/detector-registry.tsx</code>) still runs the single <code>DETECTORS</code>-ascending sort exactly as <code>dashboard-renderer.js</code> used to, and still sorts <code>action</code> components ahead of <code>reference</code> ones within that one list; the component-identity dedup it used to need is gone now that every <code>REGISTRY</code> entry maps to its own unique component (2026-09 widget/finding-type 1:1 mapping redesign). <code>cacheUtilization</code>, <code>memoryUtilization</code>, and <code>utilization</code> (<code>reference</code>-region) can still reach the active grid alongside <code>duplicatePlanSubtree</code> (<code>action</code>-region) and any other active finding, since none of them is the one always-mounted exception filtered out before that grid. <code>computeActiveWidgets</code> (<code>Alerts.tsx</code>) is still its main consumer, now filtered through <code>isAlwaysMountedType()</code> to exclude that one always-mounted component; the clean-check list bypasses <code>orderedWidgets()</code> entirely, iterating <code>Object.keys(REGISTRY)</code> per type instead (see &quot;Findings tab&quot; above). <code>DETECTORS</code>&#39; own array order and iteration, plus its cross-detector <code>suppressWhen</code> logic (e.g. <code>stageSlowness</code> deferring to <code>slowHost</code>, see <a href="./detector-contract.html#detector-contract">Detector contract</a>), live entirely in <code>packages/core/src/detectors.ts</code>/<code>packages/core/src/analyzer.ts</code>, untouched by this redesign. Which React component each finding type resolves to is still registry-owned.</p><h3 id="first-investigation-routing" tabindex="-1">First investigation routing <a class="header-anchor" href="#first-investigation-routing" aria-label="Permalink to &quot;First investigation routing&quot;">​</a></h3><p>The dashboard has a view-only per-finding route, not a single first-investigation pick any more: every Findings-tab recommendation row (rendered by <code>FixTheseFirst.tsx</code>&#39;s row components inside <code>ImpactBoard</code>) and the visible-finding control in Stage Summary (<code>StageTable.tsx</code>) each independently resolve and request their own target. <code>detector-registry.tsx</code> owns each routeable finding type&#39;s stable <code>widgetId</code>, widget title, and finding label; identity and copy come from there, not from runtime position or raw detector type. Eligible targets have a mapped, routeable registry entry and a non-empty trimmed recommendation (<code>targetForFinding</code>, <code>src/view/triage-target.ts</code>). The target&#39;s widget still renders every affected stage in its local order.</p><p><code>src/view/triage-target.ts</code> also still exports <code>selectTriageTarget</code>, the whole-catalog &quot;single highest-potential-savings finding&quot; picker that used to back a start-here callout and each per-tag headline tile in the retired <code>ProblemHeadlines.tsx</code> (potential savings first via <code>impactEstimate.wallClock.high</code>, then <code>orderedWidgets()</code> widget order, then catalog order; impact band plays no part). Ranking by potential savings replaced the earlier severity-first routing once the occupancy-weighted impact estimator gave every finding a real, comparable <code>impactEstimate.wallClock</code> figure: severity-first could point the single &quot;start here&quot; callout at a <code>skew</code>/<code>straggler</code> finding ranked <code>critical</code> on a ratio basis while its occupancy-clipped recoverable time was near zero, passing over a lower-severity finding with an order-of-magnitude larger real recoverable-time estimate right next to it. Nothing in the app calls <code>selectTriageTarget</code> any more: <code>FixTheseFirst</code> already ranks every eligible finding by impact magnitude, so a separate single-winner pick has no call site left. <code>tests/view/triage-target.test.ts</code> still covers it directly.</p><p>The route is re-derived from the current catalog before each asynchronous step. Its identity is object-reference equality against the current catalog (<code>catalog.includes(finding)</code> in <code>selectTriageTargetForFinding</code>, <code>src/view/triage-target.ts</code>) rather than a derived key: a new parse yields new finding object references, so reference presence alone detects staleness. It is not persisted and does not extend the core <code>Finding</code> contract; cross-session consumers (exports, future URL-restored state) use the core <code>Finding.id</code> instead (see <a href="./worker-protocol.html#finding-identity">Finding identity</a>).</p><p><code>Dashboard</code> owns disclosure and navigation. Every routeable <code>REGISTRY</code> widget now lives in the Findings tab (no <code>REGISTRY</code> widget renders inside Full app report any more; see &quot;Render order&quot; above), so <code>requestRoute</code> (<code>DashboardContent</code>, <code>src/view/Dashboard.tsx</code>) starts with an unconditional <code>setActiveTab(&#39;findings&#39;)</code>, whether the request came from a control already on that tab or from Full app report&#39;s own Stage Summary route link. That tab switch can unmount and remount the whole Findings subtree in the same commit as the route landing (base-ui <code>Tabs</code> fully unmounts the inactive panel; see &quot;Render order&quot; above), which two routing paths have to account for: <code>reportWidgetOpen</code> no longer clears a route just because the target <code>WidgetGridItem</code> hasn&#39;t registered yet in this commit (a child <code>WidgetCard</code> reports its own open state before its parent <code>WidgetGridItem</code>&#39;s registration effect runs on a fresh mount, so treating &quot;not registered yet&quot; as &quot;stale&quot; would drop the route before the parent even gets a chance to register it); and a paginated widget that jumps its own page to reveal a routed row (<code>GcPressure.tsx</code>, <code>StageFailed.tsx</code>, <code>TaskFailures.tsx</code>, <code>RetryWaste.tsx</code>) computes that jump during render, not only in a post-commit effect, so the row exists in the DOM in time for this same commit&#39;s anchor lookup instead of one commit later.</p><p>After the target card commits open, its wrapper scrolls with a top margin below the fixed header and its disclosure button receives visible, temporary route focus (using instant scrolling for reduced motion). Requests use a latest-request-wins token (<code>requestTokenRef</code>) that guards every step and revalidates; a stale, missing, unmounted, changed-catalog, or changed-active-file target cancels without redirecting, scrolling, or moving focus. Widget registration probes survive StrictMode double-mounts via an <code>unregistrationCheck</code> that re-validates before clearing.</p><p>This route deliberately does not include Config Audit, detector/parser or threshold changes, Stage Detail routing, or generic badges, tags, chips, or StagePills as controls.</p><h3 id="widget-placement" tabindex="-1">Widget placement <a class="header-anchor" href="#widget-placement" aria-label="Permalink to &quot;Widget placement&quot;">​</a></h3><p>The Plan Explorer (<code>src/view/widgets/PlanExplorer.tsx</code>) is embedded inside a flagged stage&#39;s own row, not as a standalone widget. Neither Task Skew nor Spill embed it any more: both dropped their own <code>PlanExplorer</code> trigger entirely (not merely re-gated behind a toggle), since the same stage&#39;s plan is already reachable, unconditionally, from <code>StageDetailDialog</code>. Shuffle I/O is <code>PlanExplorer</code>&#39;s only remaining call site: it has no row-expansion toggle of its own, so its trigger renders as a direct sibling in the row body.</p><p>Plan detection logic belongs in a <code>packages/core/src/detectors.ts</code> entry (<code>scope:&#39;sql&#39;</code>) consuming <code>planTree</code>. <code>packages/core/src/plan-summary.ts</code> is display-only summarization; do not add detection heuristics there.</p><p><code>StageHeader.tsx</code> (<code>StagePill</code> plus a <code>&lt;span&gt;</code> naming the stage) vs. bare <code>StagePill</code>/<code>StagePillGroup</code> is a deliberate row-density choice, not an inconsistency: <code>StageHeader</code> is for single-stage-per-row detail widgets where naming the stage adds context (<code>SlowHost.tsx</code>, <code>StageSlowness.tsx</code>, <code>Straggler.tsx</code>, <code>SpeculationWaste.tsx</code>, <code>GcPressure.tsx</code>, <code>Skew.tsx</code>, <code>StageShape.tsx</code>, <code>TinyTask.tsx</code>, <code>PartitionSizing.tsx</code>), while bare <code>StagePill</code>/<code>StagePillGroup</code> is for compact, multi-row lists where many stages appear per widget (<code>StageFailed.tsx</code>, <code>TaskFailures.tsx</code>, <code>RetryWaste.tsx</code>, the four split Plan Advisor widgets, <code>Spill.tsx</code>, and <code>StageTable.tsx</code>). <code>ShuffleIO.tsx</code> uses both in different parts of its own row, which is fine: it isn&#39;t a violation of the convention above.</p><p>Stage Summary Table (<code>src/view/widgets/StageTable.tsx</code>) defaults to <strong>problem stages</strong>, with a header toggle to <strong>top 10 by duration</strong>. It renders inside the Full app report tab, not as a standalone board section.</p><h3 id="full-render-sequence" tabindex="-1">Full render sequence <a class="header-anchor" href="#full-render-sequence" aria-label="Permalink to &quot;Full render sequence&quot;">​</a></h3><p>Top to bottom, in <code>Dashboard.tsx</code>&#39;s <code>FilteredBoard</code>:</p><ol><li><code>FindingFilterBar</code> (plus <code>NoMatchBanner</code> when the active filter empties both finding streams).</li><li><code>Scorecard</code>: a three-tile run-info row (Wall-clock, Efficiency, Wastage), rendered once regardless of which tab is active.</li><li>A two-tab <code>Tabs</code> (skipped entirely when the active filter empties both finding streams; <code>NoMatchBanner</code> above already covers that case), tab labels <strong>Findings</strong> and <strong>Full app report</strong>: <ul><li><strong>Findings</strong> (<code>ImpactBoard</code>): a <code>HighestImpactBar</code> callout for the single highest-impact eligible finding, if any; then one <code>&lt;section&gt;</code> per impact band in <code>Critical</code> → <code>Warning</code> → <code>Info</code> order, each rendering nothing when it has neither a recommendation row nor an active widget: a recommendation-rollup <code>Table</code> (one row per eligible-finding type, <code>REGISTRY</code>-mapped, <code>incompleteRun</code> excluded: a direct row for a type with one finding, an expandable, paginated summary row for a type with more than one) followed by a <code>WidgetGrid</code> of that band&#39;s active <code>REGISTRY</code> widgets (ranked by widget order: ascending <code>DETECTORS</code> order, action-region components ahead of reference-region ones); then, below every impact band, a small always-visible grid of just Core Usage by Locality, mounted unconditionally regardless of finding state, and a collapsed &quot;Clean checks&quot; disclosure of <code>CleanCheckRow</code> lines built per detector type (every remaining <code>REGISTRY</code> key with zero findings); or inline &quot;No findings to fix right now.&quot; text when nothing is eligible at all.</li><li><strong>Full app report</strong> (<code>ReferenceSection</code>): WallClock → Timeline → Executor Count Over Time → StageTable → Evidence availability → fixed report-lens tail (ETL Phase Attribution → What-If Executor Scaling → Compute Efficiency → Wasted Core-Hours → Core-Usage Distribution). Structural-only, as above. Fully unmounted while Findings is active (see &quot;Render order&quot; above).</li></ul></li></ol><p>Report lenses carry no impact-band chip and self-hide when their input data is absent. Every widget self-wraps in <code>WidgetCard</code> (<code>src/view/WidgetCard.tsx</code>: an <code>&lt;h3&gt;</code> title, one level below the board&#39;s <code>&lt;h2&gt;</code> section headers) for the document outline, except Scorecard, which renders its own header band.</p><p>Every card is collapsible: the title sits in a <code>CollapsibleTrigger</code> button with a chevron (<code>WidgetCard.tsx</code>), and each widget sets its own <code>defaultCollapsed</code> (most default to <code>true</code>; <code>ImpactBoard</code>&#39;s action-region grid forces <code>defaultCollapsed</code> on every <code>REGISTRY</code> widget instance it mounts, so a marginal finding doesn&#39;t default open). Route navigation (&quot;jump to this finding&quot;) goes through the grid coordinator instead of a fixed <code>tabIndex</code>: <code>WidgetGrid.tsx</code> tracks each card&#39;s disclosure button and bumps <code>openRequestGeneration</code> to force a collapsed card open and focus its trigger when the target finding has no anchored row of its own. How much a card shows beyond that is a board-wide choice, set by the Basic/Advanced density tier in the topbar.</p>',43)])])}const u=o(i,[["render",n]]);export{p as __pageData,u as default};
@@ -0,0 +1 @@
1
+ import{_ as o,o as t,c as d,a5 as a}from"./chunks/framework.DSg0KOwT.js";const p=JSON.parse('{"title":"Widget rendering","description":"","frontmatter":{},"headers":[],"relativePath":"contributor-guide/architecture/widget-rendering.md","filePath":"contributor-guide/architecture/widget-rendering.md"}'),i={name:"contributor-guide/architecture/widget-rendering.md"};function n(c,e,r,s,l,g){return t(),d("div",null,[...e[0]||(e[0]=[a("",43)])])}const u=o(i,[["render",n]]);export{p as __pageData,u as default};