sparklens 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- sparklens-0.1.0/PKG-INFO +496 -0
- sparklens-0.1.0/README.md +482 -0
- sparklens-0.1.0/ai/__init__.py +1 -0
- sparklens-0.1.0/ai/exceptions.py +57 -0
- sparklens-0.1.0/ai/explainer.py +36 -0
- sparklens-0.1.0/ai/models.py +42 -0
- sparklens-0.1.0/ai/prompt_builder.py +181 -0
- sparklens-0.1.0/ai/prompts.py +128 -0
- sparklens-0.1.0/ai/providers/base.py +30 -0
- sparklens-0.1.0/ai/providers/gemini_provider.py +183 -0
- sparklens-0.1.0/ai/providers/retrying_provider.py +67 -0
- sparklens-0.1.0/ai/recommendations.py +417 -0
- sparklens-0.1.0/analyzer/__init__.py +1 -0
- sparklens-0.1.0/analyzer/analyzer.py +341 -0
- sparklens-0.1.0/analyzer/dag_builder.py +61 -0
- sparklens-0.1.0/analyzer/extractor.py +35 -0
- sparklens-0.1.0/analyzer/normalizer.py +40 -0
- sparklens-0.1.0/analyzer/parser.py +368 -0
- sparklens-0.1.0/cli/__init__.py +0 -0
- sparklens-0.1.0/cli/main.py +154 -0
- sparklens-0.1.0/dashboard/app.py +38 -0
- sparklens-0.1.0/dashboard/pages/1_Analyzer.py +121 -0
- sparklens-0.1.0/dashboard/pages/2_Reports.py +5 -0
- sparklens-0.1.0/dashboard/pages/3_DAG.py +101 -0
- sparklens-0.1.0/dashboard/pages/4_Settings.py +5 -0
- sparklens-0.1.0/dashboard/ui_services/analysis_service.py +18 -0
- sparklens-0.1.0/dashboard/utils/renderer.py +206 -0
- sparklens-0.1.0/demo_plans/generate_plans.py +95 -0
- sparklens-0.1.0/evaluation/evaluate_engine.py +1477 -0
- sparklens-0.1.0/examples/generate_report.py +51 -0
- sparklens-0.1.0/models/__init__.py +1 -0
- sparklens-0.1.0/models/analysis_finding.py +23 -0
- sparklens-0.1.0/models/analysis_report.py +53 -0
- sparklens-0.1.0/models/analysis_result.py +43 -0
- sparklens-0.1.0/models/base.py +25 -0
- sparklens-0.1.0/models/cost_estimate.py +47 -0
- sparklens-0.1.0/models/enums.py +29 -0
- sparklens-0.1.0/models/execution_dag.py +146 -0
- sparklens-0.1.0/models/execution_edge.py +18 -0
- sparklens-0.1.0/models/execution_statistics.py +73 -0
- sparklens-0.1.0/models/parsed_plan.py +19 -0
- sparklens-0.1.0/models/physical_operator.py +50 -0
- sparklens-0.1.0/pyproject.toml +29 -0
- sparklens-0.1.0/report/__init__.py +15 -0
- sparklens-0.1.0/report/exporters.py +185 -0
- sparklens-0.1.0/report/generator.py +146 -0
- sparklens-0.1.0/report/html.py +95 -0
- sparklens-0.1.0/report/markdown.py +94 -0
- sparklens-0.1.0/report/models.py +33 -0
- sparklens-0.1.0/rules/__init__.py +1 -0
- sparklens-0.1.0/rules/aggregation_rules.py +184 -0
- sparklens-0.1.0/rules/aqe_rules.py +52 -0
- sparklens-0.1.0/rules/base.py +23 -0
- sparklens-0.1.0/rules/engine.py +97 -0
- sparklens-0.1.0/rules/exchange_rules.py +102 -0
- sparklens-0.1.0/rules/join_rules.py +137 -0
- sparklens-0.1.0/rules/partition_rules.py +74 -0
- sparklens-0.1.0/rules/performance_rules.py +155 -0
- sparklens-0.1.0/rules/pipeline_rules.py +106 -0
- sparklens-0.1.0/rules/quality_rules.py +514 -0
- sparklens-0.1.0/rules/shuffle_rules.py +53 -0
- sparklens-0.1.0/rules/sort_rules.py +49 -0
- sparklens-0.1.0/rules/stage_rules.py +73 -0
- sparklens-0.1.0/sample_plans/generate_plan.py +10 -0
- sparklens-0.1.0/services/pipeline.py +120 -0
- sparklens-0.1.0/setup.cfg +4 -0
- sparklens-0.1.0/sparklens.egg-info/PKG-INFO +496 -0
- sparklens-0.1.0/sparklens.egg-info/SOURCES.txt +101 -0
- sparklens-0.1.0/sparklens.egg-info/dependency_links.txt +1 -0
- sparklens-0.1.0/sparklens.egg-info/entry_points.txt +2 -0
- sparklens-0.1.0/sparklens.egg-info/requires.txt +7 -0
- sparklens-0.1.0/sparklens.egg-info/top_level.txt +16 -0
- sparklens-0.1.0/tests/__init__.py +0 -0
- sparklens-0.1.0/tests/integration/test_ai_pipeline.py +194 -0
- sparklens-0.1.0/tests/integration/test_end_to_end.py +52 -0
- sparklens-0.1.0/tests/integration/test_gemini_provider.py +54 -0
- sparklens-0.1.0/tests/parser/__init__.py +0 -0
- sparklens-0.1.0/tests/parser/test_real_plan.py +32 -0
- sparklens-0.1.0/tests/test_analyzer.py +18 -0
- sparklens-0.1.0/tests/test_cli.py +7 -0
- sparklens-0.1.0/tests/test_dag_builder.py +142 -0
- sparklens-0.1.0/tests/test_extractor.py +30 -0
- sparklens-0.1.0/tests/test_graph_builder.py +104 -0
- sparklens-0.1.0/tests/test_models.py +51 -0
- sparklens-0.1.0/tests/test_node_parser.py +37 -0
- sparklens-0.1.0/tests/test_normalizer.py +35 -0
- sparklens-0.1.0/tests/test_plan_parser.py +39 -0
- sparklens-0.1.0/tests/test_reasoning_regressions.py +352 -0
- sparklens-0.1.0/tests/test_recommendations.py +48 -0
- sparklens-0.1.0/tests/test_report_pipeline.py +125 -0
- sparklens-0.1.0/tests/test_retrying_provider.py +143 -0
- sparklens-0.1.0/tests/test_rule_engine.py +66 -0
- sparklens-0.1.0/tests/test_tree_parser.py +26 -0
- sparklens-0.1.0/tests/test_visualization.py +5 -0
- sparklens-0.1.0/tests/utils.py +26 -0
- sparklens-0.1.0/visualization/__init__.py +1 -0
- sparklens-0.1.0/visualization/base_renderer.py +22 -0
- sparklens-0.1.0/visualization/dag.py +141 -0
- sparklens-0.1.0/visualization/graph_builder.py +78 -0
- sparklens-0.1.0/visualization/interactive_renderer.py +184 -0
- sparklens-0.1.0/visualization/models.py +32 -0
- sparklens-0.1.0/visualization/networkx_renderer.py +169 -0
- sparklens-0.1.0/visualization/styles.py +31 -0
sparklens-0.1.0/PKG-INFO
ADDED
|
@@ -0,0 +1,496 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: sparklens
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Spark Physical Plan Analyzer and Visualizer
|
|
5
|
+
Requires-Python: >=3.12
|
|
6
|
+
Description-Content-Type: text/markdown
|
|
7
|
+
Requires-Dist: altair
|
|
8
|
+
Requires-Dist: matplotlib
|
|
9
|
+
Requires-Dist: networkx
|
|
10
|
+
Requires-Dist: pydot
|
|
11
|
+
Requires-Dist: pyvis
|
|
12
|
+
Requires-Dist: google-genai
|
|
13
|
+
Requires-Dist: python-dotenv
|
|
14
|
+
|
|
15
|
+
# SparkLens
|
|
16
|
+
|
|
17
|
+
## Deterministic Spark Execution Plan Analyzer with AI-Assisted Explanations
|
|
18
|
+
|
|
19
|
+
SparkLens is a tool for analyzing Apache Spark Physical Execution Plans.
|
|
20
|
+
|
|
21
|
+
It transforms raw Spark execution plans into:
|
|
22
|
+
|
|
23
|
+
* Structured execution DAGs
|
|
24
|
+
* Deterministic performance findings
|
|
25
|
+
* Visual execution graphs
|
|
26
|
+
* Markdown and HTML reports
|
|
27
|
+
* AI-generated explanations powered by Gemini
|
|
28
|
+
|
|
29
|
+
SparkLens is designed around a simple principle:
|
|
30
|
+
|
|
31
|
+
> AI should explain analysis, not replace analysis.
|
|
32
|
+
|
|
33
|
+
Unlike tools that send raw execution plans directly to an LLM, SparkLens first performs deterministic rule-based analysis and then uses AI to explain the findings.
|
|
34
|
+
|
|
35
|
+
---
|
|
36
|
+
|
|
37
|
+
# Why SparkLens?
|
|
38
|
+
|
|
39
|
+
Spark execution plans are powerful but difficult to interpret.
|
|
40
|
+
|
|
41
|
+
Even moderately complex plans may contain:
|
|
42
|
+
|
|
43
|
+
* Broadcast joins
|
|
44
|
+
* Shuffle exchanges
|
|
45
|
+
* Multi-stage aggregations
|
|
46
|
+
* Sorting operations
|
|
47
|
+
* Adaptive Query Execution (AQE)
|
|
48
|
+
* Partitioning strategies
|
|
49
|
+
* Stage boundaries
|
|
50
|
+
|
|
51
|
+
Understanding:
|
|
52
|
+
|
|
53
|
+
* Why a query is slow
|
|
54
|
+
* Where data is shuffled
|
|
55
|
+
* Which join strategy Spark selected
|
|
56
|
+
* Where parallelism is lost
|
|
57
|
+
* How execution stages are formed
|
|
58
|
+
|
|
59
|
+
often requires significant Spark expertise.
|
|
60
|
+
|
|
61
|
+
SparkLens automates this process and presents the results in a structured, explainable format.
|
|
62
|
+
|
|
63
|
+
---
|
|
64
|
+
|
|
65
|
+
# Core Features
|
|
66
|
+
|
|
67
|
+
## Execution Plan Parsing
|
|
68
|
+
|
|
69
|
+
Parse Spark Physical Execution Plans into structured operator representations.
|
|
70
|
+
|
|
71
|
+
---
|
|
72
|
+
|
|
73
|
+
## Execution DAG Construction
|
|
74
|
+
|
|
75
|
+
Convert execution plans into a directed acyclic graph representing execution flow.
|
|
76
|
+
|
|
77
|
+
---
|
|
78
|
+
|
|
79
|
+
## Deterministic Rule Engine
|
|
80
|
+
|
|
81
|
+
Detects Spark performance patterns and execution characteristics including:
|
|
82
|
+
|
|
83
|
+
* Broadcast Hash Joins
|
|
84
|
+
* Shuffle Operations
|
|
85
|
+
* Stage Boundaries
|
|
86
|
+
* Single Partition Bottlenecks
|
|
87
|
+
* Aggregation Pipelines
|
|
88
|
+
* Sorting Operations
|
|
89
|
+
* AQE Usage
|
|
90
|
+
* Partitioning Strategies
|
|
91
|
+
* Performance Anti-Patterns
|
|
92
|
+
|
|
93
|
+
---
|
|
94
|
+
|
|
95
|
+
## AI-Assisted Explanations
|
|
96
|
+
|
|
97
|
+
Generate:
|
|
98
|
+
|
|
99
|
+
* Executive Summaries
|
|
100
|
+
* Execution Walkthroughs
|
|
101
|
+
* Performance Impact Analysis
|
|
102
|
+
* Optimization Recommendations
|
|
103
|
+
|
|
104
|
+
using Google Gemini.
|
|
105
|
+
|
|
106
|
+
The AI layer receives structured findings generated by SparkLens and is instructed to explain those findings rather than independently analyze the plan.
|
|
107
|
+
|
|
108
|
+
---
|
|
109
|
+
|
|
110
|
+
## Visualization
|
|
111
|
+
|
|
112
|
+
Generate execution DAG visualizations as PNG images.
|
|
113
|
+
|
|
114
|
+
Visualization highlights operators associated with analysis findings.
|
|
115
|
+
|
|
116
|
+
---
|
|
117
|
+
|
|
118
|
+
## Reporting
|
|
119
|
+
|
|
120
|
+
Generate:
|
|
121
|
+
|
|
122
|
+
* Markdown Reports
|
|
123
|
+
* HTML Reports
|
|
124
|
+
|
|
125
|
+
containing both deterministic analysis and AI-generated explanations.
|
|
126
|
+
|
|
127
|
+
---
|
|
128
|
+
|
|
129
|
+
# Analysis Philosophy
|
|
130
|
+
|
|
131
|
+
SparkLens follows a deterministic-first architecture.
|
|
132
|
+
|
|
133
|
+
Instead of sending a raw execution plan directly to an LLM:
|
|
134
|
+
|
|
135
|
+
```text
|
|
136
|
+
Spark Plan
|
|
137
|
+
↓
|
|
138
|
+
LLM
|
|
139
|
+
↓
|
|
140
|
+
Explanation
|
|
141
|
+
```
|
|
142
|
+
|
|
143
|
+
SparkLens performs structured analysis first:
|
|
144
|
+
|
|
145
|
+
```text
|
|
146
|
+
Spark Plan
|
|
147
|
+
↓
|
|
148
|
+
Parser
|
|
149
|
+
↓
|
|
150
|
+
Execution DAG
|
|
151
|
+
↓
|
|
152
|
+
Rule Engine
|
|
153
|
+
↓
|
|
154
|
+
Analysis Report
|
|
155
|
+
↓
|
|
156
|
+
Recommendation Engine
|
|
157
|
+
↓
|
|
158
|
+
AI Explanation
|
|
159
|
+
```
|
|
160
|
+
|
|
161
|
+
Benefits:
|
|
162
|
+
|
|
163
|
+
* Reduced hallucinations
|
|
164
|
+
* Consistent findings
|
|
165
|
+
* Explainable analysis
|
|
166
|
+
* Reproducible results
|
|
167
|
+
* Better separation between analysis and explanation
|
|
168
|
+
|
|
169
|
+
---
|
|
170
|
+
|
|
171
|
+
# Evaluation
|
|
172
|
+
|
|
173
|
+
SparkLens findings are scored against a fixed ground-truth dataset. The evaluation runs fully offline (no API calls, no model downloads) and is deterministic: the same inputs always give the same numbers.
|
|
174
|
+
|
|
175
|
+
## What is measured
|
|
176
|
+
|
|
177
|
+
* `evaluation/dataset_expected.csv` — 50 real Spark physical plans (49 rendered with Spark 3.5 `explain("formatted")`, 1 Databricks Photon plan), each with the analysis a correct analyzer should produce, written as plain sentences (316 atomic claims)
|
|
178
|
+
* `evaluation/dataset_predicted.csv` — what SparkLens reports for each plan (`sparklens analyze <plan> --eval`: rule engine only, AI disabled)
|
|
179
|
+
|
|
180
|
+
Both paragraphs are split into atomic claims and compared claim by claim:
|
|
181
|
+
|
|
182
|
+
* **Precision** — share of SparkLens claims backed by the ground truth (false alarms lower it)
|
|
183
|
+
* **Recall** — share of ground-truth claims that SparkLens states (misses lower it)
|
|
184
|
+
* **F1** — balance of the two
|
|
185
|
+
* **Final score (F0.5)** — weights precision above recall, because false alarms erode developer trust
|
|
186
|
+
|
|
187
|
+
## How claims are matched
|
|
188
|
+
|
|
189
|
+
```text
|
|
190
|
+
SparkLens output / ground truth
|
|
191
|
+
↓
|
|
192
|
+
Atomic claims (sentences split at "so / because / but"; pronouns resolved)
|
|
193
|
+
↓
|
|
194
|
+
Spark concept normalisation ("SMJ" = "SortMergeJoin" = "sort-merge join")
|
|
195
|
+
↓
|
|
196
|
+
TF-IDF vectors (Spark-concept channel + wording channel)
|
|
197
|
+
↓
|
|
198
|
+
Directional matching (precision: backed by the ground truth; recall: stated by SparkLens)
|
|
199
|
+
↓
|
|
200
|
+
Consistency rules (stage counts, numbers, polarity, exclusive alternatives, advice vs observation)
|
|
201
|
+
```
|
|
202
|
+
|
|
203
|
+
## Validating the evaluator
|
|
204
|
+
|
|
205
|
+
`python evaluation/evaluate_engine.py validate` measures the evaluator itself on 1,377 test cases whose correct answer is known by construction:
|
|
206
|
+
|
|
207
|
+
| Test type | Example | Correct verdict |
|
|
208
|
+
|---|---|---|
|
|
209
|
+
| Paraphrase | the same claim in different words | match |
|
|
210
|
+
| Number change | "3 stages" → "4 stages" | no match |
|
|
211
|
+
| Polarity flip | enabled ↔ disabled, required ↔ avoidable | no match |
|
|
212
|
+
| Concept swap | sort-merge ↔ broadcast hash join, hash ↔ range partitioning | no match |
|
|
213
|
+
| Cross-plan claim | a claim about an operator the plan text does not contain | no match |
|
|
214
|
+
| Role swap | same words, roles swapped | no match |
|
|
215
|
+
|
|
216
|
+
It also checks invariance (reordering, duplicating or re-spacing findings never changes a score) and bounds (the ground truth as prediction scores 1.0, an empty answer 0.0, another plan's output far lower).
|
|
217
|
+
|
|
218
|
+
## Current baseline
|
|
219
|
+
|
|
220
|
+
Rule engine `c7daf3c`, 4 Oct 2026:
|
|
221
|
+
|
|
222
|
+
| Metric | Value |
|
|
223
|
+
|---|---|
|
|
224
|
+
| SparkLens precision / recall / F1 | 0.637 / 0.478 / 0.546 |
|
|
225
|
+
| SparkLens final score (F0.5) | **0.597** |
|
|
226
|
+
| Evaluator accuracy, 1,377 known-answer cases | **92.2%** (dev split 94.4%, test split 90.2%) |
|
|
227
|
+
| Evaluator false accepts / false rejects | 11.5% / 3.5% |
|
|
228
|
+
| Invariance checks / bounds | 200 of 200 / pass |
|
|
229
|
+
|
|
230
|
+
Known limitation: claims that use the same words with swapped roles ("orders is broadcast, customers is shuffled" vs the reverse) are not yet told apart.
|
|
231
|
+
|
|
232
|
+
## Running the evaluation
|
|
233
|
+
|
|
234
|
+
```bash
|
|
235
|
+
pip install -r evaluation/requirements-eval.txt
|
|
236
|
+
|
|
237
|
+
python evaluation/evaluate_engine.py selftest # unit checks of the matcher and metrics
|
|
238
|
+
python evaluation/evaluate_engine.py validate # evaluator accuracy on known-answer cases
|
|
239
|
+
python evaluation/evaluate_engine.py predict # SparkLens (--eval) on all 50 plans -> dataset_predicted.csv
|
|
240
|
+
python evaluation/evaluate_engine.py evaluate # precision / recall / F1 / final score, change vs baseline
|
|
241
|
+
python evaluation/evaluate_engine.py baseline # freeze the current result as eval_baseline.json
|
|
242
|
+
```
|
|
243
|
+
|
|
244
|
+
`predict` runs the SparkLens CLI, so SparkLens itself must be installed (`pip install -e .`). After a rule change, run `predict --force` and then `evaluate`; improved and regressed plans are listed.
|
|
245
|
+
|
|
246
|
+
To fill `dataset_predicted.csv` by hand, run `python evaluation/evaluate_engine.py export` and paste the output of `sparklens analyze evaluation/eval_plans/<id>.txt --eval` into each row.
|
|
247
|
+
|
|
248
|
+
All evaluation files live in `evaluation/`; run the commands from the repository root.
|
|
249
|
+
|
|
250
|
+
| File | Purpose |
|
|
251
|
+
|---|---|
|
|
252
|
+
| `evaluate_engine.py` | the evaluator (commands above) |
|
|
253
|
+
| `dataset_expected.csv` | ground truth |
|
|
254
|
+
| `dataset_predicted.csv` | SparkLens output being scored |
|
|
255
|
+
| `eval_validation.csv` | paraphrases and role swaps used by `validate` |
|
|
256
|
+
| `eval_baseline.json` | frozen reference result, including evaluator accuracy |
|
|
257
|
+
| `eval_review.csv` | per-claim verdicts of the last run (generated, not tracked) |
|
|
258
|
+
|
|
259
|
+
---
|
|
260
|
+
|
|
261
|
+
# Architecture
|
|
262
|
+
|
|
263
|
+
```text
|
|
264
|
+
Spark Physical Plan
|
|
265
|
+
│
|
|
266
|
+
▼
|
|
267
|
+
Plan Normalizer
|
|
268
|
+
│
|
|
269
|
+
▼
|
|
270
|
+
Plan Parser
|
|
271
|
+
│
|
|
272
|
+
▼
|
|
273
|
+
DAG Builder
|
|
274
|
+
│
|
|
275
|
+
▼
|
|
276
|
+
Execution DAG
|
|
277
|
+
│
|
|
278
|
+
▼
|
|
279
|
+
Rule Engine
|
|
280
|
+
│
|
|
281
|
+
▼
|
|
282
|
+
Analysis Report
|
|
283
|
+
│
|
|
284
|
+
▼
|
|
285
|
+
Recommendation Engine
|
|
286
|
+
│
|
|
287
|
+
▼
|
|
288
|
+
Prompt Builder
|
|
289
|
+
│
|
|
290
|
+
▼
|
|
291
|
+
Gemini AI
|
|
292
|
+
│
|
|
293
|
+
▼
|
|
294
|
+
AI Explanation
|
|
295
|
+
|
|
296
|
+
|
|
297
|
+
Execution DAG
|
|
298
|
+
│
|
|
299
|
+
▼
|
|
300
|
+
Visualization
|
|
301
|
+
│
|
|
302
|
+
▼
|
|
303
|
+
dag.png
|
|
304
|
+
|
|
305
|
+
|
|
306
|
+
Analysis Report
|
|
307
|
+
│
|
|
308
|
+
▼
|
|
309
|
+
Report Generator
|
|
310
|
+
│
|
|
311
|
+
▼
|
|
312
|
+
report.md
|
|
313
|
+
report.html
|
|
314
|
+
```
|
|
315
|
+
|
|
316
|
+
---
|
|
317
|
+
|
|
318
|
+
# Installation
|
|
319
|
+
|
|
320
|
+
Clone the repository:
|
|
321
|
+
|
|
322
|
+
```bash
|
|
323
|
+
git clone <repository-url>
|
|
324
|
+
cd SparkLens
|
|
325
|
+
```
|
|
326
|
+
|
|
327
|
+
Install SparkLens locally:
|
|
328
|
+
|
|
329
|
+
```bash
|
|
330
|
+
pip install -e .
|
|
331
|
+
```
|
|
332
|
+
|
|
333
|
+
---
|
|
334
|
+
|
|
335
|
+
# AI Setup
|
|
336
|
+
|
|
337
|
+
Create a `.env` file in the project root:
|
|
338
|
+
|
|
339
|
+
```env
|
|
340
|
+
GEMINI_API_KEY=your_api_key_here
|
|
341
|
+
```
|
|
342
|
+
|
|
343
|
+
SparkLens uses Google Gemini for natural-language explanations.
|
|
344
|
+
|
|
345
|
+
---
|
|
346
|
+
|
|
347
|
+
# Usage
|
|
348
|
+
|
|
349
|
+
## Deterministic Analysis
|
|
350
|
+
|
|
351
|
+
```bash
|
|
352
|
+
sparklens analyze sample_plans/broadcast_join.txt
|
|
353
|
+
```
|
|
354
|
+
|
|
355
|
+
Outputs:
|
|
356
|
+
|
|
357
|
+
```text
|
|
358
|
+
dag.png
|
|
359
|
+
report.md
|
|
360
|
+
report.html
|
|
361
|
+
```
|
|
362
|
+
|
|
363
|
+
---
|
|
364
|
+
|
|
365
|
+
## AI-Enhanced Analysis
|
|
366
|
+
|
|
367
|
+
```bash
|
|
368
|
+
sparklens analyze sample_plans/broadcast_join.txt --ai
|
|
369
|
+
```
|
|
370
|
+
|
|
371
|
+
Outputs:
|
|
372
|
+
|
|
373
|
+
```text
|
|
374
|
+
dag.png
|
|
375
|
+
report.md
|
|
376
|
+
report.html
|
|
377
|
+
```
|
|
378
|
+
|
|
379
|
+
with AI-generated:
|
|
380
|
+
|
|
381
|
+
* Executive Summary
|
|
382
|
+
* Execution Explanation
|
|
383
|
+
* Recommendations
|
|
384
|
+
|
|
385
|
+
## Evaluation Output
|
|
386
|
+
|
|
387
|
+
```bash
|
|
388
|
+
sparklens analyze evaluation/eval_plans/<id>.txt --eval
|
|
389
|
+
```
|
|
390
|
+
|
|
391
|
+
This prints the deterministic rule-engine findings as a single line without making AI API calls.
|
|
392
|
+
|
|
393
|
+
---
|
|
394
|
+
|
|
395
|
+
# Example Findings
|
|
396
|
+
|
|
397
|
+
```text
|
|
398
|
+
Single Partition Bottleneck
|
|
399
|
+
Severity: High
|
|
400
|
+
|
|
401
|
+
The execution plan contains a SinglePartition exchange,
|
|
402
|
+
which may reduce parallelism by processing data in a
|
|
403
|
+
single partition.
|
|
404
|
+
```
|
|
405
|
+
|
|
406
|
+
---
|
|
407
|
+
|
|
408
|
+
# Example AI Summary
|
|
409
|
+
|
|
410
|
+
```text
|
|
411
|
+
The execution plan uses a Broadcast Hash Join and Adaptive
|
|
412
|
+
Query Execution but contains a critical Single Partition
|
|
413
|
+
bottleneck that limits parallelism and may impact scalability.
|
|
414
|
+
```
|
|
415
|
+
|
|
416
|
+
---
|
|
417
|
+
|
|
418
|
+
# Project Structure
|
|
419
|
+
|
|
420
|
+
```text
|
|
421
|
+
SparkLens/
|
|
422
|
+
│
|
|
423
|
+
├── ai/
|
|
424
|
+
├── analyzer/
|
|
425
|
+
├── cli/
|
|
426
|
+
├── evaluation/
|
|
427
|
+
├── models/
|
|
428
|
+
├── report/
|
|
429
|
+
├── rules/
|
|
430
|
+
├── sample_plans/
|
|
431
|
+
├── tests/
|
|
432
|
+
├── visualization/
|
|
433
|
+
│
|
|
434
|
+
├── pyproject.toml
|
|
435
|
+
├── README.md
|
|
436
|
+
└── .env
|
|
437
|
+
```
|
|
438
|
+
|
|
439
|
+
---
|
|
440
|
+
|
|
441
|
+
# Running Tests
|
|
442
|
+
|
|
443
|
+
Execute the full test suite:
|
|
444
|
+
|
|
445
|
+
```bash
|
|
446
|
+
pytest
|
|
447
|
+
```
|
|
448
|
+
|
|
449
|
+
---
|
|
450
|
+
|
|
451
|
+
# Current Capabilities (v0.1.0)
|
|
452
|
+
|
|
453
|
+
* Spark Physical Plan Parsing
|
|
454
|
+
* Execution DAG Construction
|
|
455
|
+
* Rule-Based Analysis
|
|
456
|
+
* Recommendation Generation
|
|
457
|
+
* AI-Assisted Explanations
|
|
458
|
+
* DAG Visualization
|
|
459
|
+
* Markdown Reports
|
|
460
|
+
* HTML Reports
|
|
461
|
+
* CLI Interface
|
|
462
|
+
* Local Package Installation
|
|
463
|
+
|
|
464
|
+
---
|
|
465
|
+
|
|
466
|
+
# Roadmap
|
|
467
|
+
|
|
468
|
+
## v0.2
|
|
469
|
+
|
|
470
|
+
* Databricks Integration
|
|
471
|
+
* Rich HTML Dashboard
|
|
472
|
+
* Embedded DAG Visualizations
|
|
473
|
+
* Additional Rule Packs
|
|
474
|
+
* Enhanced Recommendation Engine
|
|
475
|
+
|
|
476
|
+
## v0.3
|
|
477
|
+
|
|
478
|
+
* Runtime Metrics Analysis
|
|
479
|
+
* Spark UI Integration
|
|
480
|
+
* Query Plan Comparison
|
|
481
|
+
* Execution Replay Visualizations
|
|
482
|
+
* Cost-Based Optimization Insights
|
|
483
|
+
|
|
484
|
+
---
|
|
485
|
+
|
|
486
|
+
# Version
|
|
487
|
+
|
|
488
|
+
Current Release:
|
|
489
|
+
|
|
490
|
+
```text
|
|
491
|
+
v0.1.0
|
|
492
|
+
```
|
|
493
|
+
|
|
494
|
+
---
|
|
495
|
+
|
|
496
|
+
Built to make Spark execution plans understandable, explainable, and actionable.
|