sparklens 0.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (103) hide show
  1. sparklens-0.1.0/PKG-INFO +496 -0
  2. sparklens-0.1.0/README.md +482 -0
  3. sparklens-0.1.0/ai/__init__.py +1 -0
  4. sparklens-0.1.0/ai/exceptions.py +57 -0
  5. sparklens-0.1.0/ai/explainer.py +36 -0
  6. sparklens-0.1.0/ai/models.py +42 -0
  7. sparklens-0.1.0/ai/prompt_builder.py +181 -0
  8. sparklens-0.1.0/ai/prompts.py +128 -0
  9. sparklens-0.1.0/ai/providers/base.py +30 -0
  10. sparklens-0.1.0/ai/providers/gemini_provider.py +183 -0
  11. sparklens-0.1.0/ai/providers/retrying_provider.py +67 -0
  12. sparklens-0.1.0/ai/recommendations.py +417 -0
  13. sparklens-0.1.0/analyzer/__init__.py +1 -0
  14. sparklens-0.1.0/analyzer/analyzer.py +341 -0
  15. sparklens-0.1.0/analyzer/dag_builder.py +61 -0
  16. sparklens-0.1.0/analyzer/extractor.py +35 -0
  17. sparklens-0.1.0/analyzer/normalizer.py +40 -0
  18. sparklens-0.1.0/analyzer/parser.py +368 -0
  19. sparklens-0.1.0/cli/__init__.py +0 -0
  20. sparklens-0.1.0/cli/main.py +154 -0
  21. sparklens-0.1.0/dashboard/app.py +38 -0
  22. sparklens-0.1.0/dashboard/pages/1_Analyzer.py +121 -0
  23. sparklens-0.1.0/dashboard/pages/2_Reports.py +5 -0
  24. sparklens-0.1.0/dashboard/pages/3_DAG.py +101 -0
  25. sparklens-0.1.0/dashboard/pages/4_Settings.py +5 -0
  26. sparklens-0.1.0/dashboard/ui_services/analysis_service.py +18 -0
  27. sparklens-0.1.0/dashboard/utils/renderer.py +206 -0
  28. sparklens-0.1.0/demo_plans/generate_plans.py +95 -0
  29. sparklens-0.1.0/evaluation/evaluate_engine.py +1477 -0
  30. sparklens-0.1.0/examples/generate_report.py +51 -0
  31. sparklens-0.1.0/models/__init__.py +1 -0
  32. sparklens-0.1.0/models/analysis_finding.py +23 -0
  33. sparklens-0.1.0/models/analysis_report.py +53 -0
  34. sparklens-0.1.0/models/analysis_result.py +43 -0
  35. sparklens-0.1.0/models/base.py +25 -0
  36. sparklens-0.1.0/models/cost_estimate.py +47 -0
  37. sparklens-0.1.0/models/enums.py +29 -0
  38. sparklens-0.1.0/models/execution_dag.py +146 -0
  39. sparklens-0.1.0/models/execution_edge.py +18 -0
  40. sparklens-0.1.0/models/execution_statistics.py +73 -0
  41. sparklens-0.1.0/models/parsed_plan.py +19 -0
  42. sparklens-0.1.0/models/physical_operator.py +50 -0
  43. sparklens-0.1.0/pyproject.toml +29 -0
  44. sparklens-0.1.0/report/__init__.py +15 -0
  45. sparklens-0.1.0/report/exporters.py +185 -0
  46. sparklens-0.1.0/report/generator.py +146 -0
  47. sparklens-0.1.0/report/html.py +95 -0
  48. sparklens-0.1.0/report/markdown.py +94 -0
  49. sparklens-0.1.0/report/models.py +33 -0
  50. sparklens-0.1.0/rules/__init__.py +1 -0
  51. sparklens-0.1.0/rules/aggregation_rules.py +184 -0
  52. sparklens-0.1.0/rules/aqe_rules.py +52 -0
  53. sparklens-0.1.0/rules/base.py +23 -0
  54. sparklens-0.1.0/rules/engine.py +97 -0
  55. sparklens-0.1.0/rules/exchange_rules.py +102 -0
  56. sparklens-0.1.0/rules/join_rules.py +137 -0
  57. sparklens-0.1.0/rules/partition_rules.py +74 -0
  58. sparklens-0.1.0/rules/performance_rules.py +155 -0
  59. sparklens-0.1.0/rules/pipeline_rules.py +106 -0
  60. sparklens-0.1.0/rules/quality_rules.py +514 -0
  61. sparklens-0.1.0/rules/shuffle_rules.py +53 -0
  62. sparklens-0.1.0/rules/sort_rules.py +49 -0
  63. sparklens-0.1.0/rules/stage_rules.py +73 -0
  64. sparklens-0.1.0/sample_plans/generate_plan.py +10 -0
  65. sparklens-0.1.0/services/pipeline.py +120 -0
  66. sparklens-0.1.0/setup.cfg +4 -0
  67. sparklens-0.1.0/sparklens.egg-info/PKG-INFO +496 -0
  68. sparklens-0.1.0/sparklens.egg-info/SOURCES.txt +101 -0
  69. sparklens-0.1.0/sparklens.egg-info/dependency_links.txt +1 -0
  70. sparklens-0.1.0/sparklens.egg-info/entry_points.txt +2 -0
  71. sparklens-0.1.0/sparklens.egg-info/requires.txt +7 -0
  72. sparklens-0.1.0/sparklens.egg-info/top_level.txt +16 -0
  73. sparklens-0.1.0/tests/__init__.py +0 -0
  74. sparklens-0.1.0/tests/integration/test_ai_pipeline.py +194 -0
  75. sparklens-0.1.0/tests/integration/test_end_to_end.py +52 -0
  76. sparklens-0.1.0/tests/integration/test_gemini_provider.py +54 -0
  77. sparklens-0.1.0/tests/parser/__init__.py +0 -0
  78. sparklens-0.1.0/tests/parser/test_real_plan.py +32 -0
  79. sparklens-0.1.0/tests/test_analyzer.py +18 -0
  80. sparklens-0.1.0/tests/test_cli.py +7 -0
  81. sparklens-0.1.0/tests/test_dag_builder.py +142 -0
  82. sparklens-0.1.0/tests/test_extractor.py +30 -0
  83. sparklens-0.1.0/tests/test_graph_builder.py +104 -0
  84. sparklens-0.1.0/tests/test_models.py +51 -0
  85. sparklens-0.1.0/tests/test_node_parser.py +37 -0
  86. sparklens-0.1.0/tests/test_normalizer.py +35 -0
  87. sparklens-0.1.0/tests/test_plan_parser.py +39 -0
  88. sparklens-0.1.0/tests/test_reasoning_regressions.py +352 -0
  89. sparklens-0.1.0/tests/test_recommendations.py +48 -0
  90. sparklens-0.1.0/tests/test_report_pipeline.py +125 -0
  91. sparklens-0.1.0/tests/test_retrying_provider.py +143 -0
  92. sparklens-0.1.0/tests/test_rule_engine.py +66 -0
  93. sparklens-0.1.0/tests/test_tree_parser.py +26 -0
  94. sparklens-0.1.0/tests/test_visualization.py +5 -0
  95. sparklens-0.1.0/tests/utils.py +26 -0
  96. sparklens-0.1.0/visualization/__init__.py +1 -0
  97. sparklens-0.1.0/visualization/base_renderer.py +22 -0
  98. sparklens-0.1.0/visualization/dag.py +141 -0
  99. sparklens-0.1.0/visualization/graph_builder.py +78 -0
  100. sparklens-0.1.0/visualization/interactive_renderer.py +184 -0
  101. sparklens-0.1.0/visualization/models.py +32 -0
  102. sparklens-0.1.0/visualization/networkx_renderer.py +169 -0
  103. sparklens-0.1.0/visualization/styles.py +31 -0
@@ -0,0 +1,496 @@
1
+ Metadata-Version: 2.4
2
+ Name: sparklens
3
+ Version: 0.1.0
4
+ Summary: Spark Physical Plan Analyzer and Visualizer
5
+ Requires-Python: >=3.12
6
+ Description-Content-Type: text/markdown
7
+ Requires-Dist: altair
8
+ Requires-Dist: matplotlib
9
+ Requires-Dist: networkx
10
+ Requires-Dist: pydot
11
+ Requires-Dist: pyvis
12
+ Requires-Dist: google-genai
13
+ Requires-Dist: python-dotenv
14
+
15
+ # SparkLens
16
+
17
+ ## Deterministic Spark Execution Plan Analyzer with AI-Assisted Explanations
18
+
19
+ SparkLens is a tool for analyzing Apache Spark Physical Execution Plans.
20
+
21
+ It transforms raw Spark execution plans into:
22
+
23
+ * Structured execution DAGs
24
+ * Deterministic performance findings
25
+ * Visual execution graphs
26
+ * Markdown and HTML reports
27
+ * AI-generated explanations powered by Gemini
28
+
29
+ SparkLens is designed around a simple principle:
30
+
31
+ > AI should explain analysis, not replace analysis.
32
+
33
+ Unlike tools that send raw execution plans directly to an LLM, SparkLens first performs deterministic rule-based analysis and then uses AI to explain the findings.
34
+
35
+ ---
36
+
37
+ # Why SparkLens?
38
+
39
+ Spark execution plans are powerful but difficult to interpret.
40
+
41
+ Even moderately complex plans may contain:
42
+
43
+ * Broadcast joins
44
+ * Shuffle exchanges
45
+ * Multi-stage aggregations
46
+ * Sorting operations
47
+ * Adaptive Query Execution (AQE)
48
+ * Partitioning strategies
49
+ * Stage boundaries
50
+
51
+ Understanding:
52
+
53
+ * Why a query is slow
54
+ * Where data is shuffled
55
+ * Which join strategy Spark selected
56
+ * Where parallelism is lost
57
+ * How execution stages are formed
58
+
59
+ often requires significant Spark expertise.
60
+
61
+ SparkLens automates this process and presents the results in a structured, explainable format.
62
+
63
+ ---
64
+
65
+ # Core Features
66
+
67
+ ## Execution Plan Parsing
68
+
69
+ Parse Spark Physical Execution Plans into structured operator representations.
70
+
71
+ ---
72
+
73
+ ## Execution DAG Construction
74
+
75
+ Convert execution plans into a directed acyclic graph representing execution flow.
76
+
77
+ ---
78
+
79
+ ## Deterministic Rule Engine
80
+
81
+ Detects Spark performance patterns and execution characteristics including:
82
+
83
+ * Broadcast Hash Joins
84
+ * Shuffle Operations
85
+ * Stage Boundaries
86
+ * Single Partition Bottlenecks
87
+ * Aggregation Pipelines
88
+ * Sorting Operations
89
+ * AQE Usage
90
+ * Partitioning Strategies
91
+ * Performance Anti-Patterns
92
+
93
+ ---
94
+
95
+ ## AI-Assisted Explanations
96
+
97
+ Generate:
98
+
99
+ * Executive Summaries
100
+ * Execution Walkthroughs
101
+ * Performance Impact Analysis
102
+ * Optimization Recommendations
103
+
104
+ using Google Gemini.
105
+
106
+ The AI layer receives structured findings generated by SparkLens and is instructed to explain those findings rather than independently analyze the plan.
107
+
108
+ ---
109
+
110
+ ## Visualization
111
+
112
+ Generate execution DAG visualizations as PNG images.
113
+
114
+ Visualization highlights operators associated with analysis findings.
115
+
116
+ ---
117
+
118
+ ## Reporting
119
+
120
+ Generate:
121
+
122
+ * Markdown Reports
123
+ * HTML Reports
124
+
125
+ containing both deterministic analysis and AI-generated explanations.
126
+
127
+ ---
128
+
129
+ # Analysis Philosophy
130
+
131
+ SparkLens follows a deterministic-first architecture.
132
+
133
+ Instead of sending a raw execution plan directly to an LLM:
134
+
135
+ ```text
136
+ Spark Plan
137
+ ↓
138
+ LLM
139
+ ↓
140
+ Explanation
141
+ ```
142
+
143
+ SparkLens performs structured analysis first:
144
+
145
+ ```text
146
+ Spark Plan
147
+ ↓
148
+ Parser
149
+ ↓
150
+ Execution DAG
151
+ ↓
152
+ Rule Engine
153
+ ↓
154
+ Analysis Report
155
+ ↓
156
+ Recommendation Engine
157
+ ↓
158
+ AI Explanation
159
+ ```
160
+
161
+ Benefits:
162
+
163
+ * Reduced hallucinations
164
+ * Consistent findings
165
+ * Explainable analysis
166
+ * Reproducible results
167
+ * Better separation between analysis and explanation
168
+
169
+ ---
170
+
171
+ # Evaluation
172
+
173
+ SparkLens findings are scored against a fixed ground-truth dataset. The evaluation runs fully offline (no API calls, no model downloads) and is deterministic: the same inputs always give the same numbers.
174
+
175
+ ## What is measured
176
+
177
+ * `evaluation/dataset_expected.csv` — 50 real Spark physical plans (49 rendered with Spark 3.5 `explain("formatted")`, 1 Databricks Photon plan), each with the analysis a correct analyzer should produce, written as plain sentences (316 atomic claims)
178
+ * `evaluation/dataset_predicted.csv` — what SparkLens reports for each plan (`sparklens analyze <plan> --eval`: rule engine only, AI disabled)
179
+
180
+ Both paragraphs are split into atomic claims and compared claim by claim:
181
+
182
+ * **Precision** — share of SparkLens claims backed by the ground truth (false alarms lower it)
183
+ * **Recall** — share of ground-truth claims that SparkLens states (misses lower it)
184
+ * **F1** — balance of the two
185
+ * **Final score (F0.5)** — weights precision above recall, because false alarms erode developer trust
186
+
187
+ ## How claims are matched
188
+
189
+ ```text
190
+ SparkLens output / ground truth
191
+ ↓
192
+ Atomic claims (sentences split at "so / because / but"; pronouns resolved)
193
+ ↓
194
+ Spark concept normalisation ("SMJ" = "SortMergeJoin" = "sort-merge join")
195
+ ↓
196
+ TF-IDF vectors (Spark-concept channel + wording channel)
197
+ ↓
198
+ Directional matching (precision: backed by the ground truth; recall: stated by SparkLens)
199
+ ↓
200
+ Consistency rules (stage counts, numbers, polarity, exclusive alternatives, advice vs observation)
201
+ ```
202
+
203
+ ## Validating the evaluator
204
+
205
+ `python evaluation/evaluate_engine.py validate` measures the evaluator itself on 1,377 test cases whose correct answer is known by construction:
206
+
207
+ | Test type | Example | Correct verdict |
208
+ |---|---|---|
209
+ | Paraphrase | the same claim in different words | match |
210
+ | Number change | "3 stages" → "4 stages" | no match |
211
+ | Polarity flip | enabled ↔ disabled, required ↔ avoidable | no match |
212
+ | Concept swap | sort-merge ↔ broadcast hash join, hash ↔ range partitioning | no match |
213
+ | Cross-plan claim | a claim about an operator the plan text does not contain | no match |
214
+ | Role swap | same words, roles swapped | no match |
215
+
216
+ It also checks invariance (reordering, duplicating or re-spacing findings never changes a score) and bounds (the ground truth as prediction scores 1.0, an empty answer 0.0, another plan's output far lower).
217
+
218
+ ## Current baseline
219
+
220
+ Rule engine `c7daf3c`, 4 Oct 2026:
221
+
222
+ | Metric | Value |
223
+ |---|---|
224
+ | SparkLens precision / recall / F1 | 0.637 / 0.478 / 0.546 |
225
+ | SparkLens final score (F0.5) | **0.597** |
226
+ | Evaluator accuracy, 1,377 known-answer cases | **92.2%** (dev split 94.4%, test split 90.2%) |
227
+ | Evaluator false accepts / false rejects | 11.5% / 3.5% |
228
+ | Invariance checks / bounds | 200 of 200 / pass |
229
+
230
+ Known limitation: claims that use the same words with swapped roles ("orders is broadcast, customers is shuffled" vs the reverse) are not yet told apart.
231
+
232
+ ## Running the evaluation
233
+
234
+ ```bash
235
+ pip install -r evaluation/requirements-eval.txt
236
+
237
+ python evaluation/evaluate_engine.py selftest # unit checks of the matcher and metrics
238
+ python evaluation/evaluate_engine.py validate # evaluator accuracy on known-answer cases
239
+ python evaluation/evaluate_engine.py predict # SparkLens (--eval) on all 50 plans -> dataset_predicted.csv
240
+ python evaluation/evaluate_engine.py evaluate # precision / recall / F1 / final score, change vs baseline
241
+ python evaluation/evaluate_engine.py baseline # freeze the current result as eval_baseline.json
242
+ ```
243
+
244
+ `predict` runs the SparkLens CLI, so SparkLens itself must be installed (`pip install -e .`). After a rule change, run `predict --force` and then `evaluate`; improved and regressed plans are listed.
245
+
246
+ To fill `dataset_predicted.csv` by hand, run `python evaluation/evaluate_engine.py export` and paste the output of `sparklens analyze evaluation/eval_plans/<id>.txt --eval` into each row.
247
+
248
+ All evaluation files live in `evaluation/`; run the commands from the repository root.
249
+
250
+ | File | Purpose |
251
+ |---|---|
252
+ | `evaluate_engine.py` | the evaluator (commands above) |
253
+ | `dataset_expected.csv` | ground truth |
254
+ | `dataset_predicted.csv` | SparkLens output being scored |
255
+ | `eval_validation.csv` | paraphrases and role swaps used by `validate` |
256
+ | `eval_baseline.json` | frozen reference result, including evaluator accuracy |
257
+ | `eval_review.csv` | per-claim verdicts of the last run (generated, not tracked) |
258
+
259
+ ---
260
+
261
+ # Architecture
262
+
263
+ ```text
264
+ Spark Physical Plan
265
+ │
266
+ ▼
267
+ Plan Normalizer
268
+ │
269
+ ▼
270
+ Plan Parser
271
+ │
272
+ ▼
273
+ DAG Builder
274
+ │
275
+ ▼
276
+ Execution DAG
277
+ │
278
+ ▼
279
+ Rule Engine
280
+ │
281
+ ▼
282
+ Analysis Report
283
+ │
284
+ ▼
285
+ Recommendation Engine
286
+ │
287
+ ▼
288
+ Prompt Builder
289
+ │
290
+ ▼
291
+ Gemini AI
292
+ │
293
+ ▼
294
+ AI Explanation
295
+
296
+
297
+ Execution DAG
298
+ │
299
+ ▼
300
+ Visualization
301
+ │
302
+ ▼
303
+ dag.png
304
+
305
+
306
+ Analysis Report
307
+ │
308
+ ▼
309
+ Report Generator
310
+ │
311
+ ▼
312
+ report.md
313
+ report.html
314
+ ```
315
+
316
+ ---
317
+
318
+ # Installation
319
+
320
+ Clone the repository:
321
+
322
+ ```bash
323
+ git clone <repository-url>
324
+ cd SparkLens
325
+ ```
326
+
327
+ Install SparkLens locally:
328
+
329
+ ```bash
330
+ pip install -e .
331
+ ```
332
+
333
+ ---
334
+
335
+ # AI Setup
336
+
337
+ Create a `.env` file in the project root:
338
+
339
+ ```env
340
+ GEMINI_API_KEY=your_api_key_here
341
+ ```
342
+
343
+ SparkLens uses Google Gemini for natural-language explanations.
344
+
345
+ ---
346
+
347
+ # Usage
348
+
349
+ ## Deterministic Analysis
350
+
351
+ ```bash
352
+ sparklens analyze sample_plans/broadcast_join.txt
353
+ ```
354
+
355
+ Outputs:
356
+
357
+ ```text
358
+ dag.png
359
+ report.md
360
+ report.html
361
+ ```
362
+
363
+ ---
364
+
365
+ ## AI-Enhanced Analysis
366
+
367
+ ```bash
368
+ sparklens analyze sample_plans/broadcast_join.txt --ai
369
+ ```
370
+
371
+ Outputs:
372
+
373
+ ```text
374
+ dag.png
375
+ report.md
376
+ report.html
377
+ ```
378
+
379
+ with AI-generated:
380
+
381
+ * Executive Summary
382
+ * Execution Explanation
383
+ * Recommendations
384
+
385
+ ## Evaluation Output
386
+
387
+ ```bash
388
+ sparklens analyze evaluation/eval_plans/<id>.txt --eval
389
+ ```
390
+
391
+ This prints the deterministic rule-engine findings as a single line without making AI API calls.
392
+
393
+ ---
394
+
395
+ # Example Findings
396
+
397
+ ```text
398
+ Single Partition Bottleneck
399
+ Severity: High
400
+
401
+ The execution plan contains a SinglePartition exchange,
402
+ which may reduce parallelism by processing data in a
403
+ single partition.
404
+ ```
405
+
406
+ ---
407
+
408
+ # Example AI Summary
409
+
410
+ ```text
411
+ The execution plan uses a Broadcast Hash Join and Adaptive
412
+ Query Execution but contains a critical Single Partition
413
+ bottleneck that limits parallelism and may impact scalability.
414
+ ```
415
+
416
+ ---
417
+
418
+ # Project Structure
419
+
420
+ ```text
421
+ SparkLens/
422
+ │
423
+ ├── ai/
424
+ ├── analyzer/
425
+ ├── cli/
426
+ ├── evaluation/
427
+ ├── models/
428
+ ├── report/
429
+ ├── rules/
430
+ ├── sample_plans/
431
+ ├── tests/
432
+ ├── visualization/
433
+ │
434
+ ├── pyproject.toml
435
+ ├── README.md
436
+ └── .env
437
+ ```
438
+
439
+ ---
440
+
441
+ # Running Tests
442
+
443
+ Execute the full test suite:
444
+
445
+ ```bash
446
+ pytest
447
+ ```
448
+
449
+ ---
450
+
451
+ # Current Capabilities (v0.1.0)
452
+
453
+ * Spark Physical Plan Parsing
454
+ * Execution DAG Construction
455
+ * Rule-Based Analysis
456
+ * Recommendation Generation
457
+ * AI-Assisted Explanations
458
+ * DAG Visualization
459
+ * Markdown Reports
460
+ * HTML Reports
461
+ * CLI Interface
462
+ * Local Package Installation
463
+
464
+ ---
465
+
466
+ # Roadmap
467
+
468
+ ## v0.2
469
+
470
+ * Databricks Integration
471
+ * Rich HTML Dashboard
472
+ * Embedded DAG Visualizations
473
+ * Additional Rule Packs
474
+ * Enhanced Recommendation Engine
475
+
476
+ ## v0.3
477
+
478
+ * Runtime Metrics Analysis
479
+ * Spark UI Integration
480
+ * Query Plan Comparison
481
+ * Execution Replay Visualizations
482
+ * Cost-Based Optimization Insights
483
+
484
+ ---
485
+
486
+ # Version
487
+
488
+ Current Release:
489
+
490
+ ```text
491
+ v0.1.0
492
+ ```
493
+
494
+ ---
495
+
496
+ Built to make Spark execution plans understandable, explainable, and actionable.