dataframe-mutator 0.1.0__tar.gz → 0.2.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- dataframe_mutator-0.2.0/PKG-INFO +172 -0
- dataframe_mutator-0.2.0/README.md +127 -0
- dataframe_mutator-0.2.0/dataframe_mutator.egg-info/PKG-INFO +172 -0
- {dataframe_mutator-0.1.0 → dataframe_mutator-0.2.0}/dataframe_mutator.egg-info/SOURCES.txt +7 -0
- dataframe_mutator-0.2.0/dataframe_mutator.egg-info/entry_points.txt +5 -0
- {dataframe_mutator-0.1.0 → dataframe_mutator-0.2.0}/dataframe_mutator.egg-info/requires.txt +1 -0
- {dataframe_mutator-0.1.0 → dataframe_mutator-0.2.0}/pyproject.toml +8 -1
- dataframe_mutator-0.2.0/src/dataframe_mutator/cli.py +159 -0
- dataframe_mutator-0.2.0/src/dataframe_mutator/config.py +95 -0
- dataframe_mutator-0.2.0/src/dataframe_mutator/integrations.py +170 -0
- dataframe_mutator-0.2.0/src/dataframe_mutator/pytest_plugin.py +78 -0
- dataframe_mutator-0.2.0/src/dataframe_mutator/reports.py +181 -0
- dataframe_mutator-0.2.0/tests/test_advanced_features.py +177 -0
- dataframe_mutator-0.1.0/PKG-INFO +0 -613
- dataframe_mutator-0.1.0/README.md +0 -569
- dataframe_mutator-0.1.0/dataframe_mutator.egg-info/PKG-INFO +0 -613
- {dataframe_mutator-0.1.0 → dataframe_mutator-0.2.0}/LICENSE +0 -0
- {dataframe_mutator-0.1.0 → dataframe_mutator-0.2.0}/dataframe_mutator.egg-info/dependency_links.txt +0 -0
- {dataframe_mutator-0.1.0 → dataframe_mutator-0.2.0}/dataframe_mutator.egg-info/top_level.txt +0 -0
- {dataframe_mutator-0.1.0 → dataframe_mutator-0.2.0}/setup.cfg +0 -0
- {dataframe_mutator-0.1.0 → dataframe_mutator-0.2.0}/src/dataframe_mutator/__init__.py +0 -0
- {dataframe_mutator-0.1.0 → dataframe_mutator-0.2.0}/tests/test_polars_100_coverage.py +0 -0
- {dataframe_mutator-0.1.0 → dataframe_mutator-0.2.0}/tests/test_polars_new_operators.py +0 -0
- {dataframe_mutator-0.1.0 → dataframe_mutator-0.2.0}/tests/test_polars_operators.py +0 -0
|
@@ -0,0 +1,172 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: dataframe-mutator
|
|
3
|
+
Version: 0.2.0
|
|
4
|
+
Summary: Production-grade mutation testing for Polars dataframes. Validate test suite quality by detecting which mutations your tests catch.
|
|
5
|
+
Author-email: Dataframe Mutator Contributors <suhrusai@gmail.com>
|
|
6
|
+
License-Expression: MIT
|
|
7
|
+
Project-URL: Homepage, https://github.com/suhrusai/dataframe-mutator
|
|
8
|
+
Project-URL: Documentation, https://github.com/suhrusai/dataframe-mutator#readme
|
|
9
|
+
Project-URL: Repository, https://github.com/suhrusai/dataframe-mutator.git
|
|
10
|
+
Project-URL: Issues, https://github.com/suhrusai/dataframe-mutator/issues
|
|
11
|
+
Keywords: mutation-testing,polars,dataframe,testing,test-quality,data-pipeline
|
|
12
|
+
Classifier: Development Status :: 4 - Beta
|
|
13
|
+
Classifier: Intended Audience :: Developers
|
|
14
|
+
Classifier: Intended Audience :: Science/Research
|
|
15
|
+
Classifier: Natural Language :: English
|
|
16
|
+
Classifier: Operating System :: OS Independent
|
|
17
|
+
Classifier: Programming Language :: Python :: 3
|
|
18
|
+
Classifier: Programming Language :: Python :: 3.8
|
|
19
|
+
Classifier: Programming Language :: Python :: 3.9
|
|
20
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
21
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
22
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
23
|
+
Classifier: Topic :: Software Development :: Libraries :: Python Modules
|
|
24
|
+
Classifier: Topic :: Scientific/Engineering :: Information Analysis
|
|
25
|
+
Requires-Python: >=3.8
|
|
26
|
+
Description-Content-Type: text/markdown
|
|
27
|
+
License-File: LICENSE
|
|
28
|
+
Requires-Dist: mutmut>=2.4.0
|
|
29
|
+
Requires-Dist: click>=8.0.0
|
|
30
|
+
Provides-Extra: polars
|
|
31
|
+
Requires-Dist: polars>=0.19.0; extra == "polars"
|
|
32
|
+
Provides-Extra: pyspark
|
|
33
|
+
Requires-Dist: pyspark>=3.0.0; extra == "pyspark"
|
|
34
|
+
Provides-Extra: pandas
|
|
35
|
+
Requires-Dist: pandas>=1.0.0; extra == "pandas"
|
|
36
|
+
Provides-Extra: dev
|
|
37
|
+
Requires-Dist: pytest>=7.0.0; extra == "dev"
|
|
38
|
+
Requires-Dist: pytest-cov>=4.0.0; extra == "dev"
|
|
39
|
+
Requires-Dist: black>=23.0.0; extra == "dev"
|
|
40
|
+
Requires-Dist: ruff>=0.1.0; extra == "dev"
|
|
41
|
+
Requires-Dist: mypy>=1.0.0; extra == "dev"
|
|
42
|
+
Provides-Extra: all
|
|
43
|
+
Requires-Dist: dataframe-mutator[dev,pandas,polars,pyspark]; extra == "all"
|
|
44
|
+
Dynamic: license-file
|
|
45
|
+
|
|
46
|
+
# 🧬 dataframe-mutator
|
|
47
|
+
|
|
48
|
+
**Production-grade mutation testing for Polars dataframes.** Validate your test suite quality by automatically detecting which mutations (logic bugs) your tests actually catch.
|
|
49
|
+
|
|
50
|
+
> **Mutation testing** runs your tests against intentionally mutated code. If tests pass despite the mutation, your test is weak. This framework makes it easy to find gaps in data pipeline test coverage.
|
|
51
|
+
|
|
52
|
+
[](https://github.com/suhrusai/dataframe-mutator/actions)
|
|
53
|
+
[](https://github.com/suhrusai/dataframe-mutator/actions)
|
|
54
|
+
[](https://github.com/suhrusai/dataframe-mutator)
|
|
55
|
+
[](https://github.com/suhrusai/dataframe-mutator#features)
|
|
56
|
+
[](https://www.python.org/)
|
|
57
|
+
[](https://github.com/suhrusai/dataframe-mutator/blob/main/LICENSE)
|
|
58
|
+
|
|
59
|
+
## ⚡ Quick Start
|
|
60
|
+
|
|
61
|
+
### Installation
|
|
62
|
+
|
|
63
|
+
```bash
|
|
64
|
+
pip install dataframe-mutator[polars]
|
|
65
|
+
```
|
|
66
|
+
|
|
67
|
+
### 2-Minute Example
|
|
68
|
+
|
|
69
|
+
```python
|
|
70
|
+
import polars as pl
|
|
71
|
+
from dataframe_mutator.polars import SmartPolarsTestRunner
|
|
72
|
+
|
|
73
|
+
# Your pipeline
|
|
74
|
+
def process_sales(df: pl.DataFrame) -> pl.DataFrame:
|
|
75
|
+
return (
|
|
76
|
+
df
|
|
77
|
+
.filter(pl.col("amount") > 100)
|
|
78
|
+
.group_by("region")
|
|
79
|
+
.agg(pl.col("amount").sum())
|
|
80
|
+
)
|
|
81
|
+
|
|
82
|
+
# Test quality
|
|
83
|
+
tester = SmartPolarsTestRunner(test_command="pytest tests/")
|
|
84
|
+
results = tester.analyze_mutation_efficiency("pipeline.py")
|
|
85
|
+
|
|
86
|
+
print(f"High-value mutations: {results['high_value_mutations']}")
|
|
87
|
+
print(f"False positives avoided: {results['potential_false_positives_avoided']:.1f}%")
|
|
88
|
+
```
|
|
89
|
+
|
|
90
|
+
## ✨ Key Features
|
|
91
|
+
|
|
92
|
+
- **109 Production Operators** — Complete Polars API coverage (filtering, aggregations, joins, nulls, strings, datetime, window functions, and more)
|
|
93
|
+
- **Smart Analysis** — AST-aware filtering eliminates false positives (6-10x faster than vanilla mutmut)
|
|
94
|
+
- **Real-World Bug Detection** — Catches boundary errors, wrong aggregations, data loss, null handling mistakes, and boolean logic bugs
|
|
95
|
+
- **Easy Integration** — Works with your existing pytest test suite
|
|
96
|
+
|
|
97
|
+
## What It Catches
|
|
98
|
+
|
|
99
|
+
✅ **Boundary mutations** — `> 0` → `>= 0`, `>= 100` → `> 100`
|
|
100
|
+
✅ **Aggregation swaps** — `sum()` → `mean()`, `count()` → `sum()`
|
|
101
|
+
✅ **Data loss bugs** — `inner_join()` → `left_join()`
|
|
102
|
+
✅ **Calculation errors** — `amount * 1.1` → `amount * 1.0`
|
|
103
|
+
✅ **Boolean logic** — `&` → `|`
|
|
104
|
+
|
|
105
|
+
## How It Works
|
|
106
|
+
|
|
107
|
+
```python
|
|
108
|
+
# 1. Quick analysis
|
|
109
|
+
tester.analyze_mutation_efficiency("src/pipeline.py")
|
|
110
|
+
# Returns: mutations found, false positives avoided, categories
|
|
111
|
+
|
|
112
|
+
# 2. Full mutation testing
|
|
113
|
+
results = tester.mutate_and_test("src/pipeline.py")
|
|
114
|
+
mutation_score = results['survival_rate'] # % of mutations caught
|
|
115
|
+
|
|
116
|
+
# 3. Interpret results
|
|
117
|
+
# 90-100% = Excellent
|
|
118
|
+
# 70-89% = Good
|
|
119
|
+
# 50-69% = Fair (add more tests)
|
|
120
|
+
# <50% = Weak (significant gaps)
|
|
121
|
+
```
|
|
122
|
+
|
|
123
|
+
## Examples & Documentation
|
|
124
|
+
|
|
125
|
+
- **[Sales ETL Pipeline Example](https://github.com/suhrusai/dataframe-mutator/tree/main/examples/etl-pipeline)** — Realistic ETL with tests and mutation testing demo
|
|
126
|
+
- **[Full Documentation](https://github.com/suhrusai/dataframe-mutator)** — Complete API reference, operator list, integration guides
|
|
127
|
+
|
|
128
|
+
## Why Mutation Testing?
|
|
129
|
+
|
|
130
|
+
**Traditional testing:** Verifies expected behavior
|
|
131
|
+
**Mutation testing:** Verifies tests catch bugs when code changes
|
|
132
|
+
|
|
133
|
+
```python
|
|
134
|
+
# Original code
|
|
135
|
+
if age > 18:
|
|
136
|
+
adult = True
|
|
137
|
+
|
|
138
|
+
# Mutated code
|
|
139
|
+
if age >= 18: # Bug: includes exactly 18-year-olds
|
|
140
|
+
adult = True
|
|
141
|
+
|
|
142
|
+
# Traditional test: ✅ PASSES (doesn't test age==18)
|
|
143
|
+
# Mutation test: ❌ FAILS (catches the boundary change)
|
|
144
|
+
```
|
|
145
|
+
|
|
146
|
+
## Integration with CI/CD
|
|
147
|
+
|
|
148
|
+
Works with GitHub Actions, GitLab CI, and other CI systems:
|
|
149
|
+
|
|
150
|
+
```python
|
|
151
|
+
from dataframe_mutator.polars import SmartPolarsTestRunner
|
|
152
|
+
|
|
153
|
+
tester = SmartPolarsTestRunner(test_command="pytest tests/")
|
|
154
|
+
results = tester.analyze_mutation_efficiency("src/pipeline.py")
|
|
155
|
+
|
|
156
|
+
if results['high_value_mutations'] < 10:
|
|
157
|
+
raise Exception("Insufficient test coverage")
|
|
158
|
+
```
|
|
159
|
+
|
|
160
|
+
## Support
|
|
161
|
+
|
|
162
|
+
- 📖 [Full Documentation](https://github.com/suhrusai/dataframe-mutator)
|
|
163
|
+
- 🔗 [GitHub Repository](https://github.com/suhrusai/dataframe-mutator)
|
|
164
|
+
- 🐛 [Report Issues](https://github.com/suhrusai/dataframe-mutator/issues)
|
|
165
|
+
|
|
166
|
+
## License
|
|
167
|
+
|
|
168
|
+
MIT — See [LICENSE](https://github.com/suhrusai/dataframe-mutator/blob/main/LICENSE)
|
|
169
|
+
|
|
170
|
+
---
|
|
171
|
+
|
|
172
|
+
**Built for data scientists and engineers who need confidence in their data pipelines.** 🚀
|
|
@@ -0,0 +1,127 @@
|
|
|
1
|
+
# 🧬 dataframe-mutator
|
|
2
|
+
|
|
3
|
+
**Production-grade mutation testing for Polars dataframes.** Validate your test suite quality by automatically detecting which mutations (logic bugs) your tests actually catch.
|
|
4
|
+
|
|
5
|
+
> **Mutation testing** runs your tests against intentionally mutated code. If tests pass despite the mutation, your test is weak. This framework makes it easy to find gaps in data pipeline test coverage.
|
|
6
|
+
|
|
7
|
+
[](https://github.com/suhrusai/dataframe-mutator/actions)
|
|
8
|
+
[](https://github.com/suhrusai/dataframe-mutator/actions)
|
|
9
|
+
[](https://github.com/suhrusai/dataframe-mutator)
|
|
10
|
+
[](https://github.com/suhrusai/dataframe-mutator#features)
|
|
11
|
+
[](https://www.python.org/)
|
|
12
|
+
[](https://github.com/suhrusai/dataframe-mutator/blob/main/LICENSE)
|
|
13
|
+
|
|
14
|
+
## ⚡ Quick Start
|
|
15
|
+
|
|
16
|
+
### Installation
|
|
17
|
+
|
|
18
|
+
```bash
|
|
19
|
+
pip install dataframe-mutator[polars]
|
|
20
|
+
```
|
|
21
|
+
|
|
22
|
+
### 2-Minute Example
|
|
23
|
+
|
|
24
|
+
```python
|
|
25
|
+
import polars as pl
|
|
26
|
+
from dataframe_mutator.polars import SmartPolarsTestRunner
|
|
27
|
+
|
|
28
|
+
# Your pipeline
|
|
29
|
+
def process_sales(df: pl.DataFrame) -> pl.DataFrame:
|
|
30
|
+
return (
|
|
31
|
+
df
|
|
32
|
+
.filter(pl.col("amount") > 100)
|
|
33
|
+
.group_by("region")
|
|
34
|
+
.agg(pl.col("amount").sum())
|
|
35
|
+
)
|
|
36
|
+
|
|
37
|
+
# Test quality
|
|
38
|
+
tester = SmartPolarsTestRunner(test_command="pytest tests/")
|
|
39
|
+
results = tester.analyze_mutation_efficiency("pipeline.py")
|
|
40
|
+
|
|
41
|
+
print(f"High-value mutations: {results['high_value_mutations']}")
|
|
42
|
+
print(f"False positives avoided: {results['potential_false_positives_avoided']:.1f}%")
|
|
43
|
+
```
|
|
44
|
+
|
|
45
|
+
## ✨ Key Features
|
|
46
|
+
|
|
47
|
+
- **109 Production Operators** — Complete Polars API coverage (filtering, aggregations, joins, nulls, strings, datetime, window functions, and more)
|
|
48
|
+
- **Smart Analysis** — AST-aware filtering eliminates false positives (6-10x faster than vanilla mutmut)
|
|
49
|
+
- **Real-World Bug Detection** — Catches boundary errors, wrong aggregations, data loss, null handling mistakes, and boolean logic bugs
|
|
50
|
+
- **Easy Integration** — Works with your existing pytest test suite
|
|
51
|
+
|
|
52
|
+
## What It Catches
|
|
53
|
+
|
|
54
|
+
✅ **Boundary mutations** — `> 0` → `>= 0`, `>= 100` → `> 100`
|
|
55
|
+
✅ **Aggregation swaps** — `sum()` → `mean()`, `count()` → `sum()`
|
|
56
|
+
✅ **Data loss bugs** — `inner_join()` → `left_join()`
|
|
57
|
+
✅ **Calculation errors** — `amount * 1.1` → `amount * 1.0`
|
|
58
|
+
✅ **Boolean logic** — `&` → `|`
|
|
59
|
+
|
|
60
|
+
## How It Works
|
|
61
|
+
|
|
62
|
+
```python
|
|
63
|
+
# 1. Quick analysis
|
|
64
|
+
tester.analyze_mutation_efficiency("src/pipeline.py")
|
|
65
|
+
# Returns: mutations found, false positives avoided, categories
|
|
66
|
+
|
|
67
|
+
# 2. Full mutation testing
|
|
68
|
+
results = tester.mutate_and_test("src/pipeline.py")
|
|
69
|
+
mutation_score = results['survival_rate'] # % of mutations caught
|
|
70
|
+
|
|
71
|
+
# 3. Interpret results
|
|
72
|
+
# 90-100% = Excellent
|
|
73
|
+
# 70-89% = Good
|
|
74
|
+
# 50-69% = Fair (add more tests)
|
|
75
|
+
# <50% = Weak (significant gaps)
|
|
76
|
+
```
|
|
77
|
+
|
|
78
|
+
## Examples & Documentation
|
|
79
|
+
|
|
80
|
+
- **[Sales ETL Pipeline Example](https://github.com/suhrusai/dataframe-mutator/tree/main/examples/etl-pipeline)** — Realistic ETL with tests and mutation testing demo
|
|
81
|
+
- **[Full Documentation](https://github.com/suhrusai/dataframe-mutator)** — Complete API reference, operator list, integration guides
|
|
82
|
+
|
|
83
|
+
## Why Mutation Testing?
|
|
84
|
+
|
|
85
|
+
**Traditional testing:** Verifies expected behavior
|
|
86
|
+
**Mutation testing:** Verifies tests catch bugs when code changes
|
|
87
|
+
|
|
88
|
+
```python
|
|
89
|
+
# Original code
|
|
90
|
+
if age > 18:
|
|
91
|
+
adult = True
|
|
92
|
+
|
|
93
|
+
# Mutated code
|
|
94
|
+
if age >= 18: # Bug: includes exactly 18-year-olds
|
|
95
|
+
adult = True
|
|
96
|
+
|
|
97
|
+
# Traditional test: ✅ PASSES (doesn't test age==18)
|
|
98
|
+
# Mutation test: ❌ FAILS (catches the boundary change)
|
|
99
|
+
```
|
|
100
|
+
|
|
101
|
+
## Integration with CI/CD
|
|
102
|
+
|
|
103
|
+
Works with GitHub Actions, GitLab CI, and other CI systems:
|
|
104
|
+
|
|
105
|
+
```python
|
|
106
|
+
from dataframe_mutator.polars import SmartPolarsTestRunner
|
|
107
|
+
|
|
108
|
+
tester = SmartPolarsTestRunner(test_command="pytest tests/")
|
|
109
|
+
results = tester.analyze_mutation_efficiency("src/pipeline.py")
|
|
110
|
+
|
|
111
|
+
if results['high_value_mutations'] < 10:
|
|
112
|
+
raise Exception("Insufficient test coverage")
|
|
113
|
+
```
|
|
114
|
+
|
|
115
|
+
## Support
|
|
116
|
+
|
|
117
|
+
- 📖 [Full Documentation](https://github.com/suhrusai/dataframe-mutator)
|
|
118
|
+
- 🔗 [GitHub Repository](https://github.com/suhrusai/dataframe-mutator)
|
|
119
|
+
- 🐛 [Report Issues](https://github.com/suhrusai/dataframe-mutator/issues)
|
|
120
|
+
|
|
121
|
+
## License
|
|
122
|
+
|
|
123
|
+
MIT — See [LICENSE](https://github.com/suhrusai/dataframe-mutator/blob/main/LICENSE)
|
|
124
|
+
|
|
125
|
+
---
|
|
126
|
+
|
|
127
|
+
**Built for data scientists and engineers who need confidence in their data pipelines.** 🚀
|
|
@@ -0,0 +1,172 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: dataframe-mutator
|
|
3
|
+
Version: 0.2.0
|
|
4
|
+
Summary: Production-grade mutation testing for Polars dataframes. Validate test suite quality by detecting which mutations your tests catch.
|
|
5
|
+
Author-email: Dataframe Mutator Contributors <suhrusai@gmail.com>
|
|
6
|
+
License-Expression: MIT
|
|
7
|
+
Project-URL: Homepage, https://github.com/suhrusai/dataframe-mutator
|
|
8
|
+
Project-URL: Documentation, https://github.com/suhrusai/dataframe-mutator#readme
|
|
9
|
+
Project-URL: Repository, https://github.com/suhrusai/dataframe-mutator.git
|
|
10
|
+
Project-URL: Issues, https://github.com/suhrusai/dataframe-mutator/issues
|
|
11
|
+
Keywords: mutation-testing,polars,dataframe,testing,test-quality,data-pipeline
|
|
12
|
+
Classifier: Development Status :: 4 - Beta
|
|
13
|
+
Classifier: Intended Audience :: Developers
|
|
14
|
+
Classifier: Intended Audience :: Science/Research
|
|
15
|
+
Classifier: Natural Language :: English
|
|
16
|
+
Classifier: Operating System :: OS Independent
|
|
17
|
+
Classifier: Programming Language :: Python :: 3
|
|
18
|
+
Classifier: Programming Language :: Python :: 3.8
|
|
19
|
+
Classifier: Programming Language :: Python :: 3.9
|
|
20
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
21
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
22
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
23
|
+
Classifier: Topic :: Software Development :: Libraries :: Python Modules
|
|
24
|
+
Classifier: Topic :: Scientific/Engineering :: Information Analysis
|
|
25
|
+
Requires-Python: >=3.8
|
|
26
|
+
Description-Content-Type: text/markdown
|
|
27
|
+
License-File: LICENSE
|
|
28
|
+
Requires-Dist: mutmut>=2.4.0
|
|
29
|
+
Requires-Dist: click>=8.0.0
|
|
30
|
+
Provides-Extra: polars
|
|
31
|
+
Requires-Dist: polars>=0.19.0; extra == "polars"
|
|
32
|
+
Provides-Extra: pyspark
|
|
33
|
+
Requires-Dist: pyspark>=3.0.0; extra == "pyspark"
|
|
34
|
+
Provides-Extra: pandas
|
|
35
|
+
Requires-Dist: pandas>=1.0.0; extra == "pandas"
|
|
36
|
+
Provides-Extra: dev
|
|
37
|
+
Requires-Dist: pytest>=7.0.0; extra == "dev"
|
|
38
|
+
Requires-Dist: pytest-cov>=4.0.0; extra == "dev"
|
|
39
|
+
Requires-Dist: black>=23.0.0; extra == "dev"
|
|
40
|
+
Requires-Dist: ruff>=0.1.0; extra == "dev"
|
|
41
|
+
Requires-Dist: mypy>=1.0.0; extra == "dev"
|
|
42
|
+
Provides-Extra: all
|
|
43
|
+
Requires-Dist: dataframe-mutator[dev,pandas,polars,pyspark]; extra == "all"
|
|
44
|
+
Dynamic: license-file
|
|
45
|
+
|
|
46
|
+
# 🧬 dataframe-mutator
|
|
47
|
+
|
|
48
|
+
**Production-grade mutation testing for Polars dataframes.** Validate your test suite quality by automatically detecting which mutations (logic bugs) your tests actually catch.
|
|
49
|
+
|
|
50
|
+
> **Mutation testing** runs your tests against intentionally mutated code. If tests pass despite the mutation, your test is weak. This framework makes it easy to find gaps in data pipeline test coverage.
|
|
51
|
+
|
|
52
|
+
[](https://github.com/suhrusai/dataframe-mutator/actions)
|
|
53
|
+
[](https://github.com/suhrusai/dataframe-mutator/actions)
|
|
54
|
+
[](https://github.com/suhrusai/dataframe-mutator)
|
|
55
|
+
[](https://github.com/suhrusai/dataframe-mutator#features)
|
|
56
|
+
[](https://www.python.org/)
|
|
57
|
+
[](https://github.com/suhrusai/dataframe-mutator/blob/main/LICENSE)
|
|
58
|
+
|
|
59
|
+
## ⚡ Quick Start
|
|
60
|
+
|
|
61
|
+
### Installation
|
|
62
|
+
|
|
63
|
+
```bash
|
|
64
|
+
pip install dataframe-mutator[polars]
|
|
65
|
+
```
|
|
66
|
+
|
|
67
|
+
### 2-Minute Example
|
|
68
|
+
|
|
69
|
+
```python
|
|
70
|
+
import polars as pl
|
|
71
|
+
from dataframe_mutator.polars import SmartPolarsTestRunner
|
|
72
|
+
|
|
73
|
+
# Your pipeline
|
|
74
|
+
def process_sales(df: pl.DataFrame) -> pl.DataFrame:
|
|
75
|
+
return (
|
|
76
|
+
df
|
|
77
|
+
.filter(pl.col("amount") > 100)
|
|
78
|
+
.group_by("region")
|
|
79
|
+
.agg(pl.col("amount").sum())
|
|
80
|
+
)
|
|
81
|
+
|
|
82
|
+
# Test quality
|
|
83
|
+
tester = SmartPolarsTestRunner(test_command="pytest tests/")
|
|
84
|
+
results = tester.analyze_mutation_efficiency("pipeline.py")
|
|
85
|
+
|
|
86
|
+
print(f"High-value mutations: {results['high_value_mutations']}")
|
|
87
|
+
print(f"False positives avoided: {results['potential_false_positives_avoided']:.1f}%")
|
|
88
|
+
```
|
|
89
|
+
|
|
90
|
+
## ✨ Key Features
|
|
91
|
+
|
|
92
|
+
- **109 Production Operators** — Complete Polars API coverage (filtering, aggregations, joins, nulls, strings, datetime, window functions, and more)
|
|
93
|
+
- **Smart Analysis** — AST-aware filtering eliminates false positives (6-10x faster than vanilla mutmut)
|
|
94
|
+
- **Real-World Bug Detection** — Catches boundary errors, wrong aggregations, data loss, null handling mistakes, and boolean logic bugs
|
|
95
|
+
- **Easy Integration** — Works with your existing pytest test suite
|
|
96
|
+
|
|
97
|
+
## What It Catches
|
|
98
|
+
|
|
99
|
+
✅ **Boundary mutations** — `> 0` → `>= 0`, `>= 100` → `> 100`
|
|
100
|
+
✅ **Aggregation swaps** — `sum()` → `mean()`, `count()` → `sum()`
|
|
101
|
+
✅ **Data loss bugs** — `inner_join()` → `left_join()`
|
|
102
|
+
✅ **Calculation errors** — `amount * 1.1` → `amount * 1.0`
|
|
103
|
+
✅ **Boolean logic** — `&` → `|`
|
|
104
|
+
|
|
105
|
+
## How It Works
|
|
106
|
+
|
|
107
|
+
```python
|
|
108
|
+
# 1. Quick analysis
|
|
109
|
+
tester.analyze_mutation_efficiency("src/pipeline.py")
|
|
110
|
+
# Returns: mutations found, false positives avoided, categories
|
|
111
|
+
|
|
112
|
+
# 2. Full mutation testing
|
|
113
|
+
results = tester.mutate_and_test("src/pipeline.py")
|
|
114
|
+
mutation_score = results['survival_rate'] # % of mutations caught
|
|
115
|
+
|
|
116
|
+
# 3. Interpret results
|
|
117
|
+
# 90-100% = Excellent
|
|
118
|
+
# 70-89% = Good
|
|
119
|
+
# 50-69% = Fair (add more tests)
|
|
120
|
+
# <50% = Weak (significant gaps)
|
|
121
|
+
```
|
|
122
|
+
|
|
123
|
+
## Examples & Documentation
|
|
124
|
+
|
|
125
|
+
- **[Sales ETL Pipeline Example](https://github.com/suhrusai/dataframe-mutator/tree/main/examples/etl-pipeline)** — Realistic ETL with tests and mutation testing demo
|
|
126
|
+
- **[Full Documentation](https://github.com/suhrusai/dataframe-mutator)** — Complete API reference, operator list, integration guides
|
|
127
|
+
|
|
128
|
+
## Why Mutation Testing?
|
|
129
|
+
|
|
130
|
+
**Traditional testing:** Verifies expected behavior
|
|
131
|
+
**Mutation testing:** Verifies tests catch bugs when code changes
|
|
132
|
+
|
|
133
|
+
```python
|
|
134
|
+
# Original code
|
|
135
|
+
if age > 18:
|
|
136
|
+
adult = True
|
|
137
|
+
|
|
138
|
+
# Mutated code
|
|
139
|
+
if age >= 18: # Bug: includes exactly 18-year-olds
|
|
140
|
+
adult = True
|
|
141
|
+
|
|
142
|
+
# Traditional test: ✅ PASSES (doesn't test age==18)
|
|
143
|
+
# Mutation test: ❌ FAILS (catches the boundary change)
|
|
144
|
+
```
|
|
145
|
+
|
|
146
|
+
## Integration with CI/CD
|
|
147
|
+
|
|
148
|
+
Works with GitHub Actions, GitLab CI, and other CI systems:
|
|
149
|
+
|
|
150
|
+
```python
|
|
151
|
+
from dataframe_mutator.polars import SmartPolarsTestRunner
|
|
152
|
+
|
|
153
|
+
tester = SmartPolarsTestRunner(test_command="pytest tests/")
|
|
154
|
+
results = tester.analyze_mutation_efficiency("src/pipeline.py")
|
|
155
|
+
|
|
156
|
+
if results['high_value_mutations'] < 10:
|
|
157
|
+
raise Exception("Insufficient test coverage")
|
|
158
|
+
```
|
|
159
|
+
|
|
160
|
+
## Support
|
|
161
|
+
|
|
162
|
+
- 📖 [Full Documentation](https://github.com/suhrusai/dataframe-mutator)
|
|
163
|
+
- 🔗 [GitHub Repository](https://github.com/suhrusai/dataframe-mutator)
|
|
164
|
+
- 🐛 [Report Issues](https://github.com/suhrusai/dataframe-mutator/issues)
|
|
165
|
+
|
|
166
|
+
## License
|
|
167
|
+
|
|
168
|
+
MIT — See [LICENSE](https://github.com/suhrusai/dataframe-mutator/blob/main/LICENSE)
|
|
169
|
+
|
|
170
|
+
---
|
|
171
|
+
|
|
172
|
+
**Built for data scientists and engineers who need confidence in their data pipelines.** 🚀
|
|
@@ -4,9 +4,16 @@ pyproject.toml
|
|
|
4
4
|
dataframe_mutator.egg-info/PKG-INFO
|
|
5
5
|
dataframe_mutator.egg-info/SOURCES.txt
|
|
6
6
|
dataframe_mutator.egg-info/dependency_links.txt
|
|
7
|
+
dataframe_mutator.egg-info/entry_points.txt
|
|
7
8
|
dataframe_mutator.egg-info/requires.txt
|
|
8
9
|
dataframe_mutator.egg-info/top_level.txt
|
|
9
10
|
src/dataframe_mutator/__init__.py
|
|
11
|
+
src/dataframe_mutator/cli.py
|
|
12
|
+
src/dataframe_mutator/config.py
|
|
13
|
+
src/dataframe_mutator/integrations.py
|
|
14
|
+
src/dataframe_mutator/pytest_plugin.py
|
|
15
|
+
src/dataframe_mutator/reports.py
|
|
16
|
+
tests/test_advanced_features.py
|
|
10
17
|
tests/test_polars_100_coverage.py
|
|
11
18
|
tests/test_polars_new_operators.py
|
|
12
19
|
tests/test_polars_operators.py
|
|
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
|
|
|
4
4
|
|
|
5
5
|
[project]
|
|
6
6
|
name = "dataframe-mutator"
|
|
7
|
-
version = "0.
|
|
7
|
+
version = "0.2.0"
|
|
8
8
|
description = "Production-grade mutation testing for Polars dataframes. Validate test suite quality by detecting which mutations your tests catch."
|
|
9
9
|
readme = "README.md"
|
|
10
10
|
requires-python = ">=3.8"
|
|
@@ -30,8 +30,15 @@ classifiers = [
|
|
|
30
30
|
]
|
|
31
31
|
dependencies = [
|
|
32
32
|
"mutmut>=2.4.0",
|
|
33
|
+
"click>=8.0.0",
|
|
33
34
|
]
|
|
34
35
|
|
|
36
|
+
[project.scripts]
|
|
37
|
+
dataframe-mutator = "dataframe_mutator.cli:cli"
|
|
38
|
+
|
|
39
|
+
[project.entry-points.pytest11]
|
|
40
|
+
dataframe-mutator = "dataframe_mutator.pytest_plugin"
|
|
41
|
+
|
|
35
42
|
[project.urls]
|
|
36
43
|
Homepage = "https://github.com/suhrusai/dataframe-mutator"
|
|
37
44
|
Documentation = "https://github.com/suhrusai/dataframe-mutator#readme"
|
|
@@ -0,0 +1,159 @@
|
|
|
1
|
+
"""Command-line interface for dataframe-mutator."""
|
|
2
|
+
|
|
3
|
+
import click
|
|
4
|
+
from pathlib import Path
|
|
5
|
+
from typing import Optional
|
|
6
|
+
from .config import MutationConfig
|
|
7
|
+
from .polars import SmartPolarsTestRunner
|
|
8
|
+
from .reports import HTMLReportGenerator, JSONExporter, BaselineTracker
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
@click.group()
|
|
12
|
+
def cli():
|
|
13
|
+
"""dataframe-mutator - Mutation testing for Polars dataframes."""
|
|
14
|
+
pass
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
@cli.command()
|
|
18
|
+
@click.argument("source", type=click.Path(exists=True), nargs=-1, required=True)
|
|
19
|
+
@click.option("--tests", default="tests/", help="Path to test directory")
|
|
20
|
+
@click.option("--config", default="dataframe-mutator.toml", help="Config file path")
|
|
21
|
+
@click.option("--output", default="text", type=click.Choice(["text", "json", "html"]),
|
|
22
|
+
help="Output format")
|
|
23
|
+
@click.option("--threshold", type=float, help="Mutation score threshold")
|
|
24
|
+
@click.option("--parallel", is_flag=True, default=True, help="Use parallel testing")
|
|
25
|
+
@click.option("--workers", type=int, help="Number of parallel workers")
|
|
26
|
+
@click.option("--save-baseline", is_flag=True, help="Save current results as baseline")
|
|
27
|
+
def analyze(source, tests, config, output, threshold, parallel, workers, save_baseline):
|
|
28
|
+
"""Analyze mutation efficiency of your code."""
|
|
29
|
+
|
|
30
|
+
# Load configuration
|
|
31
|
+
cfg = MutationConfig.from_toml(config)
|
|
32
|
+
|
|
33
|
+
# Override with CLI arguments
|
|
34
|
+
if threshold:
|
|
35
|
+
cfg.mutation_threshold = threshold
|
|
36
|
+
if not parallel:
|
|
37
|
+
cfg.parallel = False
|
|
38
|
+
if workers:
|
|
39
|
+
cfg.num_workers = workers
|
|
40
|
+
cfg.output_format = output
|
|
41
|
+
cfg.source_files = list(source)
|
|
42
|
+
cfg.test_command = f"pytest {tests}"
|
|
43
|
+
|
|
44
|
+
click.echo(f"[*] Analyzing {len(source)} file(s)...")
|
|
45
|
+
click.echo(f"[TEST] Test command: {cfg.test_command}")
|
|
46
|
+
|
|
47
|
+
try:
|
|
48
|
+
tester = SmartPolarsTestRunner(
|
|
49
|
+
test_command=cfg.test_command,
|
|
50
|
+
parallel=cfg.parallel,
|
|
51
|
+
num_workers=cfg.num_workers
|
|
52
|
+
)
|
|
53
|
+
|
|
54
|
+
all_results = {}
|
|
55
|
+
for src_file in source:
|
|
56
|
+
click.echo(f"\n[FILE] {src_file}")
|
|
57
|
+
results = tester.analyze_mutation_efficiency(src_file)
|
|
58
|
+
all_results[src_file] = results
|
|
59
|
+
|
|
60
|
+
click.echo(f" [OK] High-value mutations: {results['high_value_mutations']}")
|
|
61
|
+
click.echo(f" [STAT] False positives avoided: {results['potential_false_positives_avoided']:.1f}%")
|
|
62
|
+
|
|
63
|
+
if results['high_value_mutations'] == 0:
|
|
64
|
+
click.echo(" [WARN] No mutations found!")
|
|
65
|
+
|
|
66
|
+
# Generate reports
|
|
67
|
+
Path(cfg.output_dir).mkdir(exist_ok=True)
|
|
68
|
+
|
|
69
|
+
if output in ["text", "json", "html"]:
|
|
70
|
+
if output == "json" or output == "html":
|
|
71
|
+
exporter = JSONExporter(cfg)
|
|
72
|
+
exporter.export(all_results, f"{cfg.output_dir}/results.json")
|
|
73
|
+
click.echo(f"\n[REPORT] JSON report: {cfg.output_dir}/results.json")
|
|
74
|
+
|
|
75
|
+
if output == "html":
|
|
76
|
+
generator = HTMLReportGenerator(cfg)
|
|
77
|
+
generator.generate(all_results, f"{cfg.output_dir}/report.html")
|
|
78
|
+
click.echo(f"[REPORT] HTML report: {cfg.output_dir}/report.html")
|
|
79
|
+
|
|
80
|
+
# Save baseline if requested
|
|
81
|
+
if save_baseline:
|
|
82
|
+
tracker = BaselineTracker(cfg)
|
|
83
|
+
tracker.save_baseline(all_results)
|
|
84
|
+
click.echo(f"\n[SAVE] Baseline saved: {cfg.baseline_path}")
|
|
85
|
+
|
|
86
|
+
# Check threshold
|
|
87
|
+
avg_mutations = sum(r['high_value_mutations'] for r in all_results.values()) / len(all_results)
|
|
88
|
+
if avg_mutations < cfg.min_mutations:
|
|
89
|
+
click.echo(f"\n[FAIL] Below minimum mutations ({avg_mutations:.0f} < {cfg.min_mutations})")
|
|
90
|
+
raise click.Exit(1)
|
|
91
|
+
|
|
92
|
+
click.echo("\n[SUCCESS] Analysis complete!")
|
|
93
|
+
|
|
94
|
+
except Exception as e:
|
|
95
|
+
click.echo(f"[ERROR] Error: {e}", err=True)
|
|
96
|
+
raise click.Exit(1)
|
|
97
|
+
|
|
98
|
+
|
|
99
|
+
@cli.command()
|
|
100
|
+
@click.argument("source", type=click.Path(exists=True), nargs=-1, required=True)
|
|
101
|
+
@click.option("--tests", default="tests/", help="Path to test directory")
|
|
102
|
+
@click.option("--config", default="dataframe-mutator.toml", help="Config file path")
|
|
103
|
+
def check(source, tests, config):
|
|
104
|
+
"""Check mutation score against baseline."""
|
|
105
|
+
|
|
106
|
+
cfg = MutationConfig.from_toml(config)
|
|
107
|
+
cfg.source_files = list(source)
|
|
108
|
+
cfg.test_command = f"pytest {tests}"
|
|
109
|
+
|
|
110
|
+
tracker = BaselineTracker(cfg)
|
|
111
|
+
|
|
112
|
+
if not Path(cfg.baseline_path).exists():
|
|
113
|
+
click.echo("[FAIL] No baseline found. Run with --save-baseline first.")
|
|
114
|
+
raise click.Exit(1)
|
|
115
|
+
|
|
116
|
+
tester = SmartPolarsTestRunner(test_command=cfg.test_command)
|
|
117
|
+
baseline = tracker.load_baseline()
|
|
118
|
+
|
|
119
|
+
click.echo("[*] Comparing to baseline...")
|
|
120
|
+
|
|
121
|
+
all_pass = True
|
|
122
|
+
for src_file in source:
|
|
123
|
+
results = tester.analyze_mutation_efficiency(src_file)
|
|
124
|
+
current = results['high_value_mutations']
|
|
125
|
+
baseline_val = baseline.get(src_file, {}).get('high_value_mutations', 0)
|
|
126
|
+
|
|
127
|
+
change = current - baseline_val
|
|
128
|
+
if change >= 0:
|
|
129
|
+
click.echo(f"[OK] {src_file}: {current} mutations (+{change})")
|
|
130
|
+
else:
|
|
131
|
+
click.echo(f"[WARN] {src_file}: {current} mutations ({change})")
|
|
132
|
+
all_pass = False
|
|
133
|
+
|
|
134
|
+
if not all_pass:
|
|
135
|
+
click.echo("\n[WARN] Some files have fewer mutations than baseline")
|
|
136
|
+
raise click.Exit(1)
|
|
137
|
+
|
|
138
|
+
click.echo("\n[OK] All files meet or exceed baseline!")
|
|
139
|
+
|
|
140
|
+
|
|
141
|
+
@cli.command()
|
|
142
|
+
@click.option("--config", default="dataframe-mutator.toml", help="Config file path")
|
|
143
|
+
def init(config):
|
|
144
|
+
"""Initialize configuration file."""
|
|
145
|
+
|
|
146
|
+
cfg = MutationConfig()
|
|
147
|
+
cfg.to_json(config)
|
|
148
|
+
click.echo(f"[OK] Created {config}")
|
|
149
|
+
click.echo("Edit this file to customize mutation testing settings.")
|
|
150
|
+
|
|
151
|
+
|
|
152
|
+
@cli.command()
|
|
153
|
+
def version():
|
|
154
|
+
"""Show version."""
|
|
155
|
+
click.echo("dataframe-mutator 0.1.1")
|
|
156
|
+
|
|
157
|
+
|
|
158
|
+
if __name__ == "__main__":
|
|
159
|
+
cli()
|