csv-refine 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- csv_refine-0.1.0/LICENSE +20 -0
- csv_refine-0.1.0/PKG-INFO +162 -0
- csv_refine-0.1.0/README.md +145 -0
- csv_refine-0.1.0/pyproject.toml +32 -0
- csv_refine-0.1.0/setup.cfg +4 -0
- csv_refine-0.1.0/src/csv_refine/__init__.py +0 -0
- csv_refine-0.1.0/src/csv_refine/build_summary.py +152 -0
- csv_refine-0.1.0/src/csv_refine/cli.py +80 -0
- csv_refine-0.1.0/src/csv_refine/contract.py +571 -0
- csv_refine-0.1.0/src/csv_refine/contract_models.py +81 -0
- csv_refine-0.1.0/src/csv_refine/exceptions.py +10 -0
- csv_refine-0.1.0/src/csv_refine/logging_config.py +19 -0
- csv_refine-0.1.0/src/csv_refine/orchestration.py +67 -0
- csv_refine-0.1.0/src/csv_refine/row_processing.py +314 -0
- csv_refine-0.1.0/src/csv_refine/validate_csv.py +145 -0
- csv_refine-0.1.0/src/csv_refine/validate_unique.py +155 -0
- csv_refine-0.1.0/src/csv_refine/value_operation.py +220 -0
- csv_refine-0.1.0/src/csv_refine/write_csv.py +60 -0
- csv_refine-0.1.0/src/csv_refine/write_errors_csv.py +77 -0
- csv_refine-0.1.0/src/csv_refine/write_report.py +189 -0
- csv_refine-0.1.0/src/csv_refine.egg-info/PKG-INFO +162 -0
- csv_refine-0.1.0/src/csv_refine.egg-info/SOURCES.txt +36 -0
- csv_refine-0.1.0/src/csv_refine.egg-info/dependency_links.txt +1 -0
- csv_refine-0.1.0/src/csv_refine.egg-info/entry_points.txt +2 -0
- csv_refine-0.1.0/src/csv_refine.egg-info/requires.txt +5 -0
- csv_refine-0.1.0/src/csv_refine.egg-info/top_level.txt +1 -0
- csv_refine-0.1.0/tests/test_build_summary.py +260 -0
- csv_refine-0.1.0/tests/test_cli.py +44 -0
- csv_refine-0.1.0/tests/test_contract_models.py +120 -0
- csv_refine-0.1.0/tests/test_csv_row_processing.py +367 -0
- csv_refine-0.1.0/tests/test_logging_config.py +22 -0
- csv_refine-0.1.0/tests/test_orchestration.py +145 -0
- csv_refine-0.1.0/tests/test_validate_csv.py +211 -0
- csv_refine-0.1.0/tests/test_validate_unique.py +492 -0
- csv_refine-0.1.0/tests/test_value_operation.py +430 -0
- csv_refine-0.1.0/tests/test_write_csv.py +107 -0
- csv_refine-0.1.0/tests/test_write_errors_csv.py +170 -0
- csv_refine-0.1.0/tests/test_write_report.py +122 -0
csv_refine-0.1.0/LICENSE
ADDED
|
@@ -0,0 +1,20 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Kuypers Alexis
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software without limitation the rights
|
|
7
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
8
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
9
|
+
furnished to do so, subject to the following conditions:
|
|
10
|
+
|
|
11
|
+
The above copyright notice and this permission notice shall be included in all
|
|
12
|
+
copies or substantial portions of the Software.
|
|
13
|
+
|
|
14
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
15
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
16
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
17
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
18
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
19
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
20
|
+
SOFTWARE.
|
|
@@ -0,0 +1,162 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: csv-refine
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: A Python CLI for cleaning, filtering, and normalizing CSV files using configurable YAML data contracts.
|
|
5
|
+
License-Expression: MIT
|
|
6
|
+
Project-URL: Homepage, https://github.com/alexiskuypers/csv-refine
|
|
7
|
+
Project-URL: Repository, https://github.com/alexiskuypers/csv-refine
|
|
8
|
+
Project-URL: Issues, https://github.com/alexiskuypers/csv-refine/issues
|
|
9
|
+
Requires-Python: >=3.12
|
|
10
|
+
Description-Content-Type: text/markdown
|
|
11
|
+
License-File: LICENSE
|
|
12
|
+
Requires-Dist: PyYAML>=6.0.3
|
|
13
|
+
Requires-Dist: email-validator>=2.3.0
|
|
14
|
+
Provides-Extra: dev
|
|
15
|
+
Requires-Dist: pytest>=9.1.1; extra == "dev"
|
|
16
|
+
Dynamic: license-file
|
|
17
|
+
|
|
18
|
+
# CSV Refine
|
|
19
|
+
|
|
20
|
+
CSV Refine is a Python CLI for cleaning, filtering, and normalizing CSV files.
|
|
21
|
+
|
|
22
|
+
It lets you define exactly how each column should be validated, cleaned, and normalized through a YAML contract. CSV Refine then separates valid and invalid rows and generates reports to assess the overall health of the dataset.
|
|
23
|
+
|
|
24
|
+
## Features
|
|
25
|
+
|
|
26
|
+
- Clean and normalize messy CSV data
|
|
27
|
+
- Filter out invalid rows while preserving valid data
|
|
28
|
+
- Detect missing values, duplicates, invalid formats, and rule violations
|
|
29
|
+
- Separate valid and invalid rows into dedicated CSV files
|
|
30
|
+
- Generate JSON and HTML reports to assess CSV data quality
|
|
31
|
+
|
|
32
|
+
## Installation
|
|
33
|
+
|
|
34
|
+
```bash
|
|
35
|
+
git clone https://github.com/alexiskuypers/csv-refine.git
|
|
36
|
+
cd csv-refine
|
|
37
|
+
|
|
38
|
+
python3 -m venv .venv
|
|
39
|
+
source .venv/bin/activate
|
|
40
|
+
|
|
41
|
+
pip install -e .
|
|
42
|
+
```
|
|
43
|
+
|
|
44
|
+
Check the CLI:
|
|
45
|
+
|
|
46
|
+
```bash
|
|
47
|
+
csv-refine --help
|
|
48
|
+
```
|
|
49
|
+
|
|
50
|
+
## Quick start
|
|
51
|
+
|
|
52
|
+
```bash
|
|
53
|
+
csv-refine \
|
|
54
|
+
--contract examples/01-input/example-contract.yaml \
|
|
55
|
+
--csv examples/01-input/example.csv \
|
|
56
|
+
--mode permissive
|
|
57
|
+
```
|
|
58
|
+
|
|
59
|
+
By default, generated files are written to `output/`.
|
|
60
|
+
|
|
61
|
+
A custom output directory can be provided with `--output`.
|
|
62
|
+
|
|
63
|
+
## YAML contract
|
|
64
|
+
|
|
65
|
+
CSV Refine is configured through a YAML contract that defines exactly how each column should be validated and normalized.
|
|
66
|
+
|
|
67
|
+
A contract can define data types, nullability, uniqueness, validation rules, transformations, delimiter, and encoding.
|
|
68
|
+
|
|
69
|
+
Example:
|
|
70
|
+
|
|
71
|
+
```yaml
|
|
72
|
+
columns:
|
|
73
|
+
Transaction ID:
|
|
74
|
+
type: str
|
|
75
|
+
nullable: false
|
|
76
|
+
unique: true
|
|
77
|
+
rules:
|
|
78
|
+
starts_with: "TXN_"
|
|
79
|
+
```
|
|
80
|
+
|
|
81
|
+
For the complete contract syntax and all supported options, see the [contract README](contracts/README.md).
|
|
82
|
+
|
|
83
|
+
## Validation modes
|
|
84
|
+
|
|
85
|
+
### Permissive
|
|
86
|
+
|
|
87
|
+
Invalid rows are collected while processing continues.
|
|
88
|
+
|
|
89
|
+
### Strict
|
|
90
|
+
|
|
91
|
+
Structural, type conversion, nullability, and uniqueness failures can stop processing.
|
|
92
|
+
|
|
93
|
+
Validation-rule violations still act as row filters.
|
|
94
|
+
|
|
95
|
+
## Outputs
|
|
96
|
+
|
|
97
|
+
CSV Refine generates:
|
|
98
|
+
|
|
99
|
+
```text
|
|
100
|
+
output/
|
|
101
|
+
├── validated_example.csv
|
|
102
|
+
├── errors_example.csv
|
|
103
|
+
├── report_example.json
|
|
104
|
+
└── report_example.html
|
|
105
|
+
```
|
|
106
|
+
|
|
107
|
+
- **Validated CSV** — valid rows with transformations applied
|
|
108
|
+
- **Error CSV** — invalid rows with their detected errors
|
|
109
|
+
- **JSON report** — machine-readable summary
|
|
110
|
+
- **HTML report** — human-readable overview of CSV health
|
|
111
|
+
|
|
112
|
+
The original CSV is never modified.
|
|
113
|
+
|
|
114
|
+
## Real-world example
|
|
115
|
+
|
|
116
|
+
The repository includes a 200-row retail dataset containing missing values, invalid categories, malformed identifiers, duplicate values, invalid dates, and conversion errors.
|
|
117
|
+
|
|
118
|
+
```text
|
|
119
|
+
examples/
|
|
120
|
+
├── 01-input/
|
|
121
|
+
│ ├── example.csv
|
|
122
|
+
│ └── example-contract.yaml
|
|
123
|
+
└── 02-expected-output/
|
|
124
|
+
├── validated_example.csv
|
|
125
|
+
├── errors_example.csv
|
|
126
|
+
├── report_example.json
|
|
127
|
+
└── report_example.html
|
|
128
|
+
```
|
|
129
|
+
|
|
130
|
+
Run the example with:
|
|
131
|
+
|
|
132
|
+
```bash
|
|
133
|
+
csv-refine \
|
|
134
|
+
--contract examples/01-input/example-contract.yaml \
|
|
135
|
+
--csv examples/01-input/example.csv \
|
|
136
|
+
--mode permissive
|
|
137
|
+
```
|
|
138
|
+
|
|
139
|
+
Example result:
|
|
140
|
+
|
|
141
|
+
```text
|
|
142
|
+
Total rows: 200
|
|
143
|
+
Valid rows: 133
|
|
144
|
+
Invalid rows: 67
|
|
145
|
+
Total errors: 117
|
|
146
|
+
```
|
|
147
|
+
|
|
148
|
+
## Tests
|
|
149
|
+
|
|
150
|
+
Run the test suite with:
|
|
151
|
+
|
|
152
|
+
```bash
|
|
153
|
+
python3 -m pytest
|
|
154
|
+
```
|
|
155
|
+
|
|
156
|
+
## Stack
|
|
157
|
+
|
|
158
|
+
Python 3.12+, PyYAML, email-validator, pytest, argparse, pathlib, logging.
|
|
159
|
+
|
|
160
|
+
## Status
|
|
161
|
+
|
|
162
|
+
**Version 1 is functionally complete.**
|
|
@@ -0,0 +1,145 @@
|
|
|
1
|
+
# CSV Refine
|
|
2
|
+
|
|
3
|
+
CSV Refine is a Python CLI for cleaning, filtering, and normalizing CSV files.
|
|
4
|
+
|
|
5
|
+
It lets you define exactly how each column should be validated, cleaned, and normalized through a YAML contract. CSV Refine then separates valid and invalid rows and generates reports to assess the overall health of the dataset.
|
|
6
|
+
|
|
7
|
+
## Features
|
|
8
|
+
|
|
9
|
+
- Clean and normalize messy CSV data
|
|
10
|
+
- Filter out invalid rows while preserving valid data
|
|
11
|
+
- Detect missing values, duplicates, invalid formats, and rule violations
|
|
12
|
+
- Separate valid and invalid rows into dedicated CSV files
|
|
13
|
+
- Generate JSON and HTML reports to assess CSV data quality
|
|
14
|
+
|
|
15
|
+
## Installation
|
|
16
|
+
|
|
17
|
+
```bash
|
|
18
|
+
git clone https://github.com/alexiskuypers/csv-refine.git
|
|
19
|
+
cd csv-refine
|
|
20
|
+
|
|
21
|
+
python3 -m venv .venv
|
|
22
|
+
source .venv/bin/activate
|
|
23
|
+
|
|
24
|
+
pip install -e .
|
|
25
|
+
```
|
|
26
|
+
|
|
27
|
+
Check the CLI:
|
|
28
|
+
|
|
29
|
+
```bash
|
|
30
|
+
csv-refine --help
|
|
31
|
+
```
|
|
32
|
+
|
|
33
|
+
## Quick start
|
|
34
|
+
|
|
35
|
+
```bash
|
|
36
|
+
csv-refine \
|
|
37
|
+
--contract examples/01-input/example-contract.yaml \
|
|
38
|
+
--csv examples/01-input/example.csv \
|
|
39
|
+
--mode permissive
|
|
40
|
+
```
|
|
41
|
+
|
|
42
|
+
By default, generated files are written to `output/`.
|
|
43
|
+
|
|
44
|
+
A custom output directory can be provided with `--output`.
|
|
45
|
+
|
|
46
|
+
## YAML contract
|
|
47
|
+
|
|
48
|
+
CSV Refine is configured through a YAML contract that defines exactly how each column should be validated and normalized.
|
|
49
|
+
|
|
50
|
+
A contract can define data types, nullability, uniqueness, validation rules, transformations, delimiter, and encoding.
|
|
51
|
+
|
|
52
|
+
Example:
|
|
53
|
+
|
|
54
|
+
```yaml
|
|
55
|
+
columns:
|
|
56
|
+
Transaction ID:
|
|
57
|
+
type: str
|
|
58
|
+
nullable: false
|
|
59
|
+
unique: true
|
|
60
|
+
rules:
|
|
61
|
+
starts_with: "TXN_"
|
|
62
|
+
```
|
|
63
|
+
|
|
64
|
+
For the complete contract syntax and all supported options, see the [contract README](contracts/README.md).
|
|
65
|
+
|
|
66
|
+
## Validation modes
|
|
67
|
+
|
|
68
|
+
### Permissive
|
|
69
|
+
|
|
70
|
+
Invalid rows are collected while processing continues.
|
|
71
|
+
|
|
72
|
+
### Strict
|
|
73
|
+
|
|
74
|
+
Structural, type conversion, nullability, and uniqueness failures can stop processing.
|
|
75
|
+
|
|
76
|
+
Validation-rule violations still act as row filters.
|
|
77
|
+
|
|
78
|
+
## Outputs
|
|
79
|
+
|
|
80
|
+
CSV Refine generates:
|
|
81
|
+
|
|
82
|
+
```text
|
|
83
|
+
output/
|
|
84
|
+
├── validated_example.csv
|
|
85
|
+
├── errors_example.csv
|
|
86
|
+
├── report_example.json
|
|
87
|
+
└── report_example.html
|
|
88
|
+
```
|
|
89
|
+
|
|
90
|
+
- **Validated CSV** — valid rows with transformations applied
|
|
91
|
+
- **Error CSV** — invalid rows with their detected errors
|
|
92
|
+
- **JSON report** — machine-readable summary
|
|
93
|
+
- **HTML report** — human-readable overview of CSV health
|
|
94
|
+
|
|
95
|
+
The original CSV is never modified.
|
|
96
|
+
|
|
97
|
+
## Real-world example
|
|
98
|
+
|
|
99
|
+
The repository includes a 200-row retail dataset containing missing values, invalid categories, malformed identifiers, duplicate values, invalid dates, and conversion errors.
|
|
100
|
+
|
|
101
|
+
```text
|
|
102
|
+
examples/
|
|
103
|
+
├── 01-input/
|
|
104
|
+
│ ├── example.csv
|
|
105
|
+
│ └── example-contract.yaml
|
|
106
|
+
└── 02-expected-output/
|
|
107
|
+
├── validated_example.csv
|
|
108
|
+
├── errors_example.csv
|
|
109
|
+
├── report_example.json
|
|
110
|
+
└── report_example.html
|
|
111
|
+
```
|
|
112
|
+
|
|
113
|
+
Run the example with:
|
|
114
|
+
|
|
115
|
+
```bash
|
|
116
|
+
csv-refine \
|
|
117
|
+
--contract examples/01-input/example-contract.yaml \
|
|
118
|
+
--csv examples/01-input/example.csv \
|
|
119
|
+
--mode permissive
|
|
120
|
+
```
|
|
121
|
+
|
|
122
|
+
Example result:
|
|
123
|
+
|
|
124
|
+
```text
|
|
125
|
+
Total rows: 200
|
|
126
|
+
Valid rows: 133
|
|
127
|
+
Invalid rows: 67
|
|
128
|
+
Total errors: 117
|
|
129
|
+
```
|
|
130
|
+
|
|
131
|
+
## Tests
|
|
132
|
+
|
|
133
|
+
Run the test suite with:
|
|
134
|
+
|
|
135
|
+
```bash
|
|
136
|
+
python3 -m pytest
|
|
137
|
+
```
|
|
138
|
+
|
|
139
|
+
## Stack
|
|
140
|
+
|
|
141
|
+
Python 3.12+, PyYAML, email-validator, pytest, argparse, pathlib, logging.
|
|
142
|
+
|
|
143
|
+
## Status
|
|
144
|
+
|
|
145
|
+
**Version 1 is functionally complete.**
|
|
@@ -0,0 +1,32 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["setuptools>=77.0.3"]
|
|
3
|
+
build-backend = "setuptools.build_meta"
|
|
4
|
+
|
|
5
|
+
|
|
6
|
+
[project]
|
|
7
|
+
name = "csv-refine"
|
|
8
|
+
version = "0.1.0"
|
|
9
|
+
description = "A Python CLI for cleaning, filtering, and normalizing CSV files using configurable YAML data contracts."
|
|
10
|
+
readme = "README.md"
|
|
11
|
+
requires-python = ">=3.12"
|
|
12
|
+
license = "MIT"
|
|
13
|
+
license-files = ["LICENSE"]
|
|
14
|
+
|
|
15
|
+
dependencies = [
|
|
16
|
+
"PyYAML>=6.0.3",
|
|
17
|
+
"email-validator>=2.3.0"
|
|
18
|
+
]
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
[project.scripts]
|
|
22
|
+
csv-refine = "csv_refine.cli:run_csv_refine"
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
[project.urls]
|
|
26
|
+
Homepage = "https://github.com/alexiskuypers/csv-refine"
|
|
27
|
+
Repository = "https://github.com/alexiskuypers/csv-refine"
|
|
28
|
+
Issues = "https://github.com/alexiskuypers/csv-refine/issues"
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
[project.optional-dependencies]
|
|
32
|
+
dev = ["pytest>=9.1.1"]
|
|
File without changes
|
|
@@ -0,0 +1,152 @@
|
|
|
1
|
+
import datetime
|
|
2
|
+
import logging
|
|
3
|
+
|
|
4
|
+
from csv_refine.contract_models import Contract
|
|
5
|
+
|
|
6
|
+
logger = logging.getLogger(__name__)
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
def count_total_errors(validated_classified_rows: dict) -> tuple:
|
|
10
|
+
"""Return the total error count and the number of rows with multiple errors."""
|
|
11
|
+
count_errors = 0
|
|
12
|
+
rows_with_multiple_errors = 0
|
|
13
|
+
|
|
14
|
+
invalid_rows = validated_classified_rows["invalid_rows"]
|
|
15
|
+
|
|
16
|
+
for invalid_row in invalid_rows:
|
|
17
|
+
count_errors += len(invalid_row["errors"])
|
|
18
|
+
if len(invalid_row["errors"]) > 1:
|
|
19
|
+
rows_with_multiple_errors += 1
|
|
20
|
+
|
|
21
|
+
return count_errors, rows_with_multiple_errors
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
def collect_errors(validated_classified_rows: dict):
|
|
25
|
+
"""Return the errors collected from all invalid rows."""
|
|
26
|
+
errors = []
|
|
27
|
+
invalid_rows = validated_classified_rows["invalid_rows"]
|
|
28
|
+
for invalid_row in invalid_rows:
|
|
29
|
+
errors.extend(invalid_row["errors"])
|
|
30
|
+
return errors
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
def count_rules_violations(errors) -> list[tuple]:
|
|
34
|
+
"""Count rule violations and return them sorted by frequency."""
|
|
35
|
+
rules_violations = {
|
|
36
|
+
"max_length": 0,
|
|
37
|
+
"min_length": 0,
|
|
38
|
+
"max": 0,
|
|
39
|
+
"min": 0,
|
|
40
|
+
"starts_with": 0,
|
|
41
|
+
"ends_with": 0,
|
|
42
|
+
"regex": 0,
|
|
43
|
+
"allowed_values": 0,
|
|
44
|
+
}
|
|
45
|
+
for error in errors:
|
|
46
|
+
if "length max" in error:
|
|
47
|
+
rules_violations["max_length"] = rules_violations["max_length"] + 1
|
|
48
|
+
if "length min" in error:
|
|
49
|
+
rules_violations["min_length"] = rules_violations["min_length"] + 1
|
|
50
|
+
if "minimum autorized" in error:
|
|
51
|
+
rules_violations["min"] = rules_violations["min"] + 1
|
|
52
|
+
if "maximum autorized" in error:
|
|
53
|
+
rules_violations["max"] = rules_violations["max"] + 1
|
|
54
|
+
if "regex" in error:
|
|
55
|
+
rules_violations["regex"] = rules_violations["regex"] + 1
|
|
56
|
+
if "startwith" in error:
|
|
57
|
+
rules_violations["starts_with"] = rules_violations["starts_with"] + 1
|
|
58
|
+
if "endswith" in error:
|
|
59
|
+
rules_violations["ends_with"] = rules_violations["ends_with"] + 1
|
|
60
|
+
if "allowed values" in error:
|
|
61
|
+
rules_violations["allowed_values"] = rules_violations["allowed_values"] + 1
|
|
62
|
+
|
|
63
|
+
return sorted(rules_violations.items(), key=lambda item: item[1], reverse=True)
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
def count_errors_by_category(errors: list, rules_violations: list[tuple]):
|
|
67
|
+
"""Return error counts grouped by category."""
|
|
68
|
+
count_errors_by_category = {
|
|
69
|
+
"rules_violations": sum(values[1] for values in rules_violations),
|
|
70
|
+
"conversion_errors": 0,
|
|
71
|
+
"unique_errors": 0,
|
|
72
|
+
"nullable_errors": 0,
|
|
73
|
+
}
|
|
74
|
+
for error in errors:
|
|
75
|
+
if "cannot be null" in error:
|
|
76
|
+
count_errors_by_category["nullable_errors"] = (
|
|
77
|
+
count_errors_by_category["nullable_errors"] + 1
|
|
78
|
+
)
|
|
79
|
+
if "converted or validated" in error:
|
|
80
|
+
count_errors_by_category["conversion_errors"] = (
|
|
81
|
+
count_errors_by_category["conversion_errors"] + 1
|
|
82
|
+
)
|
|
83
|
+
if "unique" in error:
|
|
84
|
+
count_errors_by_category["unique_errors"] = (
|
|
85
|
+
count_errors_by_category["unique_errors"] + 1
|
|
86
|
+
)
|
|
87
|
+
|
|
88
|
+
return count_errors_by_category
|
|
89
|
+
|
|
90
|
+
|
|
91
|
+
def build_report(
|
|
92
|
+
validated_classified_rows: dict,
|
|
93
|
+
mode: str,
|
|
94
|
+
contract: Contract,
|
|
95
|
+
csv_name: str,
|
|
96
|
+
duration: float,
|
|
97
|
+
):
|
|
98
|
+
"""Build the validation summary report."""
|
|
99
|
+
logger.info("Start building validation report.")
|
|
100
|
+
valid_rows = len(validated_classified_rows["valid_rows"])
|
|
101
|
+
invalid_rows = len(validated_classified_rows["invalid_rows"])
|
|
102
|
+
|
|
103
|
+
total_rows = valid_rows + invalid_rows
|
|
104
|
+
|
|
105
|
+
count_errors, rows_with_multiple_errors = count_total_errors(
|
|
106
|
+
validated_classified_rows
|
|
107
|
+
)
|
|
108
|
+
collected_errors = collect_errors(validated_classified_rows)
|
|
109
|
+
rules_violations = count_rules_violations(collected_errors)
|
|
110
|
+
errors_by_category = count_errors_by_category(collected_errors, rules_violations)
|
|
111
|
+
|
|
112
|
+
if total_rows == 0:
|
|
113
|
+
valid_rows_percentage = 0
|
|
114
|
+
else:
|
|
115
|
+
valid_rows_percentage = valid_rows / total_rows * 100
|
|
116
|
+
|
|
117
|
+
if invalid_rows == 0:
|
|
118
|
+
average_errors_per_invalid_row = 0
|
|
119
|
+
status = "passed"
|
|
120
|
+
else:
|
|
121
|
+
average_errors_per_invalid_row = count_errors / invalid_rows
|
|
122
|
+
status = "completed_with_errors"
|
|
123
|
+
|
|
124
|
+
report = {
|
|
125
|
+
"status": status,
|
|
126
|
+
"input_file": csv_name,
|
|
127
|
+
"mode": mode,
|
|
128
|
+
"generated_at": datetime.datetime.now().strftime("%Y-%m-%d %H:%M:%S"),
|
|
129
|
+
"columns_count": len(contract.headers),
|
|
130
|
+
"delimiter": contract.delimiter,
|
|
131
|
+
"encoding": contract.encoding,
|
|
132
|
+
"total_rows": valid_rows + invalid_rows,
|
|
133
|
+
"valid_rows": valid_rows,
|
|
134
|
+
"invalid_rows": invalid_rows,
|
|
135
|
+
"valid_rows_percentage": round(valid_rows_percentage, 2),
|
|
136
|
+
"total_errors": count_errors,
|
|
137
|
+
"rows_with_multiple_errors": rows_with_multiple_errors,
|
|
138
|
+
"average_errors_per_invalid_row": round(average_errors_per_invalid_row, 2),
|
|
139
|
+
"rules_violations": dict(rules_violations),
|
|
140
|
+
"errors_by_category": errors_by_category,
|
|
141
|
+
"processing_duration_ms": round(duration, 2),
|
|
142
|
+
}
|
|
143
|
+
logger.info(
|
|
144
|
+
logger.info(
|
|
145
|
+
f"Validation report built successfully: "
|
|
146
|
+
f"status={status}, "
|
|
147
|
+
f"valid_rows={valid_rows}, "
|
|
148
|
+
f"invalid_rows={invalid_rows}, "
|
|
149
|
+
f"total_errors={count_errors}"
|
|
150
|
+
)
|
|
151
|
+
)
|
|
152
|
+
return report
|
|
@@ -0,0 +1,80 @@
|
|
|
1
|
+
import argparse
|
|
2
|
+
import logging
|
|
3
|
+
import webbrowser
|
|
4
|
+
from pathlib import Path
|
|
5
|
+
|
|
6
|
+
from csv_refine.exceptions import YAMLContractError, CSVError
|
|
7
|
+
from csv_refine.logging_config import configure_logging
|
|
8
|
+
from csv_refine.orchestration import orchestration
|
|
9
|
+
|
|
10
|
+
logger = logging.getLogger(__name__)
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
def create_parser() -> argparse.ArgumentParser:
|
|
14
|
+
"""Create and configure the CLI argument parser."""
|
|
15
|
+
parser = argparse.ArgumentParser(
|
|
16
|
+
description="A Python CLI for validating and normalizing CSV files using YAML data contracts.",
|
|
17
|
+
formatter_class=argparse.ArgumentDefaultsHelpFormatter,
|
|
18
|
+
)
|
|
19
|
+
parser.add_argument(
|
|
20
|
+
"--contract", type=Path, required=True, help="Path to the YAML contract."
|
|
21
|
+
)
|
|
22
|
+
|
|
23
|
+
parser.add_argument(
|
|
24
|
+
"--csv", type=Path, required=True, help="Path to the CSV file to process."
|
|
25
|
+
)
|
|
26
|
+
|
|
27
|
+
parser.add_argument(
|
|
28
|
+
"--output",
|
|
29
|
+
help="Output directory.",
|
|
30
|
+
default="output",
|
|
31
|
+
)
|
|
32
|
+
parser.add_argument(
|
|
33
|
+
"--mode",
|
|
34
|
+
type=str,
|
|
35
|
+
choices=["strict", "permissive"],
|
|
36
|
+
help="Validation mode.",
|
|
37
|
+
default="permissive",
|
|
38
|
+
)
|
|
39
|
+
return parser
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
def run_csv_refine() -> int:
|
|
43
|
+
"""Run the CSV Refine CLI and return the appropriate exit code."""
|
|
44
|
+
configure_logging()
|
|
45
|
+
parser = create_parser()
|
|
46
|
+
args = parser.parse_args()
|
|
47
|
+
|
|
48
|
+
try:
|
|
49
|
+
report, html_output = orchestration(
|
|
50
|
+
yaml_contract_path=args.contract,
|
|
51
|
+
csv_path=args.csv,
|
|
52
|
+
mode=args.mode,
|
|
53
|
+
output=Path(args.output),
|
|
54
|
+
)
|
|
55
|
+
|
|
56
|
+
except YAMLContractError as error:
|
|
57
|
+
logger.error(f"Error(s) caused by contract, error(s): '{error}'.")
|
|
58
|
+
return 1
|
|
59
|
+
|
|
60
|
+
except CSVError as error:
|
|
61
|
+
logger.error(f"Error(s) caused by csv, error(s): '{error}'.")
|
|
62
|
+
return 2
|
|
63
|
+
|
|
64
|
+
webbrowser.open(html_output.resolve().as_uri())
|
|
65
|
+
print_summary(report=report)
|
|
66
|
+
return 0
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
def print_summary(report: dict) -> None:
|
|
70
|
+
"""Print a concise processing summary for the user."""
|
|
71
|
+
print(
|
|
72
|
+
f"Columns counted: {report['columns_count']}\n"
|
|
73
|
+
f"Total rows: {report['total_rows']}\n"
|
|
74
|
+
f"Total valid rows: {report['valid_rows']}\n"
|
|
75
|
+
f"Errors: {report['total_errors']}."
|
|
76
|
+
)
|
|
77
|
+
|
|
78
|
+
|
|
79
|
+
if __name__ == "__main__":
|
|
80
|
+
raise SystemExit(run_csv_refine())
|