csv-refine 0.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (38) hide show
  1. csv_refine-0.1.0/LICENSE +20 -0
  2. csv_refine-0.1.0/PKG-INFO +162 -0
  3. csv_refine-0.1.0/README.md +145 -0
  4. csv_refine-0.1.0/pyproject.toml +32 -0
  5. csv_refine-0.1.0/setup.cfg +4 -0
  6. csv_refine-0.1.0/src/csv_refine/__init__.py +0 -0
  7. csv_refine-0.1.0/src/csv_refine/build_summary.py +152 -0
  8. csv_refine-0.1.0/src/csv_refine/cli.py +80 -0
  9. csv_refine-0.1.0/src/csv_refine/contract.py +571 -0
  10. csv_refine-0.1.0/src/csv_refine/contract_models.py +81 -0
  11. csv_refine-0.1.0/src/csv_refine/exceptions.py +10 -0
  12. csv_refine-0.1.0/src/csv_refine/logging_config.py +19 -0
  13. csv_refine-0.1.0/src/csv_refine/orchestration.py +67 -0
  14. csv_refine-0.1.0/src/csv_refine/row_processing.py +314 -0
  15. csv_refine-0.1.0/src/csv_refine/validate_csv.py +145 -0
  16. csv_refine-0.1.0/src/csv_refine/validate_unique.py +155 -0
  17. csv_refine-0.1.0/src/csv_refine/value_operation.py +220 -0
  18. csv_refine-0.1.0/src/csv_refine/write_csv.py +60 -0
  19. csv_refine-0.1.0/src/csv_refine/write_errors_csv.py +77 -0
  20. csv_refine-0.1.0/src/csv_refine/write_report.py +189 -0
  21. csv_refine-0.1.0/src/csv_refine.egg-info/PKG-INFO +162 -0
  22. csv_refine-0.1.0/src/csv_refine.egg-info/SOURCES.txt +36 -0
  23. csv_refine-0.1.0/src/csv_refine.egg-info/dependency_links.txt +1 -0
  24. csv_refine-0.1.0/src/csv_refine.egg-info/entry_points.txt +2 -0
  25. csv_refine-0.1.0/src/csv_refine.egg-info/requires.txt +5 -0
  26. csv_refine-0.1.0/src/csv_refine.egg-info/top_level.txt +1 -0
  27. csv_refine-0.1.0/tests/test_build_summary.py +260 -0
  28. csv_refine-0.1.0/tests/test_cli.py +44 -0
  29. csv_refine-0.1.0/tests/test_contract_models.py +120 -0
  30. csv_refine-0.1.0/tests/test_csv_row_processing.py +367 -0
  31. csv_refine-0.1.0/tests/test_logging_config.py +22 -0
  32. csv_refine-0.1.0/tests/test_orchestration.py +145 -0
  33. csv_refine-0.1.0/tests/test_validate_csv.py +211 -0
  34. csv_refine-0.1.0/tests/test_validate_unique.py +492 -0
  35. csv_refine-0.1.0/tests/test_value_operation.py +430 -0
  36. csv_refine-0.1.0/tests/test_write_csv.py +107 -0
  37. csv_refine-0.1.0/tests/test_write_errors_csv.py +170 -0
  38. csv_refine-0.1.0/tests/test_write_report.py +122 -0
@@ -0,0 +1,20 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Kuypers Alexis
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software without limitation the rights
7
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
8
+ copies of the Software, and to permit persons to whom the Software is
9
+ furnished to do so, subject to the following conditions:
10
+
11
+ The above copyright notice and this permission notice shall be included in all
12
+ copies or substantial portions of the Software.
13
+
14
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
15
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
16
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
17
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
18
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
19
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
20
+ SOFTWARE.
@@ -0,0 +1,162 @@
1
+ Metadata-Version: 2.4
2
+ Name: csv-refine
3
+ Version: 0.1.0
4
+ Summary: A Python CLI for cleaning, filtering, and normalizing CSV files using configurable YAML data contracts.
5
+ License-Expression: MIT
6
+ Project-URL: Homepage, https://github.com/alexiskuypers/csv-refine
7
+ Project-URL: Repository, https://github.com/alexiskuypers/csv-refine
8
+ Project-URL: Issues, https://github.com/alexiskuypers/csv-refine/issues
9
+ Requires-Python: >=3.12
10
+ Description-Content-Type: text/markdown
11
+ License-File: LICENSE
12
+ Requires-Dist: PyYAML>=6.0.3
13
+ Requires-Dist: email-validator>=2.3.0
14
+ Provides-Extra: dev
15
+ Requires-Dist: pytest>=9.1.1; extra == "dev"
16
+ Dynamic: license-file
17
+
18
+ # CSV Refine
19
+
20
+ CSV Refine is a Python CLI for cleaning, filtering, and normalizing CSV files.
21
+
22
+ It lets you define exactly how each column should be validated, cleaned, and normalized through a YAML contract. CSV Refine then separates valid and invalid rows and generates reports to assess the overall health of the dataset.
23
+
24
+ ## Features
25
+
26
+ - Clean and normalize messy CSV data
27
+ - Filter out invalid rows while preserving valid data
28
+ - Detect missing values, duplicates, invalid formats, and rule violations
29
+ - Separate valid and invalid rows into dedicated CSV files
30
+ - Generate JSON and HTML reports to assess CSV data quality
31
+
32
+ ## Installation
33
+
34
+ ```bash
35
+ git clone https://github.com/alexiskuypers/csv-refine.git
36
+ cd csv-refine
37
+
38
+ python3 -m venv .venv
39
+ source .venv/bin/activate
40
+
41
+ pip install -e .
42
+ ```
43
+
44
+ Check the CLI:
45
+
46
+ ```bash
47
+ csv-refine --help
48
+ ```
49
+
50
+ ## Quick start
51
+
52
+ ```bash
53
+ csv-refine \
54
+ --contract examples/01-input/example-contract.yaml \
55
+ --csv examples/01-input/example.csv \
56
+ --mode permissive
57
+ ```
58
+
59
+ By default, generated files are written to `output/`.
60
+
61
+ A custom output directory can be provided with `--output`.
62
+
63
+ ## YAML contract
64
+
65
+ CSV Refine is configured through a YAML contract that defines exactly how each column should be validated and normalized.
66
+
67
+ A contract can define data types, nullability, uniqueness, validation rules, transformations, delimiter, and encoding.
68
+
69
+ Example:
70
+
71
+ ```yaml
72
+ columns:
73
+ Transaction ID:
74
+ type: str
75
+ nullable: false
76
+ unique: true
77
+ rules:
78
+ starts_with: "TXN_"
79
+ ```
80
+
81
+ For the complete contract syntax and all supported options, see the [contract README](contracts/README.md).
82
+
83
+ ## Validation modes
84
+
85
+ ### Permissive
86
+
87
+ Invalid rows are collected while processing continues.
88
+
89
+ ### Strict
90
+
91
+ Structural, type conversion, nullability, and uniqueness failures can stop processing.
92
+
93
+ Validation-rule violations still act as row filters.
94
+
95
+ ## Outputs
96
+
97
+ CSV Refine generates:
98
+
99
+ ```text
100
+ output/
101
+ ├── validated_example.csv
102
+ ├── errors_example.csv
103
+ ├── report_example.json
104
+ └── report_example.html
105
+ ```
106
+
107
+ - **Validated CSV** — valid rows with transformations applied
108
+ - **Error CSV** — invalid rows with their detected errors
109
+ - **JSON report** — machine-readable summary
110
+ - **HTML report** — human-readable overview of CSV health
111
+
112
+ The original CSV is never modified.
113
+
114
+ ## Real-world example
115
+
116
+ The repository includes a 200-row retail dataset containing missing values, invalid categories, malformed identifiers, duplicate values, invalid dates, and conversion errors.
117
+
118
+ ```text
119
+ examples/
120
+ ├── 01-input/
121
+ │ ├── example.csv
122
+ │ └── example-contract.yaml
123
+ └── 02-expected-output/
124
+ ├── validated_example.csv
125
+ ├── errors_example.csv
126
+ ├── report_example.json
127
+ └── report_example.html
128
+ ```
129
+
130
+ Run the example with:
131
+
132
+ ```bash
133
+ csv-refine \
134
+ --contract examples/01-input/example-contract.yaml \
135
+ --csv examples/01-input/example.csv \
136
+ --mode permissive
137
+ ```
138
+
139
+ Example result:
140
+
141
+ ```text
142
+ Total rows: 200
143
+ Valid rows: 133
144
+ Invalid rows: 67
145
+ Total errors: 117
146
+ ```
147
+
148
+ ## Tests
149
+
150
+ Run the test suite with:
151
+
152
+ ```bash
153
+ python3 -m pytest
154
+ ```
155
+
156
+ ## Stack
157
+
158
+ Python 3.12+, PyYAML, email-validator, pytest, argparse, pathlib, logging.
159
+
160
+ ## Status
161
+
162
+ **Version 1 is functionally complete.**
@@ -0,0 +1,145 @@
1
+ # CSV Refine
2
+
3
+ CSV Refine is a Python CLI for cleaning, filtering, and normalizing CSV files.
4
+
5
+ It lets you define exactly how each column should be validated, cleaned, and normalized through a YAML contract. CSV Refine then separates valid and invalid rows and generates reports to assess the overall health of the dataset.
6
+
7
+ ## Features
8
+
9
+ - Clean and normalize messy CSV data
10
+ - Filter out invalid rows while preserving valid data
11
+ - Detect missing values, duplicates, invalid formats, and rule violations
12
+ - Separate valid and invalid rows into dedicated CSV files
13
+ - Generate JSON and HTML reports to assess CSV data quality
14
+
15
+ ## Installation
16
+
17
+ ```bash
18
+ git clone https://github.com/alexiskuypers/csv-refine.git
19
+ cd csv-refine
20
+
21
+ python3 -m venv .venv
22
+ source .venv/bin/activate
23
+
24
+ pip install -e .
25
+ ```
26
+
27
+ Check the CLI:
28
+
29
+ ```bash
30
+ csv-refine --help
31
+ ```
32
+
33
+ ## Quick start
34
+
35
+ ```bash
36
+ csv-refine \
37
+ --contract examples/01-input/example-contract.yaml \
38
+ --csv examples/01-input/example.csv \
39
+ --mode permissive
40
+ ```
41
+
42
+ By default, generated files are written to `output/`.
43
+
44
+ A custom output directory can be provided with `--output`.
45
+
46
+ ## YAML contract
47
+
48
+ CSV Refine is configured through a YAML contract that defines exactly how each column should be validated and normalized.
49
+
50
+ A contract can define data types, nullability, uniqueness, validation rules, transformations, delimiter, and encoding.
51
+
52
+ Example:
53
+
54
+ ```yaml
55
+ columns:
56
+ Transaction ID:
57
+ type: str
58
+ nullable: false
59
+ unique: true
60
+ rules:
61
+ starts_with: "TXN_"
62
+ ```
63
+
64
+ For the complete contract syntax and all supported options, see the [contract README](contracts/README.md).
65
+
66
+ ## Validation modes
67
+
68
+ ### Permissive
69
+
70
+ Invalid rows are collected while processing continues.
71
+
72
+ ### Strict
73
+
74
+ Structural, type conversion, nullability, and uniqueness failures can stop processing.
75
+
76
+ Validation-rule violations still act as row filters.
77
+
78
+ ## Outputs
79
+
80
+ CSV Refine generates:
81
+
82
+ ```text
83
+ output/
84
+ ├── validated_example.csv
85
+ ├── errors_example.csv
86
+ ├── report_example.json
87
+ └── report_example.html
88
+ ```
89
+
90
+ - **Validated CSV** — valid rows with transformations applied
91
+ - **Error CSV** — invalid rows with their detected errors
92
+ - **JSON report** — machine-readable summary
93
+ - **HTML report** — human-readable overview of CSV health
94
+
95
+ The original CSV is never modified.
96
+
97
+ ## Real-world example
98
+
99
+ The repository includes a 200-row retail dataset containing missing values, invalid categories, malformed identifiers, duplicate values, invalid dates, and conversion errors.
100
+
101
+ ```text
102
+ examples/
103
+ ├── 01-input/
104
+ │ ├── example.csv
105
+ │ └── example-contract.yaml
106
+ └── 02-expected-output/
107
+ ├── validated_example.csv
108
+ ├── errors_example.csv
109
+ ├── report_example.json
110
+ └── report_example.html
111
+ ```
112
+
113
+ Run the example with:
114
+
115
+ ```bash
116
+ csv-refine \
117
+ --contract examples/01-input/example-contract.yaml \
118
+ --csv examples/01-input/example.csv \
119
+ --mode permissive
120
+ ```
121
+
122
+ Example result:
123
+
124
+ ```text
125
+ Total rows: 200
126
+ Valid rows: 133
127
+ Invalid rows: 67
128
+ Total errors: 117
129
+ ```
130
+
131
+ ## Tests
132
+
133
+ Run the test suite with:
134
+
135
+ ```bash
136
+ python3 -m pytest
137
+ ```
138
+
139
+ ## Stack
140
+
141
+ Python 3.12+, PyYAML, email-validator, pytest, argparse, pathlib, logging.
142
+
143
+ ## Status
144
+
145
+ **Version 1 is functionally complete.**
@@ -0,0 +1,32 @@
1
+ [build-system]
2
+ requires = ["setuptools>=77.0.3"]
3
+ build-backend = "setuptools.build_meta"
4
+
5
+
6
+ [project]
7
+ name = "csv-refine"
8
+ version = "0.1.0"
9
+ description = "A Python CLI for cleaning, filtering, and normalizing CSV files using configurable YAML data contracts."
10
+ readme = "README.md"
11
+ requires-python = ">=3.12"
12
+ license = "MIT"
13
+ license-files = ["LICENSE"]
14
+
15
+ dependencies = [
16
+ "PyYAML>=6.0.3",
17
+ "email-validator>=2.3.0"
18
+ ]
19
+
20
+
21
+ [project.scripts]
22
+ csv-refine = "csv_refine.cli:run_csv_refine"
23
+
24
+
25
+ [project.urls]
26
+ Homepage = "https://github.com/alexiskuypers/csv-refine"
27
+ Repository = "https://github.com/alexiskuypers/csv-refine"
28
+ Issues = "https://github.com/alexiskuypers/csv-refine/issues"
29
+
30
+
31
+ [project.optional-dependencies]
32
+ dev = ["pytest>=9.1.1"]
@@ -0,0 +1,4 @@
1
+ [egg_info]
2
+ tag_build =
3
+ tag_date = 0
4
+
File without changes
@@ -0,0 +1,152 @@
1
+ import datetime
2
+ import logging
3
+
4
+ from csv_refine.contract_models import Contract
5
+
6
+ logger = logging.getLogger(__name__)
7
+
8
+
9
+ def count_total_errors(validated_classified_rows: dict) -> tuple:
10
+ """Return the total error count and the number of rows with multiple errors."""
11
+ count_errors = 0
12
+ rows_with_multiple_errors = 0
13
+
14
+ invalid_rows = validated_classified_rows["invalid_rows"]
15
+
16
+ for invalid_row in invalid_rows:
17
+ count_errors += len(invalid_row["errors"])
18
+ if len(invalid_row["errors"]) > 1:
19
+ rows_with_multiple_errors += 1
20
+
21
+ return count_errors, rows_with_multiple_errors
22
+
23
+
24
+ def collect_errors(validated_classified_rows: dict):
25
+ """Return the errors collected from all invalid rows."""
26
+ errors = []
27
+ invalid_rows = validated_classified_rows["invalid_rows"]
28
+ for invalid_row in invalid_rows:
29
+ errors.extend(invalid_row["errors"])
30
+ return errors
31
+
32
+
33
+ def count_rules_violations(errors) -> list[tuple]:
34
+ """Count rule violations and return them sorted by frequency."""
35
+ rules_violations = {
36
+ "max_length": 0,
37
+ "min_length": 0,
38
+ "max": 0,
39
+ "min": 0,
40
+ "starts_with": 0,
41
+ "ends_with": 0,
42
+ "regex": 0,
43
+ "allowed_values": 0,
44
+ }
45
+ for error in errors:
46
+ if "length max" in error:
47
+ rules_violations["max_length"] = rules_violations["max_length"] + 1
48
+ if "length min" in error:
49
+ rules_violations["min_length"] = rules_violations["min_length"] + 1
50
+ if "minimum autorized" in error:
51
+ rules_violations["min"] = rules_violations["min"] + 1
52
+ if "maximum autorized" in error:
53
+ rules_violations["max"] = rules_violations["max"] + 1
54
+ if "regex" in error:
55
+ rules_violations["regex"] = rules_violations["regex"] + 1
56
+ if "startwith" in error:
57
+ rules_violations["starts_with"] = rules_violations["starts_with"] + 1
58
+ if "endswith" in error:
59
+ rules_violations["ends_with"] = rules_violations["ends_with"] + 1
60
+ if "allowed values" in error:
61
+ rules_violations["allowed_values"] = rules_violations["allowed_values"] + 1
62
+
63
+ return sorted(rules_violations.items(), key=lambda item: item[1], reverse=True)
64
+
65
+
66
+ def count_errors_by_category(errors: list, rules_violations: list[tuple]):
67
+ """Return error counts grouped by category."""
68
+ count_errors_by_category = {
69
+ "rules_violations": sum(values[1] for values in rules_violations),
70
+ "conversion_errors": 0,
71
+ "unique_errors": 0,
72
+ "nullable_errors": 0,
73
+ }
74
+ for error in errors:
75
+ if "cannot be null" in error:
76
+ count_errors_by_category["nullable_errors"] = (
77
+ count_errors_by_category["nullable_errors"] + 1
78
+ )
79
+ if "converted or validated" in error:
80
+ count_errors_by_category["conversion_errors"] = (
81
+ count_errors_by_category["conversion_errors"] + 1
82
+ )
83
+ if "unique" in error:
84
+ count_errors_by_category["unique_errors"] = (
85
+ count_errors_by_category["unique_errors"] + 1
86
+ )
87
+
88
+ return count_errors_by_category
89
+
90
+
91
+ def build_report(
92
+ validated_classified_rows: dict,
93
+ mode: str,
94
+ contract: Contract,
95
+ csv_name: str,
96
+ duration: float,
97
+ ):
98
+ """Build the validation summary report."""
99
+ logger.info("Start building validation report.")
100
+ valid_rows = len(validated_classified_rows["valid_rows"])
101
+ invalid_rows = len(validated_classified_rows["invalid_rows"])
102
+
103
+ total_rows = valid_rows + invalid_rows
104
+
105
+ count_errors, rows_with_multiple_errors = count_total_errors(
106
+ validated_classified_rows
107
+ )
108
+ collected_errors = collect_errors(validated_classified_rows)
109
+ rules_violations = count_rules_violations(collected_errors)
110
+ errors_by_category = count_errors_by_category(collected_errors, rules_violations)
111
+
112
+ if total_rows == 0:
113
+ valid_rows_percentage = 0
114
+ else:
115
+ valid_rows_percentage = valid_rows / total_rows * 100
116
+
117
+ if invalid_rows == 0:
118
+ average_errors_per_invalid_row = 0
119
+ status = "passed"
120
+ else:
121
+ average_errors_per_invalid_row = count_errors / invalid_rows
122
+ status = "completed_with_errors"
123
+
124
+ report = {
125
+ "status": status,
126
+ "input_file": csv_name,
127
+ "mode": mode,
128
+ "generated_at": datetime.datetime.now().strftime("%Y-%m-%d %H:%M:%S"),
129
+ "columns_count": len(contract.headers),
130
+ "delimiter": contract.delimiter,
131
+ "encoding": contract.encoding,
132
+ "total_rows": valid_rows + invalid_rows,
133
+ "valid_rows": valid_rows,
134
+ "invalid_rows": invalid_rows,
135
+ "valid_rows_percentage": round(valid_rows_percentage, 2),
136
+ "total_errors": count_errors,
137
+ "rows_with_multiple_errors": rows_with_multiple_errors,
138
+ "average_errors_per_invalid_row": round(average_errors_per_invalid_row, 2),
139
+ "rules_violations": dict(rules_violations),
140
+ "errors_by_category": errors_by_category,
141
+ "processing_duration_ms": round(duration, 2),
142
+ }
143
+ logger.info(
144
+ logger.info(
145
+ f"Validation report built successfully: "
146
+ f"status={status}, "
147
+ f"valid_rows={valid_rows}, "
148
+ f"invalid_rows={invalid_rows}, "
149
+ f"total_errors={count_errors}"
150
+ )
151
+ )
152
+ return report
@@ -0,0 +1,80 @@
1
+ import argparse
2
+ import logging
3
+ import webbrowser
4
+ from pathlib import Path
5
+
6
+ from csv_refine.exceptions import YAMLContractError, CSVError
7
+ from csv_refine.logging_config import configure_logging
8
+ from csv_refine.orchestration import orchestration
9
+
10
+ logger = logging.getLogger(__name__)
11
+
12
+
13
+ def create_parser() -> argparse.ArgumentParser:
14
+ """Create and configure the CLI argument parser."""
15
+ parser = argparse.ArgumentParser(
16
+ description="A Python CLI for validating and normalizing CSV files using YAML data contracts.",
17
+ formatter_class=argparse.ArgumentDefaultsHelpFormatter,
18
+ )
19
+ parser.add_argument(
20
+ "--contract", type=Path, required=True, help="Path to the YAML contract."
21
+ )
22
+
23
+ parser.add_argument(
24
+ "--csv", type=Path, required=True, help="Path to the CSV file to process."
25
+ )
26
+
27
+ parser.add_argument(
28
+ "--output",
29
+ help="Output directory.",
30
+ default="output",
31
+ )
32
+ parser.add_argument(
33
+ "--mode",
34
+ type=str,
35
+ choices=["strict", "permissive"],
36
+ help="Validation mode.",
37
+ default="permissive",
38
+ )
39
+ return parser
40
+
41
+
42
+ def run_csv_refine() -> int:
43
+ """Run the CSV Refine CLI and return the appropriate exit code."""
44
+ configure_logging()
45
+ parser = create_parser()
46
+ args = parser.parse_args()
47
+
48
+ try:
49
+ report, html_output = orchestration(
50
+ yaml_contract_path=args.contract,
51
+ csv_path=args.csv,
52
+ mode=args.mode,
53
+ output=Path(args.output),
54
+ )
55
+
56
+ except YAMLContractError as error:
57
+ logger.error(f"Error(s) caused by contract, error(s): '{error}'.")
58
+ return 1
59
+
60
+ except CSVError as error:
61
+ logger.error(f"Error(s) caused by csv, error(s): '{error}'.")
62
+ return 2
63
+
64
+ webbrowser.open(html_output.resolve().as_uri())
65
+ print_summary(report=report)
66
+ return 0
67
+
68
+
69
+ def print_summary(report: dict) -> None:
70
+ """Print a concise processing summary for the user."""
71
+ print(
72
+ f"Columns counted: {report['columns_count']}\n"
73
+ f"Total rows: {report['total_rows']}\n"
74
+ f"Total valid rows: {report['valid_rows']}\n"
75
+ f"Errors: {report['total_errors']}."
76
+ )
77
+
78
+
79
+ if __name__ == "__main__":
80
+ raise SystemExit(run_csv_refine())