csv-refine 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
csv_refine/__init__.py ADDED
File without changes
@@ -0,0 +1,152 @@
1
+ import datetime
2
+ import logging
3
+
4
+ from csv_refine.contract_models import Contract
5
+
6
+ logger = logging.getLogger(__name__)
7
+
8
+
9
+ def count_total_errors(validated_classified_rows: dict) -> tuple:
10
+ """Return the total error count and the number of rows with multiple errors."""
11
+ count_errors = 0
12
+ rows_with_multiple_errors = 0
13
+
14
+ invalid_rows = validated_classified_rows["invalid_rows"]
15
+
16
+ for invalid_row in invalid_rows:
17
+ count_errors += len(invalid_row["errors"])
18
+ if len(invalid_row["errors"]) > 1:
19
+ rows_with_multiple_errors += 1
20
+
21
+ return count_errors, rows_with_multiple_errors
22
+
23
+
24
+ def collect_errors(validated_classified_rows: dict):
25
+ """Return the errors collected from all invalid rows."""
26
+ errors = []
27
+ invalid_rows = validated_classified_rows["invalid_rows"]
28
+ for invalid_row in invalid_rows:
29
+ errors.extend(invalid_row["errors"])
30
+ return errors
31
+
32
+
33
+ def count_rules_violations(errors) -> list[tuple]:
34
+ """Count rule violations and return them sorted by frequency."""
35
+ rules_violations = {
36
+ "max_length": 0,
37
+ "min_length": 0,
38
+ "max": 0,
39
+ "min": 0,
40
+ "starts_with": 0,
41
+ "ends_with": 0,
42
+ "regex": 0,
43
+ "allowed_values": 0,
44
+ }
45
+ for error in errors:
46
+ if "length max" in error:
47
+ rules_violations["max_length"] = rules_violations["max_length"] + 1
48
+ if "length min" in error:
49
+ rules_violations["min_length"] = rules_violations["min_length"] + 1
50
+ if "minimum autorized" in error:
51
+ rules_violations["min"] = rules_violations["min"] + 1
52
+ if "maximum autorized" in error:
53
+ rules_violations["max"] = rules_violations["max"] + 1
54
+ if "regex" in error:
55
+ rules_violations["regex"] = rules_violations["regex"] + 1
56
+ if "startwith" in error:
57
+ rules_violations["starts_with"] = rules_violations["starts_with"] + 1
58
+ if "endswith" in error:
59
+ rules_violations["ends_with"] = rules_violations["ends_with"] + 1
60
+ if "allowed values" in error:
61
+ rules_violations["allowed_values"] = rules_violations["allowed_values"] + 1
62
+
63
+ return sorted(rules_violations.items(), key=lambda item: item[1], reverse=True)
64
+
65
+
66
+ def count_errors_by_category(errors: list, rules_violations: list[tuple]):
67
+ """Return error counts grouped by category."""
68
+ count_errors_by_category = {
69
+ "rules_violations": sum(values[1] for values in rules_violations),
70
+ "conversion_errors": 0,
71
+ "unique_errors": 0,
72
+ "nullable_errors": 0,
73
+ }
74
+ for error in errors:
75
+ if "cannot be null" in error:
76
+ count_errors_by_category["nullable_errors"] = (
77
+ count_errors_by_category["nullable_errors"] + 1
78
+ )
79
+ if "converted or validated" in error:
80
+ count_errors_by_category["conversion_errors"] = (
81
+ count_errors_by_category["conversion_errors"] + 1
82
+ )
83
+ if "unique" in error:
84
+ count_errors_by_category["unique_errors"] = (
85
+ count_errors_by_category["unique_errors"] + 1
86
+ )
87
+
88
+ return count_errors_by_category
89
+
90
+
91
+ def build_report(
92
+ validated_classified_rows: dict,
93
+ mode: str,
94
+ contract: Contract,
95
+ csv_name: str,
96
+ duration: float,
97
+ ):
98
+ """Build the validation summary report."""
99
+ logger.info("Start building validation report.")
100
+ valid_rows = len(validated_classified_rows["valid_rows"])
101
+ invalid_rows = len(validated_classified_rows["invalid_rows"])
102
+
103
+ total_rows = valid_rows + invalid_rows
104
+
105
+ count_errors, rows_with_multiple_errors = count_total_errors(
106
+ validated_classified_rows
107
+ )
108
+ collected_errors = collect_errors(validated_classified_rows)
109
+ rules_violations = count_rules_violations(collected_errors)
110
+ errors_by_category = count_errors_by_category(collected_errors, rules_violations)
111
+
112
+ if total_rows == 0:
113
+ valid_rows_percentage = 0
114
+ else:
115
+ valid_rows_percentage = valid_rows / total_rows * 100
116
+
117
+ if invalid_rows == 0:
118
+ average_errors_per_invalid_row = 0
119
+ status = "passed"
120
+ else:
121
+ average_errors_per_invalid_row = count_errors / invalid_rows
122
+ status = "completed_with_errors"
123
+
124
+ report = {
125
+ "status": status,
126
+ "input_file": csv_name,
127
+ "mode": mode,
128
+ "generated_at": datetime.datetime.now().strftime("%Y-%m-%d %H:%M:%S"),
129
+ "columns_count": len(contract.headers),
130
+ "delimiter": contract.delimiter,
131
+ "encoding": contract.encoding,
132
+ "total_rows": valid_rows + invalid_rows,
133
+ "valid_rows": valid_rows,
134
+ "invalid_rows": invalid_rows,
135
+ "valid_rows_percentage": round(valid_rows_percentage, 2),
136
+ "total_errors": count_errors,
137
+ "rows_with_multiple_errors": rows_with_multiple_errors,
138
+ "average_errors_per_invalid_row": round(average_errors_per_invalid_row, 2),
139
+ "rules_violations": dict(rules_violations),
140
+ "errors_by_category": errors_by_category,
141
+ "processing_duration_ms": round(duration, 2),
142
+ }
143
+ logger.info(
144
+ logger.info(
145
+ f"Validation report built successfully: "
146
+ f"status={status}, "
147
+ f"valid_rows={valid_rows}, "
148
+ f"invalid_rows={invalid_rows}, "
149
+ f"total_errors={count_errors}"
150
+ )
151
+ )
152
+ return report
csv_refine/cli.py ADDED
@@ -0,0 +1,80 @@
1
+ import argparse
2
+ import logging
3
+ import webbrowser
4
+ from pathlib import Path
5
+
6
+ from csv_refine.exceptions import YAMLContractError, CSVError
7
+ from csv_refine.logging_config import configure_logging
8
+ from csv_refine.orchestration import orchestration
9
+
10
+ logger = logging.getLogger(__name__)
11
+
12
+
13
+ def create_parser() -> argparse.ArgumentParser:
14
+ """Create and configure the CLI argument parser."""
15
+ parser = argparse.ArgumentParser(
16
+ description="A Python CLI for validating and normalizing CSV files using YAML data contracts.",
17
+ formatter_class=argparse.ArgumentDefaultsHelpFormatter,
18
+ )
19
+ parser.add_argument(
20
+ "--contract", type=Path, required=True, help="Path to the YAML contract."
21
+ )
22
+
23
+ parser.add_argument(
24
+ "--csv", type=Path, required=True, help="Path to the CSV file to process."
25
+ )
26
+
27
+ parser.add_argument(
28
+ "--output",
29
+ help="Output directory.",
30
+ default="output",
31
+ )
32
+ parser.add_argument(
33
+ "--mode",
34
+ type=str,
35
+ choices=["strict", "permissive"],
36
+ help="Validation mode.",
37
+ default="permissive",
38
+ )
39
+ return parser
40
+
41
+
42
+ def run_csv_refine() -> int:
43
+ """Run the CSV Refine CLI and return the appropriate exit code."""
44
+ configure_logging()
45
+ parser = create_parser()
46
+ args = parser.parse_args()
47
+
48
+ try:
49
+ report, html_output = orchestration(
50
+ yaml_contract_path=args.contract,
51
+ csv_path=args.csv,
52
+ mode=args.mode,
53
+ output=Path(args.output),
54
+ )
55
+
56
+ except YAMLContractError as error:
57
+ logger.error(f"Error(s) caused by contract, error(s): '{error}'.")
58
+ return 1
59
+
60
+ except CSVError as error:
61
+ logger.error(f"Error(s) caused by csv, error(s): '{error}'.")
62
+ return 2
63
+
64
+ webbrowser.open(html_output.resolve().as_uri())
65
+ print_summary(report=report)
66
+ return 0
67
+
68
+
69
+ def print_summary(report: dict) -> None:
70
+ """Print a concise processing summary for the user."""
71
+ print(
72
+ f"Columns counted: {report['columns_count']}\n"
73
+ f"Total rows: {report['total_rows']}\n"
74
+ f"Total valid rows: {report['valid_rows']}\n"
75
+ f"Errors: {report['total_errors']}."
76
+ )
77
+
78
+
79
+ if __name__ == "__main__":
80
+ raise SystemExit(run_csv_refine())