csvsmith 0.7.2__tar.gz → 0.8.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {csvsmith-0.7.2/src/csvsmith.egg-info → csvsmith-0.8.0}/PKG-INFO +26 -11
- {csvsmith-0.7.2 → csvsmith-0.8.0}/README.rst +22 -10
- {csvsmith-0.7.2 → csvsmith-0.8.0}/pyproject.toml +7 -1
- {csvsmith-0.7.2 → csvsmith-0.8.0}/src/csvsmith/__init__.py +8 -2
- {csvsmith-0.7.2 → csvsmith-0.8.0}/src/csvsmith/cli.py +62 -1
- csvsmith-0.8.0/src/csvsmith/tools/strict_concat.py +101 -0
- csvsmith-0.8.0/src/csvsmith/utils/clean_numeric.py +124 -0
- {csvsmith-0.7.2 → csvsmith-0.8.0/src/csvsmith.egg-info}/PKG-INFO +26 -11
- {csvsmith-0.7.2 → csvsmith-0.8.0}/src/csvsmith.egg-info/SOURCES.txt +2 -0
- csvsmith-0.8.0/src/csvsmith.egg-info/requires.txt +5 -0
- {csvsmith-0.7.2 → csvsmith-0.8.0}/tests/test_clean_numeric.py +33 -1
- {csvsmith-0.7.2 → csvsmith-0.8.0}/tests/test_cli.py +21 -2
- csvsmith-0.8.0/tests/test_strict_concat.py +102 -0
- csvsmith-0.7.2/src/csvsmith/utils/clean_numeric.py +0 -124
- csvsmith-0.7.2/src/csvsmith.egg-info/requires.txt +0 -1
- {csvsmith-0.7.2 → csvsmith-0.8.0}/LICENSE +0 -0
- {csvsmith-0.7.2 → csvsmith-0.8.0}/setup.cfg +0 -0
- {csvsmith-0.7.2 → csvsmith-0.8.0}/src/csvsmith/tools/__init__.py +0 -0
- {csvsmith-0.7.2 → csvsmith-0.8.0}/src/csvsmith/tools/classify.py +0 -0
- {csvsmith-0.7.2 → csvsmith-0.8.0}/src/csvsmith/tools/excel2csv.py +0 -0
- {csvsmith-0.7.2 → csvsmith-0.8.0}/src/csvsmith/tools/filter_rows.py +0 -0
- {csvsmith-0.7.2 → csvsmith-0.8.0}/src/csvsmith/tools/find_matches_in_csv.py +0 -0
- {csvsmith-0.7.2 → csvsmith-0.8.0}/src/csvsmith/tools/move_files.py +0 -0
- {csvsmith-0.7.2 → csvsmith-0.8.0}/src/csvsmith/tools/row_dedup.py +0 -0
- {csvsmith-0.7.2 → csvsmith-0.8.0}/src/csvsmith/utils/__init__.py +0 -0
- {csvsmith-0.7.2 → csvsmith-0.8.0}/src/csvsmith/utils/distance.py +0 -0
- {csvsmith-0.7.2 → csvsmith-0.8.0}/src/csvsmith/utils/io.py +0 -0
- {csvsmith-0.7.2 → csvsmith-0.8.0}/src/csvsmith/utils/normalize.py +0 -0
- {csvsmith-0.7.2 → csvsmith-0.8.0}/src/csvsmith.egg-info/dependency_links.txt +0 -0
- {csvsmith-0.7.2 → csvsmith-0.8.0}/src/csvsmith.egg-info/entry_points.txt +0 -0
- {csvsmith-0.7.2 → csvsmith-0.8.0}/src/csvsmith.egg-info/top_level.txt +0 -0
- {csvsmith-0.7.2 → csvsmith-0.8.0}/tests/test_classify.py +0 -0
- {csvsmith-0.7.2 → csvsmith-0.8.0}/tests/test_excel2csv.py +0 -0
- {csvsmith-0.7.2 → csvsmith-0.8.0}/tests/test_filter_rows.py +0 -0
- {csvsmith-0.7.2 → csvsmith-0.8.0}/tests/test_find_matches_in_csv.py +0 -0
- {csvsmith-0.7.2 → csvsmith-0.8.0}/tests/test_move_files.py +0 -0
- {csvsmith-0.7.2 → csvsmith-0.8.0}/tests/test_normalize.py +0 -0
- {csvsmith-0.7.2 → csvsmith-0.8.0}/tests/test_row_dedup.py +0 -0
- {csvsmith-0.7.2 → csvsmith-0.8.0}/tests/test_string_distance.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: csvsmith
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.8.0
|
|
4
4
|
Summary: Small CSV utilities: row deduplication, classification, row filtering, and CLI helpers.
|
|
5
5
|
Author-email: Eiichi YAMAMOTO <info@yeiichi.com>
|
|
6
6
|
License: MIT License
|
|
@@ -39,6 +39,9 @@ Requires-Python: >=3.10
|
|
|
39
39
|
Description-Content-Type: text/x-rst
|
|
40
40
|
License-File: LICENSE
|
|
41
41
|
Requires-Dist: openpyxl>=3.1
|
|
42
|
+
Provides-Extra: docs
|
|
43
|
+
Requires-Dist: sphinx<9,>=8; extra == "docs"
|
|
44
|
+
Requires-Dist: furo>=2024.8.6; extra == "docs"
|
|
42
45
|
Dynamic: license-file
|
|
43
46
|
|
|
44
47
|
csvsmith
|
|
@@ -84,6 +87,7 @@ Features
|
|
|
84
87
|
- Convert Excel workbooks to CSV
|
|
85
88
|
- Move files by suffix
|
|
86
89
|
- Find matching values inside CSV files
|
|
90
|
+
- Concatenate CSV files with identical headers
|
|
87
91
|
- Use the tools either from Python or from the command line
|
|
88
92
|
|
|
89
93
|
Installation
|
|
@@ -111,17 +115,11 @@ You can use the library from Python:
|
|
|
111
115
|
|
|
112
116
|
.. code-block:: python
|
|
113
117
|
|
|
114
|
-
from csvsmith import
|
|
115
|
-
clean_numeric,
|
|
116
|
-
dedupe_with_report,
|
|
117
|
-
excel_to_csv,
|
|
118
|
-
find_matches_in_csv,
|
|
119
|
-
move_by_suffix,
|
|
120
|
-
)
|
|
118
|
+
from csvsmith.utils.clean_numeric import clean_currency_numeric
|
|
121
119
|
|
|
122
|
-
print(
|
|
120
|
+
print(clean_currency_numeric("$1,234.56"))
|
|
123
121
|
|
|
124
|
-
|
|
122
|
+
For command-line usage, use single quotes around values containing ``$``:
|
|
125
123
|
|
|
126
124
|
.. code-block:: console
|
|
127
125
|
|
|
@@ -138,6 +136,17 @@ Clean numeric values:
|
|
|
138
136
|
|
|
139
137
|
csvsmith clean-numeric "1,234.56" --sep "," --decimal "."
|
|
140
138
|
|
|
139
|
+
Clean currency-prefixed numeric values:
|
|
140
|
+
|
|
141
|
+
.. code-block:: console
|
|
142
|
+
|
|
143
|
+
csvsmith clean-currency-numeric '$1,234.56' --sep "," --decimal "."
|
|
144
|
+
|
|
145
|
+
.. note::
|
|
146
|
+
|
|
147
|
+
Use single quotes for values containing ``$``. Double quotes may trigger
|
|
148
|
+
shell expansion and change the input unexpectedly.
|
|
149
|
+
|
|
141
150
|
Filter rows in a CSV:
|
|
142
151
|
|
|
143
152
|
.. code-block:: console
|
|
@@ -174,6 +183,12 @@ Find matches in a CSV:
|
|
|
174
183
|
|
|
175
184
|
csvsmith find-matches input.csv target --ignore-case --ignore-whitespace
|
|
176
185
|
|
|
186
|
+
Concatenate CSV files:
|
|
187
|
+
|
|
188
|
+
.. code-block:: console
|
|
189
|
+
|
|
190
|
+
csvsmith strict-concat file1.csv file2.csv -o combined.csv
|
|
191
|
+
|
|
177
192
|
Find matches in a CSV
|
|
178
193
|
---------------------
|
|
179
194
|
|
|
@@ -233,7 +248,7 @@ File and conversion helpers:
|
|
|
233
248
|
|
|
234
249
|
.. code-block:: python
|
|
235
250
|
|
|
236
|
-
from csvsmith import excel_to_csv, move_by_suffix
|
|
251
|
+
from csvsmith import excel_to_csv, move_by_suffix, strict_concat_rows, save_csv
|
|
237
252
|
|
|
238
253
|
String comparison utilities:
|
|
239
254
|
|
|
@@ -41,6 +41,7 @@ Features
|
|
|
41
41
|
- Convert Excel workbooks to CSV
|
|
42
42
|
- Move files by suffix
|
|
43
43
|
- Find matching values inside CSV files
|
|
44
|
+
- Concatenate CSV files with identical headers
|
|
44
45
|
- Use the tools either from Python or from the command line
|
|
45
46
|
|
|
46
47
|
Installation
|
|
@@ -68,17 +69,11 @@ You can use the library from Python:
|
|
|
68
69
|
|
|
69
70
|
.. code-block:: python
|
|
70
71
|
|
|
71
|
-
from csvsmith import
|
|
72
|
-
clean_numeric,
|
|
73
|
-
dedupe_with_report,
|
|
74
|
-
excel_to_csv,
|
|
75
|
-
find_matches_in_csv,
|
|
76
|
-
move_by_suffix,
|
|
77
|
-
)
|
|
72
|
+
from csvsmith.utils.clean_numeric import clean_currency_numeric
|
|
78
73
|
|
|
79
|
-
print(
|
|
74
|
+
print(clean_currency_numeric("$1,234.56"))
|
|
80
75
|
|
|
81
|
-
|
|
76
|
+
For command-line usage, use single quotes around values containing ``$``:
|
|
82
77
|
|
|
83
78
|
.. code-block:: console
|
|
84
79
|
|
|
@@ -95,6 +90,17 @@ Clean numeric values:
|
|
|
95
90
|
|
|
96
91
|
csvsmith clean-numeric "1,234.56" --sep "," --decimal "."
|
|
97
92
|
|
|
93
|
+
Clean currency-prefixed numeric values:
|
|
94
|
+
|
|
95
|
+
.. code-block:: console
|
|
96
|
+
|
|
97
|
+
csvsmith clean-currency-numeric '$1,234.56' --sep "," --decimal "."
|
|
98
|
+
|
|
99
|
+
.. note::
|
|
100
|
+
|
|
101
|
+
Use single quotes for values containing ``$``. Double quotes may trigger
|
|
102
|
+
shell expansion and change the input unexpectedly.
|
|
103
|
+
|
|
98
104
|
Filter rows in a CSV:
|
|
99
105
|
|
|
100
106
|
.. code-block:: console
|
|
@@ -131,6 +137,12 @@ Find matches in a CSV:
|
|
|
131
137
|
|
|
132
138
|
csvsmith find-matches input.csv target --ignore-case --ignore-whitespace
|
|
133
139
|
|
|
140
|
+
Concatenate CSV files:
|
|
141
|
+
|
|
142
|
+
.. code-block:: console
|
|
143
|
+
|
|
144
|
+
csvsmith strict-concat file1.csv file2.csv -o combined.csv
|
|
145
|
+
|
|
134
146
|
Find matches in a CSV
|
|
135
147
|
---------------------
|
|
136
148
|
|
|
@@ -190,7 +202,7 @@ File and conversion helpers:
|
|
|
190
202
|
|
|
191
203
|
.. code-block:: python
|
|
192
204
|
|
|
193
|
-
from csvsmith import excel_to_csv, move_by_suffix
|
|
205
|
+
from csvsmith import excel_to_csv, move_by_suffix, strict_concat_rows, save_csv
|
|
194
206
|
|
|
195
207
|
String comparison utilities:
|
|
196
208
|
|
|
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
|
|
|
4
4
|
|
|
5
5
|
[project]
|
|
6
6
|
name = "csvsmith"
|
|
7
|
-
version = "0.
|
|
7
|
+
version = "0.8.0"
|
|
8
8
|
description = "Small CSV utilities: row deduplication, classification, row filtering, and CLI helpers."
|
|
9
9
|
readme = "README.rst"
|
|
10
10
|
requires-python = ">=3.10"
|
|
@@ -35,6 +35,12 @@ dependencies = [
|
|
|
35
35
|
"openpyxl>=3.1",
|
|
36
36
|
]
|
|
37
37
|
|
|
38
|
+
[project.optional-dependencies]
|
|
39
|
+
docs = [
|
|
40
|
+
"sphinx>=8,<9",
|
|
41
|
+
"furo>=2024.8.6",
|
|
42
|
+
]
|
|
43
|
+
|
|
38
44
|
[project.urls]
|
|
39
45
|
Homepage = "https://github.com/yeiichi/csvsmith"
|
|
40
46
|
Repository = "https://github.com/yeiichi/csvsmith"
|
|
@@ -16,6 +16,8 @@ Public API:
|
|
|
16
16
|
- Relation
|
|
17
17
|
- Result
|
|
18
18
|
- analyze_pair
|
|
19
|
+
- strict_concat_rows
|
|
20
|
+
- save_csv
|
|
19
21
|
|
|
20
22
|
Compatibility aliases:
|
|
21
23
|
- CSVCleaner
|
|
@@ -28,10 +30,11 @@ Submodules:
|
|
|
28
30
|
- csvsmith.filter_rows
|
|
29
31
|
- csvsmith.excel2csv
|
|
30
32
|
- csvsmith.move_files
|
|
33
|
+
- csvsmith.strict_concat
|
|
31
34
|
- csvsmith.cli (CLI entrypoint)
|
|
32
35
|
"""
|
|
33
36
|
|
|
34
|
-
__version__ = "0.
|
|
37
|
+
__version__ = "0.8.0"
|
|
35
38
|
|
|
36
39
|
from .tools.classify import CSVClassifier
|
|
37
40
|
from .tools.excel2csv import excel_to_csv
|
|
@@ -43,6 +46,7 @@ from .tools.row_dedup import (
|
|
|
43
46
|
find_duplicate_rows,
|
|
44
47
|
dedupe_with_report,
|
|
45
48
|
)
|
|
49
|
+
from .tools.strict_concat import save_csv, strict_concat_rows
|
|
46
50
|
from .utils.clean_numeric import clean_numeric
|
|
47
51
|
from .utils.distance import StringDistance, Relation, Result, analyze_pair
|
|
48
52
|
from .utils.io import (
|
|
@@ -68,4 +72,6 @@ __all__ = [
|
|
|
68
72
|
"Result",
|
|
69
73
|
"analyze_pair",
|
|
70
74
|
"clean_numeric",
|
|
71
|
-
|
|
75
|
+
"strict_concat_rows",
|
|
76
|
+
"save_csv",
|
|
77
|
+
]
|
|
@@ -9,13 +9,14 @@ from typing import Optional, Sequence
|
|
|
9
9
|
from . import __version__
|
|
10
10
|
from .tools.classify import CSVClassifier
|
|
11
11
|
from .tools.excel2csv import excel_to_csv
|
|
12
|
-
from .utils.clean_numeric import clean_numeric
|
|
12
|
+
from .utils.clean_numeric import clean_currency_numeric, clean_numeric
|
|
13
13
|
from .tools.filter_rows import DropRowsBySubstring
|
|
14
14
|
from .tools.move_files import move_by_suffix, normalize_suffixes
|
|
15
15
|
from .tools.row_dedup import (
|
|
16
16
|
dedupe_with_report,
|
|
17
17
|
find_duplicate_rows,
|
|
18
18
|
)
|
|
19
|
+
from .tools.strict_concat import save_csv, strict_concat_rows
|
|
19
20
|
from .tools.find_matches_in_csv import find_matches_in_csv
|
|
20
21
|
from .utils.distance import analyze_pair
|
|
21
22
|
from .utils.io import read_csv_rows, write_csv_rows
|
|
@@ -149,6 +150,21 @@ def cmd_clean_numeric(args: argparse.Namespace) -> int:
|
|
|
149
150
|
return 0
|
|
150
151
|
|
|
151
152
|
|
|
153
|
+
def cmd_clean_currency_numeric(args: argparse.Namespace) -> int:
|
|
154
|
+
try:
|
|
155
|
+
cleaned = clean_currency_numeric(
|
|
156
|
+
args.value,
|
|
157
|
+
sep=args.sep,
|
|
158
|
+
decimal=args.decimal,
|
|
159
|
+
relaxed=args.relaxed,
|
|
160
|
+
)
|
|
161
|
+
print(cleaned)
|
|
162
|
+
except ValueError as e:
|
|
163
|
+
print(f"Error: {e}", file=sys.stderr)
|
|
164
|
+
return 1
|
|
165
|
+
return 0
|
|
166
|
+
|
|
167
|
+
|
|
152
168
|
def cmd_find_matches(args: argparse.Namespace) -> int:
|
|
153
169
|
results = find_matches_in_csv(
|
|
154
170
|
args.input,
|
|
@@ -165,6 +181,23 @@ def cmd_find_matches(args: argparse.Namespace) -> int:
|
|
|
165
181
|
return 0
|
|
166
182
|
|
|
167
183
|
|
|
184
|
+
def cmd_strict_concat(args: argparse.Namespace) -> int:
|
|
185
|
+
input_dir = Path(args.input_dir)
|
|
186
|
+
if not input_dir.is_dir():
|
|
187
|
+
print(f"Error: input directory not found: {input_dir}", file=sys.stderr)
|
|
188
|
+
return 1
|
|
189
|
+
|
|
190
|
+
try:
|
|
191
|
+
rows = strict_concat_rows(input_dir)
|
|
192
|
+
save_csv(rows, args.output)
|
|
193
|
+
except Exception as e:
|
|
194
|
+
print(f"Error: {e}", file=sys.stderr)
|
|
195
|
+
return 1
|
|
196
|
+
|
|
197
|
+
print(f"Wrote concatenated CSV to: {args.output}")
|
|
198
|
+
return 0
|
|
199
|
+
|
|
200
|
+
|
|
168
201
|
def _add_find_matches_parser(subparsers) -> None:
|
|
169
202
|
parser = subparsers.add_parser("find-matches", help="Find matches in a CSV file.")
|
|
170
203
|
parser.add_argument("input", help="Input CSV file.")
|
|
@@ -175,6 +208,16 @@ def _add_find_matches_parser(subparsers) -> None:
|
|
|
175
208
|
parser.set_defaults(func=cmd_find_matches)
|
|
176
209
|
|
|
177
210
|
|
|
211
|
+
def _add_strict_concat_parser(subparsers) -> None:
|
|
212
|
+
parser = subparsers.add_parser(
|
|
213
|
+
"strict-concat",
|
|
214
|
+
help="Concatenate CSVs in a directory only when all headers match exactly.",
|
|
215
|
+
)
|
|
216
|
+
parser.add_argument("input_dir", help="Directory containing CSV files to concatenate.")
|
|
217
|
+
parser.add_argument("-o", "--output", required=True, help="Output CSV file path.")
|
|
218
|
+
parser.set_defaults(func=cmd_strict_concat)
|
|
219
|
+
|
|
220
|
+
|
|
178
221
|
def _add_row_duplicates_parser(subparsers) -> None:
|
|
179
222
|
parser = subparsers.add_parser("row-duplicates", help="Find duplicate rows in a CSV.")
|
|
180
223
|
parser.add_argument("input", help="Input CSV file.")
|
|
@@ -272,6 +315,22 @@ def _add_clean_numeric_parser(subparsers) -> None:
|
|
|
272
315
|
parser.set_defaults(func=cmd_clean_numeric)
|
|
273
316
|
|
|
274
317
|
|
|
318
|
+
def _add_clean_currency_numeric_parser(subparsers) -> None:
|
|
319
|
+
parser = subparsers.add_parser(
|
|
320
|
+
"clean-currency-numeric",
|
|
321
|
+
help="Clean and convert a currency-prefixed numeric string to float.",
|
|
322
|
+
)
|
|
323
|
+
parser.add_argument("value", help="Numeric value to clean.")
|
|
324
|
+
parser.add_argument("--sep", default=",", help="Group separator (default: ,).")
|
|
325
|
+
parser.add_argument("--decimal", default=".", help="Decimal separator (default: .).")
|
|
326
|
+
parser.add_argument(
|
|
327
|
+
"--relaxed",
|
|
328
|
+
action="store_true",
|
|
329
|
+
help="Return the original input when it is not numeric.",
|
|
330
|
+
)
|
|
331
|
+
parser.set_defaults(func=cmd_clean_currency_numeric)
|
|
332
|
+
|
|
333
|
+
|
|
275
334
|
def build_parser() -> argparse.ArgumentParser:
|
|
276
335
|
parser = argparse.ArgumentParser(
|
|
277
336
|
prog="csvsmith",
|
|
@@ -286,6 +345,7 @@ def build_parser() -> argparse.ArgumentParser:
|
|
|
286
345
|
|
|
287
346
|
_add_row_duplicates_parser(subparsers)
|
|
288
347
|
_add_find_matches_parser(subparsers)
|
|
348
|
+
_add_strict_concat_parser(subparsers)
|
|
289
349
|
_add_dedupe_parser(subparsers)
|
|
290
350
|
_add_classify_parser(subparsers)
|
|
291
351
|
_add_move_files_parser(subparsers)
|
|
@@ -293,6 +353,7 @@ def build_parser() -> argparse.ArgumentParser:
|
|
|
293
353
|
_add_drop_rows_parser(subparsers)
|
|
294
354
|
_add_string_distance_parser(subparsers)
|
|
295
355
|
_add_clean_numeric_parser(subparsers)
|
|
356
|
+
_add_clean_currency_numeric_parser(subparsers)
|
|
296
357
|
|
|
297
358
|
return parser
|
|
298
359
|
|
|
@@ -0,0 +1,101 @@
|
|
|
1
|
+
import csv
|
|
2
|
+
from collections.abc import Iterable
|
|
3
|
+
from pathlib import Path
|
|
4
|
+
|
|
5
|
+
|
|
6
|
+
def find_csvs(csv_dir: Path | str) -> list[Path]:
|
|
7
|
+
"""
|
|
8
|
+
Find all CSV files in the specified directory.
|
|
9
|
+
|
|
10
|
+
This function searches for all files with a ``.csv`` extension in the given
|
|
11
|
+
directory and returns a sorted list of their paths.
|
|
12
|
+
|
|
13
|
+
:param csv_dir: The directory to search for CSV files. This can be provided
|
|
14
|
+
as either a ``Path`` object or a string representing the path to the
|
|
15
|
+
directory.
|
|
16
|
+
:type csv_dir: Path | str
|
|
17
|
+
:return: Sorted list of paths to all ``.csv`` files found in the specified
|
|
18
|
+
directory.
|
|
19
|
+
:rtype: list[Path]
|
|
20
|
+
"""
|
|
21
|
+
return sorted(Path(csv_dir).glob("*.csv"))
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
def read_header(csv_path: Path) -> list[str]:
|
|
25
|
+
"""
|
|
26
|
+
Reads the header row of a given CSV file.
|
|
27
|
+
|
|
28
|
+
This function opens a CSV file located at the specified path, reads its first
|
|
29
|
+
row, and returns it as a list of strings. The file is assumed to be encoded
|
|
30
|
+
in UTF-8 with optional BOM (Byte Order Mark). If the CSV file is empty, a
|
|
31
|
+
ValueError is raised indicating the problem. The CSV file is expected to be
|
|
32
|
+
opened in read mode with no newline translation.
|
|
33
|
+
|
|
34
|
+
:param csv_path: The path to the CSV file to read the header from.
|
|
35
|
+
:type csv_path: Path
|
|
36
|
+
:return: A list of strings representing the header row of the CSV file.
|
|
37
|
+
:rtype: list[str]
|
|
38
|
+
:raises ValueError: If the CSV file is empty and no header can be retrieved.
|
|
39
|
+
"""
|
|
40
|
+
with csv_path.open(encoding="utf-8-sig", newline="") as f:
|
|
41
|
+
reader = csv.reader(f)
|
|
42
|
+
try:
|
|
43
|
+
return next(reader)
|
|
44
|
+
except StopIteration as e:
|
|
45
|
+
raise ValueError(f"Empty CSV: {csv_path}") from e
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
def _validate_headers_match(csv_paths: list[Path]) -> list[str]:
|
|
49
|
+
"""
|
|
50
|
+
Validates that the headers of all provided CSV files match each other. The function compares the
|
|
51
|
+
header of each CSV file in the input list against the header of the first file. If a mismatch
|
|
52
|
+
is encountered, a ValueError is raised indicating the problematic file. The function returns
|
|
53
|
+
the header of the first file if all headers are consistent.
|
|
54
|
+
|
|
55
|
+
:param csv_paths: A list of Path objects representing the file paths to the CSV files to validate.
|
|
56
|
+
:type csv_paths: list[Path]
|
|
57
|
+
:return: A list of strings representing the matched header of the first CSV file.
|
|
58
|
+
:rtype: list[str]
|
|
59
|
+
"""
|
|
60
|
+
expected_header = read_header(csv_paths[0])
|
|
61
|
+
for csv_path in csv_paths[1:]:
|
|
62
|
+
if read_header(csv_path) != expected_header:
|
|
63
|
+
raise ValueError(f"Header mismatch: {csv_path}")
|
|
64
|
+
return expected_header
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
def strict_concat_rows(csv_dir: Path | str) -> list[list[str]]:
|
|
68
|
+
"""
|
|
69
|
+
Concatenates rows from multiple CSV files into a list of lists of strings, ensuring
|
|
70
|
+
the headers across all CSV files match. The output includes a new column indicating
|
|
71
|
+
the file stem.
|
|
72
|
+
|
|
73
|
+
:param csv_dir: Directory containing the CSV files or a specific path to a CSV file.
|
|
74
|
+
:type csv_dir: Path | str
|
|
75
|
+
:return: A list of lists, where each inner list represents a row from the concatenated
|
|
76
|
+
CSV files. The first row contains the headers, including a "file_stem" column.
|
|
77
|
+
:rtype: list[list[str]]
|
|
78
|
+
:raises FileNotFoundError: If no CSV files are found in the provided directory.
|
|
79
|
+
"""
|
|
80
|
+
csv_paths = find_csvs(csv_dir)
|
|
81
|
+
if not csv_paths:
|
|
82
|
+
raise FileNotFoundError(f"No CSV files found in: {csv_dir}")
|
|
83
|
+
|
|
84
|
+
expected_header = _validate_headers_match(csv_paths)
|
|
85
|
+
out_rows: list[list[str]] = [["file_stem", *expected_header]]
|
|
86
|
+
|
|
87
|
+
for csv_path in csv_paths:
|
|
88
|
+
with csv_path.open(encoding="utf-8-sig", newline="") as f:
|
|
89
|
+
reader = csv.reader(f)
|
|
90
|
+
next(reader) # skip header
|
|
91
|
+
for row in reader:
|
|
92
|
+
out_rows.append([csv_path.stem, *row])
|
|
93
|
+
return out_rows
|
|
94
|
+
|
|
95
|
+
|
|
96
|
+
def save_csv(rows: Iterable[list[str]], out_path: Path | str) -> None:
|
|
97
|
+
"""Write rows to out_path."""
|
|
98
|
+
out_path = Path(out_path)
|
|
99
|
+
with out_path.open("w", encoding="utf-8", newline="") as f:
|
|
100
|
+
writer = csv.writer(f)
|
|
101
|
+
writer.writerows(rows)
|
|
@@ -0,0 +1,124 @@
|
|
|
1
|
+
import re
|
|
2
|
+
from typing import Any
|
|
3
|
+
|
|
4
|
+
NON_BREAKING_SPACE = "\xa0"
|
|
5
|
+
SEPARATOR_PATTERN = re.compile(r"[ _\xa0]")
|
|
6
|
+
NUMBER_PATTERN = re.compile(r"^-?(?:\d+|\d*\.\d+)$")
|
|
7
|
+
INVALID_NUMBER_MESSAGE = "Could not convert {value!r} to a valid number."
|
|
8
|
+
CURRENCY_PREFIX_PATTERN = re.compile(r"^[\$€£¥₹]")
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
def strip_currency_prefix(value: Any) -> Any:
|
|
12
|
+
"""
|
|
13
|
+
Remove a single common currency symbol from the start of a value.
|
|
14
|
+
"""
|
|
15
|
+
text = str(value).strip()
|
|
16
|
+
if text and CURRENCY_PREFIX_PATTERN.match(text):
|
|
17
|
+
return text[1:].strip()
|
|
18
|
+
return value
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
def _normalize_numeric_text(value: Any, *, sep: str, decimal: str) -> str:
|
|
22
|
+
"""
|
|
23
|
+
Normalize a numeric text string for consistent formatting.
|
|
24
|
+
|
|
25
|
+
Converts a given value to a string representation and ensures normalization of numeric formatting,
|
|
26
|
+
such as removing group separators, converting localized decimal separators, and handling negative
|
|
27
|
+
values enclosed in parentheses.
|
|
28
|
+
"""
|
|
29
|
+
numeric_text = str(value).strip()
|
|
30
|
+
|
|
31
|
+
if numeric_text.startswith("(") and numeric_text.endswith(")"):
|
|
32
|
+
numeric_text = f"-{numeric_text[1:-1]}"
|
|
33
|
+
|
|
34
|
+
if sep:
|
|
35
|
+
numeric_text = numeric_text.replace(sep, "")
|
|
36
|
+
|
|
37
|
+
if decimal != ".":
|
|
38
|
+
numeric_text = numeric_text.replace(decimal, ".")
|
|
39
|
+
|
|
40
|
+
return numeric_text
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
def _has_valid_grouping(numeric_text: str, *, decimal: str) -> bool:
|
|
44
|
+
"""
|
|
45
|
+
Checks whether a numeric text string has valid grouping based on a specified decimal character.
|
|
46
|
+
"""
|
|
47
|
+
if not numeric_text:
|
|
48
|
+
return False
|
|
49
|
+
|
|
50
|
+
unsigned_text = numeric_text[1:] if numeric_text.startswith("-") else numeric_text
|
|
51
|
+
|
|
52
|
+
if unsigned_text.count(decimal) > 1:
|
|
53
|
+
return False
|
|
54
|
+
|
|
55
|
+
integer_text, _, fraction_text = unsigned_text.partition(decimal)
|
|
56
|
+
|
|
57
|
+
if not integer_text and not fraction_text:
|
|
58
|
+
return False
|
|
59
|
+
|
|
60
|
+
for part in (integer_text, fraction_text):
|
|
61
|
+
if not part:
|
|
62
|
+
continue
|
|
63
|
+
if part.startswith("_") or part.endswith("_"):
|
|
64
|
+
return False
|
|
65
|
+
if part.startswith(" ") or part.endswith(" "):
|
|
66
|
+
return False
|
|
67
|
+
if part.startswith(NON_BREAKING_SPACE) or part.endswith(NON_BREAKING_SPACE):
|
|
68
|
+
return False
|
|
69
|
+
if "__" in part or " " in part or NON_BREAKING_SPACE * 2 in part:
|
|
70
|
+
return False
|
|
71
|
+
|
|
72
|
+
stripped_text = SEPARATOR_PATTERN.sub("", numeric_text)
|
|
73
|
+
return bool(NUMBER_PATTERN.fullmatch(stripped_text))
|
|
74
|
+
|
|
75
|
+
|
|
76
|
+
def _strip_group_separators(numeric_text: str) -> str:
|
|
77
|
+
return SEPARATOR_PATTERN.sub("", numeric_text)
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
def _invalid_number_error(value: Any) -> ValueError:
|
|
81
|
+
return ValueError(INVALID_NUMBER_MESSAGE.format(value=value))
|
|
82
|
+
|
|
83
|
+
|
|
84
|
+
def clean_numeric(
|
|
85
|
+
value: Any, *, sep: str = ",", decimal: str = ".", relaxed: bool = False
|
|
86
|
+
) -> float | Any:
|
|
87
|
+
"""
|
|
88
|
+
Cleans and converts a given input to a float by normalizing its numeric representation.
|
|
89
|
+
"""
|
|
90
|
+
if value is None:
|
|
91
|
+
return 0.0
|
|
92
|
+
|
|
93
|
+
normalized_text = _normalize_numeric_text(value, sep=sep, decimal=decimal)
|
|
94
|
+
|
|
95
|
+
if not _has_valid_grouping(normalized_text, decimal=decimal):
|
|
96
|
+
if relaxed:
|
|
97
|
+
return value
|
|
98
|
+
raise _invalid_number_error(value)
|
|
99
|
+
|
|
100
|
+
candidate_text = _strip_group_separators(normalized_text)
|
|
101
|
+
|
|
102
|
+
try:
|
|
103
|
+
return float(candidate_text)
|
|
104
|
+
except ValueError as exc:
|
|
105
|
+
if relaxed:
|
|
106
|
+
return value
|
|
107
|
+
raise _invalid_number_error(value) from exc
|
|
108
|
+
|
|
109
|
+
|
|
110
|
+
def clean_currency_numeric(
|
|
111
|
+
value: Any, *, sep: str = ",", decimal: str = ".", relaxed: bool = False
|
|
112
|
+
) -> float | Any:
|
|
113
|
+
"""
|
|
114
|
+
Cleans and converts a currency-prefixed numeric string to a float.
|
|
115
|
+
"""
|
|
116
|
+
if value is None:
|
|
117
|
+
return 0.0
|
|
118
|
+
|
|
119
|
+
return clean_numeric(
|
|
120
|
+
strip_currency_prefix(value),
|
|
121
|
+
sep=sep,
|
|
122
|
+
decimal=decimal,
|
|
123
|
+
relaxed=relaxed,
|
|
124
|
+
)
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: csvsmith
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.8.0
|
|
4
4
|
Summary: Small CSV utilities: row deduplication, classification, row filtering, and CLI helpers.
|
|
5
5
|
Author-email: Eiichi YAMAMOTO <info@yeiichi.com>
|
|
6
6
|
License: MIT License
|
|
@@ -39,6 +39,9 @@ Requires-Python: >=3.10
|
|
|
39
39
|
Description-Content-Type: text/x-rst
|
|
40
40
|
License-File: LICENSE
|
|
41
41
|
Requires-Dist: openpyxl>=3.1
|
|
42
|
+
Provides-Extra: docs
|
|
43
|
+
Requires-Dist: sphinx<9,>=8; extra == "docs"
|
|
44
|
+
Requires-Dist: furo>=2024.8.6; extra == "docs"
|
|
42
45
|
Dynamic: license-file
|
|
43
46
|
|
|
44
47
|
csvsmith
|
|
@@ -84,6 +87,7 @@ Features
|
|
|
84
87
|
- Convert Excel workbooks to CSV
|
|
85
88
|
- Move files by suffix
|
|
86
89
|
- Find matching values inside CSV files
|
|
90
|
+
- Concatenate CSV files with identical headers
|
|
87
91
|
- Use the tools either from Python or from the command line
|
|
88
92
|
|
|
89
93
|
Installation
|
|
@@ -111,17 +115,11 @@ You can use the library from Python:
|
|
|
111
115
|
|
|
112
116
|
.. code-block:: python
|
|
113
117
|
|
|
114
|
-
from csvsmith import
|
|
115
|
-
clean_numeric,
|
|
116
|
-
dedupe_with_report,
|
|
117
|
-
excel_to_csv,
|
|
118
|
-
find_matches_in_csv,
|
|
119
|
-
move_by_suffix,
|
|
120
|
-
)
|
|
118
|
+
from csvsmith.utils.clean_numeric import clean_currency_numeric
|
|
121
119
|
|
|
122
|
-
print(
|
|
120
|
+
print(clean_currency_numeric("$1,234.56"))
|
|
123
121
|
|
|
124
|
-
|
|
122
|
+
For command-line usage, use single quotes around values containing ``$``:
|
|
125
123
|
|
|
126
124
|
.. code-block:: console
|
|
127
125
|
|
|
@@ -138,6 +136,17 @@ Clean numeric values:
|
|
|
138
136
|
|
|
139
137
|
csvsmith clean-numeric "1,234.56" --sep "," --decimal "."
|
|
140
138
|
|
|
139
|
+
Clean currency-prefixed numeric values:
|
|
140
|
+
|
|
141
|
+
.. code-block:: console
|
|
142
|
+
|
|
143
|
+
csvsmith clean-currency-numeric '$1,234.56' --sep "," --decimal "."
|
|
144
|
+
|
|
145
|
+
.. note::
|
|
146
|
+
|
|
147
|
+
Use single quotes for values containing ``$``. Double quotes may trigger
|
|
148
|
+
shell expansion and change the input unexpectedly.
|
|
149
|
+
|
|
141
150
|
Filter rows in a CSV:
|
|
142
151
|
|
|
143
152
|
.. code-block:: console
|
|
@@ -174,6 +183,12 @@ Find matches in a CSV:
|
|
|
174
183
|
|
|
175
184
|
csvsmith find-matches input.csv target --ignore-case --ignore-whitespace
|
|
176
185
|
|
|
186
|
+
Concatenate CSV files:
|
|
187
|
+
|
|
188
|
+
.. code-block:: console
|
|
189
|
+
|
|
190
|
+
csvsmith strict-concat file1.csv file2.csv -o combined.csv
|
|
191
|
+
|
|
177
192
|
Find matches in a CSV
|
|
178
193
|
---------------------
|
|
179
194
|
|
|
@@ -233,7 +248,7 @@ File and conversion helpers:
|
|
|
233
248
|
|
|
234
249
|
.. code-block:: python
|
|
235
250
|
|
|
236
|
-
from csvsmith import excel_to_csv, move_by_suffix
|
|
251
|
+
from csvsmith import excel_to_csv, move_by_suffix, strict_concat_rows, save_csv
|
|
237
252
|
|
|
238
253
|
String comparison utilities:
|
|
239
254
|
|
|
@@ -16,6 +16,7 @@ src/csvsmith/tools/filter_rows.py
|
|
|
16
16
|
src/csvsmith/tools/find_matches_in_csv.py
|
|
17
17
|
src/csvsmith/tools/move_files.py
|
|
18
18
|
src/csvsmith/tools/row_dedup.py
|
|
19
|
+
src/csvsmith/tools/strict_concat.py
|
|
19
20
|
src/csvsmith/utils/__init__.py
|
|
20
21
|
src/csvsmith/utils/clean_numeric.py
|
|
21
22
|
src/csvsmith/utils/distance.py
|
|
@@ -30,4 +31,5 @@ tests/test_find_matches_in_csv.py
|
|
|
30
31
|
tests/test_move_files.py
|
|
31
32
|
tests/test_normalize.py
|
|
32
33
|
tests/test_row_dedup.py
|
|
34
|
+
tests/test_strict_concat.py
|
|
33
35
|
tests/test_string_distance.py
|
|
@@ -1,5 +1,12 @@
|
|
|
1
1
|
import pytest
|
|
2
|
-
from csvsmith.
|
|
2
|
+
from csvsmith.cli import build_parser, main
|
|
3
|
+
from csvsmith.utils.clean_numeric import clean_currency_numeric, clean_numeric
|
|
4
|
+
|
|
5
|
+
|
|
6
|
+
def test_main_help():
|
|
7
|
+
with pytest.raises(SystemExit) as excinfo:
|
|
8
|
+
main(["--help"])
|
|
9
|
+
assert excinfo.value.code == 0
|
|
3
10
|
|
|
4
11
|
|
|
5
12
|
def test_clean_numeric_with_valid_integer_string() -> None:
|
|
@@ -34,3 +41,28 @@ def test_clean_numeric_with_multiple_decimal_points_raises_valueerror() -> None:
|
|
|
34
41
|
|
|
35
42
|
def test_clean_numeric_with_none_value() -> None:
|
|
36
43
|
assert clean_numeric(None) == 0.0
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
def test_clean_currency_numeric_with_currency_prefix() -> None:
|
|
47
|
+
assert clean_currency_numeric("$1,000") == 1000.0
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
def test_clean_currency_numeric_with_euro_prefix() -> None:
|
|
51
|
+
assert clean_currency_numeric("€1,000.50") == 1000.5
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
def test_clean_numeric_still_rejects_currency_prefix() -> None:
|
|
55
|
+
with pytest.raises(ValueError, match=r"Could not convert '\$1,000'.*"):
|
|
56
|
+
clean_numeric("$1,000")
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
def test_cli_parses_clean_currency_numeric_command():
|
|
60
|
+
parser = build_parser()
|
|
61
|
+
args = parser.parse_args(
|
|
62
|
+
["clean-currency-numeric", "$1,234.56", "--sep", ",", "--decimal", "."]
|
|
63
|
+
)
|
|
64
|
+
|
|
65
|
+
assert args.command == "clean-currency-numeric"
|
|
66
|
+
assert args.value == "$1,234.56"
|
|
67
|
+
assert args.sep == ","
|
|
68
|
+
assert args.decimal == "."
|
|
@@ -138,12 +138,22 @@ def test_cli_parses_clean_numeric_command():
|
|
|
138
138
|
assert args.decimal == "."
|
|
139
139
|
|
|
140
140
|
|
|
141
|
-
def
|
|
141
|
+
def test_cli_parses_clean_currency_numeric_command():
|
|
142
142
|
parser = build_parser()
|
|
143
143
|
args = parser.parse_args(
|
|
144
|
-
["clean-numeric", "
|
|
144
|
+
["clean-currency-numeric", "$1,234.56", "--sep", ",", "--decimal", "."]
|
|
145
145
|
)
|
|
146
146
|
|
|
147
|
+
assert args.command == "clean-currency-numeric"
|
|
148
|
+
assert args.value == "$1,234.56"
|
|
149
|
+
assert args.sep == ","
|
|
150
|
+
assert args.decimal == "."
|
|
151
|
+
|
|
152
|
+
|
|
153
|
+
def test_cli_parses_clean_numeric_command_with_relaxed_mode():
|
|
154
|
+
parser = build_parser()
|
|
155
|
+
args = parser.parse_args(["clean-numeric", "not-a-number", "--relaxed"])
|
|
156
|
+
|
|
147
157
|
assert args.command == "clean-numeric"
|
|
148
158
|
assert args.value == "not-a-number"
|
|
149
159
|
assert args.relaxed is True
|
|
@@ -178,3 +188,12 @@ def test_cli_parses_find_matches_command():
|
|
|
178
188
|
assert args.ignore_case is True
|
|
179
189
|
assert args.ignore_whitespace is True
|
|
180
190
|
assert args.no_nfkc is True
|
|
191
|
+
|
|
192
|
+
|
|
193
|
+
def test_cli_parses_strict_concat_command():
|
|
194
|
+
parser = build_parser()
|
|
195
|
+
args = parser.parse_args(["strict-concat", "input_dir", "-o", "output.csv"])
|
|
196
|
+
|
|
197
|
+
assert args.command == "strict-concat"
|
|
198
|
+
assert args.input_dir == "input_dir"
|
|
199
|
+
assert args.output == "output.csv"
|
|
@@ -0,0 +1,102 @@
|
|
|
1
|
+
import csv
|
|
2
|
+
from pathlib import Path
|
|
3
|
+
|
|
4
|
+
import pytest
|
|
5
|
+
|
|
6
|
+
from csvsmith.tools.strict_concat import strict_concat_rows
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
def write_csv(path: Path, rows: list[list[str]]):
|
|
10
|
+
with path.open("w", newline="", encoding="utf-8") as f:
|
|
11
|
+
writer = csv.writer(f)
|
|
12
|
+
writer.writerows(rows)
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
def test_strict_concat_happy(tmp_path):
|
|
16
|
+
f1 = tmp_path / "a.csv"
|
|
17
|
+
f2 = tmp_path / "b.csv"
|
|
18
|
+
|
|
19
|
+
write_csv(f1, [["id", "name"], ["1", "Alice"]])
|
|
20
|
+
write_csv(f2, [["id", "name"], ["2", "Bob"]])
|
|
21
|
+
|
|
22
|
+
rows = strict_concat_rows(tmp_path)
|
|
23
|
+
|
|
24
|
+
assert rows == [
|
|
25
|
+
["file_stem", "id", "name"],
|
|
26
|
+
["a", "1", "Alice"],
|
|
27
|
+
["b", "2", "Bob"],
|
|
28
|
+
]
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
def test_strict_concat_header_mismatch(tmp_path):
|
|
32
|
+
write_csv(tmp_path / "a.csv", [["id", "name"], ["1", "Alice"]])
|
|
33
|
+
write_csv(tmp_path / "b.csv", [["id", "age"], ["2", "30"]])
|
|
34
|
+
|
|
35
|
+
with pytest.raises(ValueError, match="Header mismatch"):
|
|
36
|
+
strict_concat_rows(tmp_path)
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
def test_strict_concat_empty_dir(tmp_path):
|
|
40
|
+
with pytest.raises(FileNotFoundError):
|
|
41
|
+
strict_concat_rows(tmp_path)
|
|
42
|
+
|
|
43
|
+
def test_strict_concat_with_non_csv_files(tmp_path):
|
|
44
|
+
# Create sample CSV files.
|
|
45
|
+
write_csv(tmp_path / "a.csv", [["id", "name"], ["1", "Alice"]])
|
|
46
|
+
write_csv(tmp_path / "b.csv", [["id", "name"], ["2", "Bob"]])
|
|
47
|
+
|
|
48
|
+
# Create some non-CSV files in the same directory.
|
|
49
|
+
(tmp_path / "file1.txt").write_text("This is a text file.", encoding="utf-8")
|
|
50
|
+
(tmp_path / "file2.doc").write_text("This is a Word document.", encoding="utf-8")
|
|
51
|
+
|
|
52
|
+
# Call function and check results ignore non-CSV files.
|
|
53
|
+
rows = strict_concat_rows(tmp_path)
|
|
54
|
+
assert rows == [
|
|
55
|
+
["file_stem", "id", "name"],
|
|
56
|
+
["a", "1", "Alice"],
|
|
57
|
+
["b", "2", "Bob"],
|
|
58
|
+
]
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
def test_strict_concat_empty_csv(tmp_path):
|
|
62
|
+
(tmp_path / "a.csv").write_text("", encoding="utf-8")
|
|
63
|
+
|
|
64
|
+
with pytest.raises(ValueError, match="Empty CSV"):
|
|
65
|
+
strict_concat_rows(tmp_path)
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
def test_header_only_files(tmp_path):
|
|
69
|
+
write_csv(tmp_path / "a.csv", [["id", "name"]])
|
|
70
|
+
write_csv(tmp_path / "b.csv", [["id", "name"]])
|
|
71
|
+
|
|
72
|
+
rows = strict_concat_rows(tmp_path)
|
|
73
|
+
|
|
74
|
+
assert rows == [["file_stem", "id", "name"]]
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
def test_strict_concat_cli(tmp_path, capsys):
|
|
78
|
+
from csvsmith.cli import main
|
|
79
|
+
f1 = tmp_path / "a.csv"
|
|
80
|
+
f2 = tmp_path / "b.csv"
|
|
81
|
+
out = tmp_path / "out.csv"
|
|
82
|
+
|
|
83
|
+
write_csv(f1, [["id", "name"], ["1", "Alice"]])
|
|
84
|
+
write_csv(f2, [["id", "name"], ["2", "Bob"]])
|
|
85
|
+
|
|
86
|
+
exit_code = main(["strict-concat", str(tmp_path), "-o", str(out)])
|
|
87
|
+
|
|
88
|
+
assert exit_code == 0
|
|
89
|
+
assert out.exists()
|
|
90
|
+
|
|
91
|
+
with out.open(encoding="utf-8") as f:
|
|
92
|
+
reader = csv.reader(f)
|
|
93
|
+
rows = list(reader)
|
|
94
|
+
|
|
95
|
+
assert rows == [
|
|
96
|
+
["file_stem", "id", "name"],
|
|
97
|
+
["a", "1", "Alice"],
|
|
98
|
+
["b", "2", "Bob"],
|
|
99
|
+
]
|
|
100
|
+
|
|
101
|
+
captured = capsys.readouterr()
|
|
102
|
+
assert "Wrote concatenated CSV to:" in captured.out
|
|
@@ -1,124 +0,0 @@
|
|
|
1
|
-
import re
|
|
2
|
-
from typing import Any
|
|
3
|
-
|
|
4
|
-
NON_BREAKING_SPACE = "\xa0"
|
|
5
|
-
SEPARATOR_PATTERN = re.compile(r"[ _\xa0]")
|
|
6
|
-
NUMBER_PATTERN = re.compile(r"^-?(?:\d+|\d*\.\d+)$")
|
|
7
|
-
|
|
8
|
-
|
|
9
|
-
def _normalize_numeric_text(value: Any, *, sep: str, decimal: str) -> str:
|
|
10
|
-
"""
|
|
11
|
-
Normalize a numeric text string for consistent formatting.
|
|
12
|
-
|
|
13
|
-
Converts a given value to a string representation and ensures normalization of numeric formatting,
|
|
14
|
-
such as removing group separators, converting localized decimal separators, and handling negative
|
|
15
|
-
values enclosed in parentheses.
|
|
16
|
-
|
|
17
|
-
:param value: The value to be normalized, which may be of any type.
|
|
18
|
-
:type value: Any
|
|
19
|
-
:param sep: The character used as a group separator in the input value, which will be removed
|
|
20
|
-
during normalization.
|
|
21
|
-
:type sep: str
|
|
22
|
-
:param decimal: The character used as the decimal separator in the input value, which will
|
|
23
|
-
be replaced with a standard period ('.') during normalization.
|
|
24
|
-
:type decimal: str
|
|
25
|
-
:return: A normalized numeric string with consistent formatting.
|
|
26
|
-
:rtype: str
|
|
27
|
-
"""
|
|
28
|
-
numeric_text = str(value).strip()
|
|
29
|
-
|
|
30
|
-
if numeric_text.startswith("(") and numeric_text.endswith(")"):
|
|
31
|
-
numeric_text = f"-{numeric_text[1:-1]}"
|
|
32
|
-
|
|
33
|
-
if sep:
|
|
34
|
-
numeric_text = numeric_text.replace(sep, "")
|
|
35
|
-
|
|
36
|
-
if decimal != ".":
|
|
37
|
-
numeric_text = numeric_text.replace(decimal, ".")
|
|
38
|
-
|
|
39
|
-
return numeric_text
|
|
40
|
-
|
|
41
|
-
|
|
42
|
-
def _has_valid_grouping(numeric_text: str, *, decimal: str) -> bool:
|
|
43
|
-
"""
|
|
44
|
-
Checks whether a numeric text string has valid grouping based on a specified decimal character.
|
|
45
|
-
|
|
46
|
-
This function validates the structure of the given numeric text to determine if it adheres to allowed
|
|
47
|
-
grouping conventions. It ensures the string does not contain invalid or misplaced group separators,
|
|
48
|
-
decimal points, or spacing characters.
|
|
49
|
-
|
|
50
|
-
:param numeric_text: A string representing the numeric text to be validated.
|
|
51
|
-
:type numeric_text: str
|
|
52
|
-
:param decimal: A string representing the character used as the decimal point.
|
|
53
|
-
:type decimal: str
|
|
54
|
-
:return: True if the numeric text satisfies the grouping rules; otherwise, False.
|
|
55
|
-
:rtype: bool
|
|
56
|
-
"""
|
|
57
|
-
if not numeric_text:
|
|
58
|
-
return False
|
|
59
|
-
|
|
60
|
-
unsigned_text = numeric_text[1:] if numeric_text.startswith("-") else numeric_text
|
|
61
|
-
|
|
62
|
-
if unsigned_text.count(decimal) > 1:
|
|
63
|
-
return False
|
|
64
|
-
|
|
65
|
-
integer_text, _, fraction_text = unsigned_text.partition(decimal)
|
|
66
|
-
|
|
67
|
-
if not integer_text and not fraction_text:
|
|
68
|
-
return False
|
|
69
|
-
|
|
70
|
-
for part in (integer_text, fraction_text):
|
|
71
|
-
if not part:
|
|
72
|
-
continue
|
|
73
|
-
if part.startswith("_") or part.endswith("_"):
|
|
74
|
-
return False
|
|
75
|
-
if part.startswith(" ") or part.endswith(" "):
|
|
76
|
-
return False
|
|
77
|
-
if part.startswith(NON_BREAKING_SPACE) or part.endswith(NON_BREAKING_SPACE):
|
|
78
|
-
return False
|
|
79
|
-
if "__" in part or " " in part or NON_BREAKING_SPACE * 2 in part:
|
|
80
|
-
return False
|
|
81
|
-
|
|
82
|
-
stripped_text = SEPARATOR_PATTERN.sub("", numeric_text)
|
|
83
|
-
return bool(NUMBER_PATTERN.fullmatch(stripped_text))
|
|
84
|
-
|
|
85
|
-
|
|
86
|
-
def clean_numeric(
|
|
87
|
-
value: Any, *, sep: str = ",", decimal: str = ".", relaxed: bool = False
|
|
88
|
-
) -> float | Any:
|
|
89
|
-
"""
|
|
90
|
-
Cleans and converts a given input to a float by normalizing its numeric representation.
|
|
91
|
-
Handles separators and decimal points based on the provided arguments. If the input
|
|
92
|
-
value is invalid or cannot be converted, a ValueError is raised unless relaxed mode
|
|
93
|
-
is enabled.
|
|
94
|
-
|
|
95
|
-
:param value: The input value to be cleaned and converted.
|
|
96
|
-
:type value: Any
|
|
97
|
-
:param sep: The character used as a thousands separator in the input value. Default is ",".
|
|
98
|
-
:type sep: str
|
|
99
|
-
:param decimal: The character used as a decimal point in the input value. Default is ".".
|
|
100
|
-
:type decimal: str
|
|
101
|
-
:param relaxed: If True, return the original input when it is not numeric.
|
|
102
|
-
:type relaxed: bool
|
|
103
|
-
:return: The cleaned and converted numeric value as a float, or the original value in relaxed mode.
|
|
104
|
-
:rtype: float | Any
|
|
105
|
-
:raises ValueError: If the input value cannot be converted to a valid number and relaxed is False.
|
|
106
|
-
"""
|
|
107
|
-
if value is None:
|
|
108
|
-
return 0.0
|
|
109
|
-
|
|
110
|
-
normalized_number_text = _normalize_numeric_text(value, sep=sep, decimal=decimal)
|
|
111
|
-
|
|
112
|
-
if not _has_valid_grouping(normalized_number_text, decimal=decimal):
|
|
113
|
-
if relaxed:
|
|
114
|
-
return value
|
|
115
|
-
raise ValueError(f"Could not convert {value!r} to a valid number.")
|
|
116
|
-
|
|
117
|
-
numeric_text = SEPARATOR_PATTERN.sub("", normalized_number_text)
|
|
118
|
-
|
|
119
|
-
try:
|
|
120
|
-
return float(numeric_text)
|
|
121
|
-
except ValueError as exc:
|
|
122
|
-
if relaxed:
|
|
123
|
-
return value
|
|
124
|
-
raise ValueError(f"Could not convert {value!r} to a valid number.") from exc
|
|
@@ -1 +0,0 @@
|
|
|
1
|
-
openpyxl>=3.1
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|