csvsmith 0.7.2__tar.gz → 0.8.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (39) hide show
  1. {csvsmith-0.7.2/src/csvsmith.egg-info → csvsmith-0.8.0}/PKG-INFO +26 -11
  2. {csvsmith-0.7.2 → csvsmith-0.8.0}/README.rst +22 -10
  3. {csvsmith-0.7.2 → csvsmith-0.8.0}/pyproject.toml +7 -1
  4. {csvsmith-0.7.2 → csvsmith-0.8.0}/src/csvsmith/__init__.py +8 -2
  5. {csvsmith-0.7.2 → csvsmith-0.8.0}/src/csvsmith/cli.py +62 -1
  6. csvsmith-0.8.0/src/csvsmith/tools/strict_concat.py +101 -0
  7. csvsmith-0.8.0/src/csvsmith/utils/clean_numeric.py +124 -0
  8. {csvsmith-0.7.2 → csvsmith-0.8.0/src/csvsmith.egg-info}/PKG-INFO +26 -11
  9. {csvsmith-0.7.2 → csvsmith-0.8.0}/src/csvsmith.egg-info/SOURCES.txt +2 -0
  10. csvsmith-0.8.0/src/csvsmith.egg-info/requires.txt +5 -0
  11. {csvsmith-0.7.2 → csvsmith-0.8.0}/tests/test_clean_numeric.py +33 -1
  12. {csvsmith-0.7.2 → csvsmith-0.8.0}/tests/test_cli.py +21 -2
  13. csvsmith-0.8.0/tests/test_strict_concat.py +102 -0
  14. csvsmith-0.7.2/src/csvsmith/utils/clean_numeric.py +0 -124
  15. csvsmith-0.7.2/src/csvsmith.egg-info/requires.txt +0 -1
  16. {csvsmith-0.7.2 → csvsmith-0.8.0}/LICENSE +0 -0
  17. {csvsmith-0.7.2 → csvsmith-0.8.0}/setup.cfg +0 -0
  18. {csvsmith-0.7.2 → csvsmith-0.8.0}/src/csvsmith/tools/__init__.py +0 -0
  19. {csvsmith-0.7.2 → csvsmith-0.8.0}/src/csvsmith/tools/classify.py +0 -0
  20. {csvsmith-0.7.2 → csvsmith-0.8.0}/src/csvsmith/tools/excel2csv.py +0 -0
  21. {csvsmith-0.7.2 → csvsmith-0.8.0}/src/csvsmith/tools/filter_rows.py +0 -0
  22. {csvsmith-0.7.2 → csvsmith-0.8.0}/src/csvsmith/tools/find_matches_in_csv.py +0 -0
  23. {csvsmith-0.7.2 → csvsmith-0.8.0}/src/csvsmith/tools/move_files.py +0 -0
  24. {csvsmith-0.7.2 → csvsmith-0.8.0}/src/csvsmith/tools/row_dedup.py +0 -0
  25. {csvsmith-0.7.2 → csvsmith-0.8.0}/src/csvsmith/utils/__init__.py +0 -0
  26. {csvsmith-0.7.2 → csvsmith-0.8.0}/src/csvsmith/utils/distance.py +0 -0
  27. {csvsmith-0.7.2 → csvsmith-0.8.0}/src/csvsmith/utils/io.py +0 -0
  28. {csvsmith-0.7.2 → csvsmith-0.8.0}/src/csvsmith/utils/normalize.py +0 -0
  29. {csvsmith-0.7.2 → csvsmith-0.8.0}/src/csvsmith.egg-info/dependency_links.txt +0 -0
  30. {csvsmith-0.7.2 → csvsmith-0.8.0}/src/csvsmith.egg-info/entry_points.txt +0 -0
  31. {csvsmith-0.7.2 → csvsmith-0.8.0}/src/csvsmith.egg-info/top_level.txt +0 -0
  32. {csvsmith-0.7.2 → csvsmith-0.8.0}/tests/test_classify.py +0 -0
  33. {csvsmith-0.7.2 → csvsmith-0.8.0}/tests/test_excel2csv.py +0 -0
  34. {csvsmith-0.7.2 → csvsmith-0.8.0}/tests/test_filter_rows.py +0 -0
  35. {csvsmith-0.7.2 → csvsmith-0.8.0}/tests/test_find_matches_in_csv.py +0 -0
  36. {csvsmith-0.7.2 → csvsmith-0.8.0}/tests/test_move_files.py +0 -0
  37. {csvsmith-0.7.2 → csvsmith-0.8.0}/tests/test_normalize.py +0 -0
  38. {csvsmith-0.7.2 → csvsmith-0.8.0}/tests/test_row_dedup.py +0 -0
  39. {csvsmith-0.7.2 → csvsmith-0.8.0}/tests/test_string_distance.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: csvsmith
3
- Version: 0.7.2
3
+ Version: 0.8.0
4
4
  Summary: Small CSV utilities: row deduplication, classification, row filtering, and CLI helpers.
5
5
  Author-email: Eiichi YAMAMOTO <info@yeiichi.com>
6
6
  License: MIT License
@@ -39,6 +39,9 @@ Requires-Python: >=3.10
39
39
  Description-Content-Type: text/x-rst
40
40
  License-File: LICENSE
41
41
  Requires-Dist: openpyxl>=3.1
42
+ Provides-Extra: docs
43
+ Requires-Dist: sphinx<9,>=8; extra == "docs"
44
+ Requires-Dist: furo>=2024.8.6; extra == "docs"
42
45
  Dynamic: license-file
43
46
 
44
47
  csvsmith
@@ -84,6 +87,7 @@ Features
84
87
  - Convert Excel workbooks to CSV
85
88
  - Move files by suffix
86
89
  - Find matching values inside CSV files
90
+ - Concatenate CSV files with identical headers
87
91
  - Use the tools either from Python or from the command line
88
92
 
89
93
  Installation
@@ -111,17 +115,11 @@ You can use the library from Python:
111
115
 
112
116
  .. code-block:: python
113
117
 
114
- from csvsmith import (
115
- clean_numeric,
116
- dedupe_with_report,
117
- excel_to_csv,
118
- find_matches_in_csv,
119
- move_by_suffix,
120
- )
118
+ from csvsmith.utils.clean_numeric import clean_currency_numeric
121
119
 
122
- print(clean_numeric("1,234.56"))
120
+ print(clean_currency_numeric("$1,234.56"))
123
121
 
124
- Or use the command-line interface:
122
+ For command-line usage, use single quotes around values containing ``$``:
125
123
 
126
124
  .. code-block:: console
127
125
 
@@ -138,6 +136,17 @@ Clean numeric values:
138
136
 
139
137
  csvsmith clean-numeric "1,234.56" --sep "," --decimal "."
140
138
 
139
+ Clean currency-prefixed numeric values:
140
+
141
+ .. code-block:: console
142
+
143
+ csvsmith clean-currency-numeric '$1,234.56' --sep "," --decimal "."
144
+
145
+ .. note::
146
+
147
+ Use single quotes for values containing ``$``. Double quotes may trigger
148
+ shell expansion and change the input unexpectedly.
149
+
141
150
  Filter rows in a CSV:
142
151
 
143
152
  .. code-block:: console
@@ -174,6 +183,12 @@ Find matches in a CSV:
174
183
 
175
184
  csvsmith find-matches input.csv target --ignore-case --ignore-whitespace
176
185
 
186
+ Concatenate CSV files:
187
+
188
+ .. code-block:: console
189
+
190
+ csvsmith strict-concat file1.csv file2.csv -o combined.csv
191
+
177
192
  Find matches in a CSV
178
193
  ---------------------
179
194
 
@@ -233,7 +248,7 @@ File and conversion helpers:
233
248
 
234
249
  .. code-block:: python
235
250
 
236
- from csvsmith import excel_to_csv, move_by_suffix
251
+ from csvsmith import excel_to_csv, move_by_suffix, strict_concat_rows, save_csv
237
252
 
238
253
  String comparison utilities:
239
254
 
@@ -41,6 +41,7 @@ Features
41
41
  - Convert Excel workbooks to CSV
42
42
  - Move files by suffix
43
43
  - Find matching values inside CSV files
44
+ - Concatenate CSV files with identical headers
44
45
  - Use the tools either from Python or from the command line
45
46
 
46
47
  Installation
@@ -68,17 +69,11 @@ You can use the library from Python:
68
69
 
69
70
  .. code-block:: python
70
71
 
71
- from csvsmith import (
72
- clean_numeric,
73
- dedupe_with_report,
74
- excel_to_csv,
75
- find_matches_in_csv,
76
- move_by_suffix,
77
- )
72
+ from csvsmith.utils.clean_numeric import clean_currency_numeric
78
73
 
79
- print(clean_numeric("1,234.56"))
74
+ print(clean_currency_numeric("$1,234.56"))
80
75
 
81
- Or use the command-line interface:
76
+ For command-line usage, use single quotes around values containing ``$``:
82
77
 
83
78
  .. code-block:: console
84
79
 
@@ -95,6 +90,17 @@ Clean numeric values:
95
90
 
96
91
  csvsmith clean-numeric "1,234.56" --sep "," --decimal "."
97
92
 
93
+ Clean currency-prefixed numeric values:
94
+
95
+ .. code-block:: console
96
+
97
+ csvsmith clean-currency-numeric '$1,234.56' --sep "," --decimal "."
98
+
99
+ .. note::
100
+
101
+ Use single quotes for values containing ``$``. Double quotes may trigger
102
+ shell expansion and change the input unexpectedly.
103
+
98
104
  Filter rows in a CSV:
99
105
 
100
106
  .. code-block:: console
@@ -131,6 +137,12 @@ Find matches in a CSV:
131
137
 
132
138
  csvsmith find-matches input.csv target --ignore-case --ignore-whitespace
133
139
 
140
+ Concatenate CSV files:
141
+
142
+ .. code-block:: console
143
+
144
+ csvsmith strict-concat file1.csv file2.csv -o combined.csv
145
+
134
146
  Find matches in a CSV
135
147
  ---------------------
136
148
 
@@ -190,7 +202,7 @@ File and conversion helpers:
190
202
 
191
203
  .. code-block:: python
192
204
 
193
- from csvsmith import excel_to_csv, move_by_suffix
205
+ from csvsmith import excel_to_csv, move_by_suffix, strict_concat_rows, save_csv
194
206
 
195
207
  String comparison utilities:
196
208
 
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
4
4
 
5
5
  [project]
6
6
  name = "csvsmith"
7
- version = "0.7.2"
7
+ version = "0.8.0"
8
8
  description = "Small CSV utilities: row deduplication, classification, row filtering, and CLI helpers."
9
9
  readme = "README.rst"
10
10
  requires-python = ">=3.10"
@@ -35,6 +35,12 @@ dependencies = [
35
35
  "openpyxl>=3.1",
36
36
  ]
37
37
 
38
+ [project.optional-dependencies]
39
+ docs = [
40
+ "sphinx>=8,<9",
41
+ "furo>=2024.8.6",
42
+ ]
43
+
38
44
  [project.urls]
39
45
  Homepage = "https://github.com/yeiichi/csvsmith"
40
46
  Repository = "https://github.com/yeiichi/csvsmith"
@@ -16,6 +16,8 @@ Public API:
16
16
  - Relation
17
17
  - Result
18
18
  - analyze_pair
19
+ - strict_concat_rows
20
+ - save_csv
19
21
 
20
22
  Compatibility aliases:
21
23
  - CSVCleaner
@@ -28,10 +30,11 @@ Submodules:
28
30
  - csvsmith.filter_rows
29
31
  - csvsmith.excel2csv
30
32
  - csvsmith.move_files
33
+ - csvsmith.strict_concat
31
34
  - csvsmith.cli (CLI entrypoint)
32
35
  """
33
36
 
34
- __version__ = "0.7.2"
37
+ __version__ = "0.8.0"
35
38
 
36
39
  from .tools.classify import CSVClassifier
37
40
  from .tools.excel2csv import excel_to_csv
@@ -43,6 +46,7 @@ from .tools.row_dedup import (
43
46
  find_duplicate_rows,
44
47
  dedupe_with_report,
45
48
  )
49
+ from .tools.strict_concat import save_csv, strict_concat_rows
46
50
  from .utils.clean_numeric import clean_numeric
47
51
  from .utils.distance import StringDistance, Relation, Result, analyze_pair
48
52
  from .utils.io import (
@@ -68,4 +72,6 @@ __all__ = [
68
72
  "Result",
69
73
  "analyze_pair",
70
74
  "clean_numeric",
71
- ]
75
+ "strict_concat_rows",
76
+ "save_csv",
77
+ ]
@@ -9,13 +9,14 @@ from typing import Optional, Sequence
9
9
  from . import __version__
10
10
  from .tools.classify import CSVClassifier
11
11
  from .tools.excel2csv import excel_to_csv
12
- from .utils.clean_numeric import clean_numeric
12
+ from .utils.clean_numeric import clean_currency_numeric, clean_numeric
13
13
  from .tools.filter_rows import DropRowsBySubstring
14
14
  from .tools.move_files import move_by_suffix, normalize_suffixes
15
15
  from .tools.row_dedup import (
16
16
  dedupe_with_report,
17
17
  find_duplicate_rows,
18
18
  )
19
+ from .tools.strict_concat import save_csv, strict_concat_rows
19
20
  from .tools.find_matches_in_csv import find_matches_in_csv
20
21
  from .utils.distance import analyze_pair
21
22
  from .utils.io import read_csv_rows, write_csv_rows
@@ -149,6 +150,21 @@ def cmd_clean_numeric(args: argparse.Namespace) -> int:
149
150
  return 0
150
151
 
151
152
 
153
+ def cmd_clean_currency_numeric(args: argparse.Namespace) -> int:
154
+ try:
155
+ cleaned = clean_currency_numeric(
156
+ args.value,
157
+ sep=args.sep,
158
+ decimal=args.decimal,
159
+ relaxed=args.relaxed,
160
+ )
161
+ print(cleaned)
162
+ except ValueError as e:
163
+ print(f"Error: {e}", file=sys.stderr)
164
+ return 1
165
+ return 0
166
+
167
+
152
168
  def cmd_find_matches(args: argparse.Namespace) -> int:
153
169
  results = find_matches_in_csv(
154
170
  args.input,
@@ -165,6 +181,23 @@ def cmd_find_matches(args: argparse.Namespace) -> int:
165
181
  return 0
166
182
 
167
183
 
184
+ def cmd_strict_concat(args: argparse.Namespace) -> int:
185
+ input_dir = Path(args.input_dir)
186
+ if not input_dir.is_dir():
187
+ print(f"Error: input directory not found: {input_dir}", file=sys.stderr)
188
+ return 1
189
+
190
+ try:
191
+ rows = strict_concat_rows(input_dir)
192
+ save_csv(rows, args.output)
193
+ except Exception as e:
194
+ print(f"Error: {e}", file=sys.stderr)
195
+ return 1
196
+
197
+ print(f"Wrote concatenated CSV to: {args.output}")
198
+ return 0
199
+
200
+
168
201
  def _add_find_matches_parser(subparsers) -> None:
169
202
  parser = subparsers.add_parser("find-matches", help="Find matches in a CSV file.")
170
203
  parser.add_argument("input", help="Input CSV file.")
@@ -175,6 +208,16 @@ def _add_find_matches_parser(subparsers) -> None:
175
208
  parser.set_defaults(func=cmd_find_matches)
176
209
 
177
210
 
211
+ def _add_strict_concat_parser(subparsers) -> None:
212
+ parser = subparsers.add_parser(
213
+ "strict-concat",
214
+ help="Concatenate CSVs in a directory only when all headers match exactly.",
215
+ )
216
+ parser.add_argument("input_dir", help="Directory containing CSV files to concatenate.")
217
+ parser.add_argument("-o", "--output", required=True, help="Output CSV file path.")
218
+ parser.set_defaults(func=cmd_strict_concat)
219
+
220
+
178
221
  def _add_row_duplicates_parser(subparsers) -> None:
179
222
  parser = subparsers.add_parser("row-duplicates", help="Find duplicate rows in a CSV.")
180
223
  parser.add_argument("input", help="Input CSV file.")
@@ -272,6 +315,22 @@ def _add_clean_numeric_parser(subparsers) -> None:
272
315
  parser.set_defaults(func=cmd_clean_numeric)
273
316
 
274
317
 
318
+ def _add_clean_currency_numeric_parser(subparsers) -> None:
319
+ parser = subparsers.add_parser(
320
+ "clean-currency-numeric",
321
+ help="Clean and convert a currency-prefixed numeric string to float.",
322
+ )
323
+ parser.add_argument("value", help="Numeric value to clean.")
324
+ parser.add_argument("--sep", default=",", help="Group separator (default: ,).")
325
+ parser.add_argument("--decimal", default=".", help="Decimal separator (default: .).")
326
+ parser.add_argument(
327
+ "--relaxed",
328
+ action="store_true",
329
+ help="Return the original input when it is not numeric.",
330
+ )
331
+ parser.set_defaults(func=cmd_clean_currency_numeric)
332
+
333
+
275
334
  def build_parser() -> argparse.ArgumentParser:
276
335
  parser = argparse.ArgumentParser(
277
336
  prog="csvsmith",
@@ -286,6 +345,7 @@ def build_parser() -> argparse.ArgumentParser:
286
345
 
287
346
  _add_row_duplicates_parser(subparsers)
288
347
  _add_find_matches_parser(subparsers)
348
+ _add_strict_concat_parser(subparsers)
289
349
  _add_dedupe_parser(subparsers)
290
350
  _add_classify_parser(subparsers)
291
351
  _add_move_files_parser(subparsers)
@@ -293,6 +353,7 @@ def build_parser() -> argparse.ArgumentParser:
293
353
  _add_drop_rows_parser(subparsers)
294
354
  _add_string_distance_parser(subparsers)
295
355
  _add_clean_numeric_parser(subparsers)
356
+ _add_clean_currency_numeric_parser(subparsers)
296
357
 
297
358
  return parser
298
359
 
@@ -0,0 +1,101 @@
1
+ import csv
2
+ from collections.abc import Iterable
3
+ from pathlib import Path
4
+
5
+
6
+ def find_csvs(csv_dir: Path | str) -> list[Path]:
7
+ """
8
+ Find all CSV files in the specified directory.
9
+
10
+ This function searches for all files with a ``.csv`` extension in the given
11
+ directory and returns a sorted list of their paths.
12
+
13
+ :param csv_dir: The directory to search for CSV files. This can be provided
14
+ as either a ``Path`` object or a string representing the path to the
15
+ directory.
16
+ :type csv_dir: Path | str
17
+ :return: Sorted list of paths to all ``.csv`` files found in the specified
18
+ directory.
19
+ :rtype: list[Path]
20
+ """
21
+ return sorted(Path(csv_dir).glob("*.csv"))
22
+
23
+
24
+ def read_header(csv_path: Path) -> list[str]:
25
+ """
26
+ Reads the header row of a given CSV file.
27
+
28
+ This function opens a CSV file located at the specified path, reads its first
29
+ row, and returns it as a list of strings. The file is assumed to be encoded
30
+ in UTF-8 with optional BOM (Byte Order Mark). If the CSV file is empty, a
31
+ ValueError is raised indicating the problem. The CSV file is expected to be
32
+ opened in read mode with no newline translation.
33
+
34
+ :param csv_path: The path to the CSV file to read the header from.
35
+ :type csv_path: Path
36
+ :return: A list of strings representing the header row of the CSV file.
37
+ :rtype: list[str]
38
+ :raises ValueError: If the CSV file is empty and no header can be retrieved.
39
+ """
40
+ with csv_path.open(encoding="utf-8-sig", newline="") as f:
41
+ reader = csv.reader(f)
42
+ try:
43
+ return next(reader)
44
+ except StopIteration as e:
45
+ raise ValueError(f"Empty CSV: {csv_path}") from e
46
+
47
+
48
+ def _validate_headers_match(csv_paths: list[Path]) -> list[str]:
49
+ """
50
+ Validates that the headers of all provided CSV files match each other. The function compares the
51
+ header of each CSV file in the input list against the header of the first file. If a mismatch
52
+ is encountered, a ValueError is raised indicating the problematic file. The function returns
53
+ the header of the first file if all headers are consistent.
54
+
55
+ :param csv_paths: A list of Path objects representing the file paths to the CSV files to validate.
56
+ :type csv_paths: list[Path]
57
+ :return: A list of strings representing the matched header of the first CSV file.
58
+ :rtype: list[str]
59
+ """
60
+ expected_header = read_header(csv_paths[0])
61
+ for csv_path in csv_paths[1:]:
62
+ if read_header(csv_path) != expected_header:
63
+ raise ValueError(f"Header mismatch: {csv_path}")
64
+ return expected_header
65
+
66
+
67
+ def strict_concat_rows(csv_dir: Path | str) -> list[list[str]]:
68
+ """
69
+ Concatenates rows from multiple CSV files into a list of lists of strings, ensuring
70
+ the headers across all CSV files match. The output includes a new column indicating
71
+ the file stem.
72
+
73
+ :param csv_dir: Directory containing the CSV files or a specific path to a CSV file.
74
+ :type csv_dir: Path | str
75
+ :return: A list of lists, where each inner list represents a row from the concatenated
76
+ CSV files. The first row contains the headers, including a "file_stem" column.
77
+ :rtype: list[list[str]]
78
+ :raises FileNotFoundError: If no CSV files are found in the provided directory.
79
+ """
80
+ csv_paths = find_csvs(csv_dir)
81
+ if not csv_paths:
82
+ raise FileNotFoundError(f"No CSV files found in: {csv_dir}")
83
+
84
+ expected_header = _validate_headers_match(csv_paths)
85
+ out_rows: list[list[str]] = [["file_stem", *expected_header]]
86
+
87
+ for csv_path in csv_paths:
88
+ with csv_path.open(encoding="utf-8-sig", newline="") as f:
89
+ reader = csv.reader(f)
90
+ next(reader) # skip header
91
+ for row in reader:
92
+ out_rows.append([csv_path.stem, *row])
93
+ return out_rows
94
+
95
+
96
+ def save_csv(rows: Iterable[list[str]], out_path: Path | str) -> None:
97
+ """Write rows to out_path."""
98
+ out_path = Path(out_path)
99
+ with out_path.open("w", encoding="utf-8", newline="") as f:
100
+ writer = csv.writer(f)
101
+ writer.writerows(rows)
@@ -0,0 +1,124 @@
1
+ import re
2
+ from typing import Any
3
+
4
+ NON_BREAKING_SPACE = "\xa0"
5
+ SEPARATOR_PATTERN = re.compile(r"[ _\xa0]")
6
+ NUMBER_PATTERN = re.compile(r"^-?(?:\d+|\d*\.\d+)$")
7
+ INVALID_NUMBER_MESSAGE = "Could not convert {value!r} to a valid number."
8
+ CURRENCY_PREFIX_PATTERN = re.compile(r"^[\$€£¥₹]")
9
+
10
+
11
+ def strip_currency_prefix(value: Any) -> Any:
12
+ """
13
+ Remove a single common currency symbol from the start of a value.
14
+ """
15
+ text = str(value).strip()
16
+ if text and CURRENCY_PREFIX_PATTERN.match(text):
17
+ return text[1:].strip()
18
+ return value
19
+
20
+
21
+ def _normalize_numeric_text(value: Any, *, sep: str, decimal: str) -> str:
22
+ """
23
+ Normalize a numeric text string for consistent formatting.
24
+
25
+ Converts a given value to a string representation and ensures normalization of numeric formatting,
26
+ such as removing group separators, converting localized decimal separators, and handling negative
27
+ values enclosed in parentheses.
28
+ """
29
+ numeric_text = str(value).strip()
30
+
31
+ if numeric_text.startswith("(") and numeric_text.endswith(")"):
32
+ numeric_text = f"-{numeric_text[1:-1]}"
33
+
34
+ if sep:
35
+ numeric_text = numeric_text.replace(sep, "")
36
+
37
+ if decimal != ".":
38
+ numeric_text = numeric_text.replace(decimal, ".")
39
+
40
+ return numeric_text
41
+
42
+
43
+ def _has_valid_grouping(numeric_text: str, *, decimal: str) -> bool:
44
+ """
45
+ Checks whether a numeric text string has valid grouping based on a specified decimal character.
46
+ """
47
+ if not numeric_text:
48
+ return False
49
+
50
+ unsigned_text = numeric_text[1:] if numeric_text.startswith("-") else numeric_text
51
+
52
+ if unsigned_text.count(decimal) > 1:
53
+ return False
54
+
55
+ integer_text, _, fraction_text = unsigned_text.partition(decimal)
56
+
57
+ if not integer_text and not fraction_text:
58
+ return False
59
+
60
+ for part in (integer_text, fraction_text):
61
+ if not part:
62
+ continue
63
+ if part.startswith("_") or part.endswith("_"):
64
+ return False
65
+ if part.startswith(" ") or part.endswith(" "):
66
+ return False
67
+ if part.startswith(NON_BREAKING_SPACE) or part.endswith(NON_BREAKING_SPACE):
68
+ return False
69
+ if "__" in part or " " in part or NON_BREAKING_SPACE * 2 in part:
70
+ return False
71
+
72
+ stripped_text = SEPARATOR_PATTERN.sub("", numeric_text)
73
+ return bool(NUMBER_PATTERN.fullmatch(stripped_text))
74
+
75
+
76
+ def _strip_group_separators(numeric_text: str) -> str:
77
+ return SEPARATOR_PATTERN.sub("", numeric_text)
78
+
79
+
80
+ def _invalid_number_error(value: Any) -> ValueError:
81
+ return ValueError(INVALID_NUMBER_MESSAGE.format(value=value))
82
+
83
+
84
+ def clean_numeric(
85
+ value: Any, *, sep: str = ",", decimal: str = ".", relaxed: bool = False
86
+ ) -> float | Any:
87
+ """
88
+ Cleans and converts a given input to a float by normalizing its numeric representation.
89
+ """
90
+ if value is None:
91
+ return 0.0
92
+
93
+ normalized_text = _normalize_numeric_text(value, sep=sep, decimal=decimal)
94
+
95
+ if not _has_valid_grouping(normalized_text, decimal=decimal):
96
+ if relaxed:
97
+ return value
98
+ raise _invalid_number_error(value)
99
+
100
+ candidate_text = _strip_group_separators(normalized_text)
101
+
102
+ try:
103
+ return float(candidate_text)
104
+ except ValueError as exc:
105
+ if relaxed:
106
+ return value
107
+ raise _invalid_number_error(value) from exc
108
+
109
+
110
+ def clean_currency_numeric(
111
+ value: Any, *, sep: str = ",", decimal: str = ".", relaxed: bool = False
112
+ ) -> float | Any:
113
+ """
114
+ Cleans and converts a currency-prefixed numeric string to a float.
115
+ """
116
+ if value is None:
117
+ return 0.0
118
+
119
+ return clean_numeric(
120
+ strip_currency_prefix(value),
121
+ sep=sep,
122
+ decimal=decimal,
123
+ relaxed=relaxed,
124
+ )
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: csvsmith
3
- Version: 0.7.2
3
+ Version: 0.8.0
4
4
  Summary: Small CSV utilities: row deduplication, classification, row filtering, and CLI helpers.
5
5
  Author-email: Eiichi YAMAMOTO <info@yeiichi.com>
6
6
  License: MIT License
@@ -39,6 +39,9 @@ Requires-Python: >=3.10
39
39
  Description-Content-Type: text/x-rst
40
40
  License-File: LICENSE
41
41
  Requires-Dist: openpyxl>=3.1
42
+ Provides-Extra: docs
43
+ Requires-Dist: sphinx<9,>=8; extra == "docs"
44
+ Requires-Dist: furo>=2024.8.6; extra == "docs"
42
45
  Dynamic: license-file
43
46
 
44
47
  csvsmith
@@ -84,6 +87,7 @@ Features
84
87
  - Convert Excel workbooks to CSV
85
88
  - Move files by suffix
86
89
  - Find matching values inside CSV files
90
+ - Concatenate CSV files with identical headers
87
91
  - Use the tools either from Python or from the command line
88
92
 
89
93
  Installation
@@ -111,17 +115,11 @@ You can use the library from Python:
111
115
 
112
116
  .. code-block:: python
113
117
 
114
- from csvsmith import (
115
- clean_numeric,
116
- dedupe_with_report,
117
- excel_to_csv,
118
- find_matches_in_csv,
119
- move_by_suffix,
120
- )
118
+ from csvsmith.utils.clean_numeric import clean_currency_numeric
121
119
 
122
- print(clean_numeric("1,234.56"))
120
+ print(clean_currency_numeric("$1,234.56"))
123
121
 
124
- Or use the command-line interface:
122
+ For command-line usage, use single quotes around values containing ``$``:
125
123
 
126
124
  .. code-block:: console
127
125
 
@@ -138,6 +136,17 @@ Clean numeric values:
138
136
 
139
137
  csvsmith clean-numeric "1,234.56" --sep "," --decimal "."
140
138
 
139
+ Clean currency-prefixed numeric values:
140
+
141
+ .. code-block:: console
142
+
143
+ csvsmith clean-currency-numeric '$1,234.56' --sep "," --decimal "."
144
+
145
+ .. note::
146
+
147
+ Use single quotes for values containing ``$``. Double quotes may trigger
148
+ shell expansion and change the input unexpectedly.
149
+
141
150
  Filter rows in a CSV:
142
151
 
143
152
  .. code-block:: console
@@ -174,6 +183,12 @@ Find matches in a CSV:
174
183
 
175
184
  csvsmith find-matches input.csv target --ignore-case --ignore-whitespace
176
185
 
186
+ Concatenate CSV files:
187
+
188
+ .. code-block:: console
189
+
190
+ csvsmith strict-concat file1.csv file2.csv -o combined.csv
191
+
177
192
  Find matches in a CSV
178
193
  ---------------------
179
194
 
@@ -233,7 +248,7 @@ File and conversion helpers:
233
248
 
234
249
  .. code-block:: python
235
250
 
236
- from csvsmith import excel_to_csv, move_by_suffix
251
+ from csvsmith import excel_to_csv, move_by_suffix, strict_concat_rows, save_csv
237
252
 
238
253
  String comparison utilities:
239
254
 
@@ -16,6 +16,7 @@ src/csvsmith/tools/filter_rows.py
16
16
  src/csvsmith/tools/find_matches_in_csv.py
17
17
  src/csvsmith/tools/move_files.py
18
18
  src/csvsmith/tools/row_dedup.py
19
+ src/csvsmith/tools/strict_concat.py
19
20
  src/csvsmith/utils/__init__.py
20
21
  src/csvsmith/utils/clean_numeric.py
21
22
  src/csvsmith/utils/distance.py
@@ -30,4 +31,5 @@ tests/test_find_matches_in_csv.py
30
31
  tests/test_move_files.py
31
32
  tests/test_normalize.py
32
33
  tests/test_row_dedup.py
34
+ tests/test_strict_concat.py
33
35
  tests/test_string_distance.py
@@ -0,0 +1,5 @@
1
+ openpyxl>=3.1
2
+
3
+ [docs]
4
+ sphinx<9,>=8
5
+ furo>=2024.8.6
@@ -1,5 +1,12 @@
1
1
  import pytest
2
- from csvsmith.utils.clean_numeric import clean_numeric
2
+ from csvsmith.cli import build_parser, main
3
+ from csvsmith.utils.clean_numeric import clean_currency_numeric, clean_numeric
4
+
5
+
6
+ def test_main_help():
7
+ with pytest.raises(SystemExit) as excinfo:
8
+ main(["--help"])
9
+ assert excinfo.value.code == 0
3
10
 
4
11
 
5
12
  def test_clean_numeric_with_valid_integer_string() -> None:
@@ -34,3 +41,28 @@ def test_clean_numeric_with_multiple_decimal_points_raises_valueerror() -> None:
34
41
 
35
42
  def test_clean_numeric_with_none_value() -> None:
36
43
  assert clean_numeric(None) == 0.0
44
+
45
+
46
+ def test_clean_currency_numeric_with_currency_prefix() -> None:
47
+ assert clean_currency_numeric("$1,000") == 1000.0
48
+
49
+
50
+ def test_clean_currency_numeric_with_euro_prefix() -> None:
51
+ assert clean_currency_numeric("€1,000.50") == 1000.5
52
+
53
+
54
+ def test_clean_numeric_still_rejects_currency_prefix() -> None:
55
+ with pytest.raises(ValueError, match=r"Could not convert '\$1,000'.*"):
56
+ clean_numeric("$1,000")
57
+
58
+
59
+ def test_cli_parses_clean_currency_numeric_command():
60
+ parser = build_parser()
61
+ args = parser.parse_args(
62
+ ["clean-currency-numeric", "$1,234.56", "--sep", ",", "--decimal", "."]
63
+ )
64
+
65
+ assert args.command == "clean-currency-numeric"
66
+ assert args.value == "$1,234.56"
67
+ assert args.sep == ","
68
+ assert args.decimal == "."
@@ -138,12 +138,22 @@ def test_cli_parses_clean_numeric_command():
138
138
  assert args.decimal == "."
139
139
 
140
140
 
141
- def test_cli_parses_clean_numeric_command_with_relaxed_mode():
141
+ def test_cli_parses_clean_currency_numeric_command():
142
142
  parser = build_parser()
143
143
  args = parser.parse_args(
144
- ["clean-numeric", "not-a-number", "--relaxed"]
144
+ ["clean-currency-numeric", "$1,234.56", "--sep", ",", "--decimal", "."]
145
145
  )
146
146
 
147
+ assert args.command == "clean-currency-numeric"
148
+ assert args.value == "$1,234.56"
149
+ assert args.sep == ","
150
+ assert args.decimal == "."
151
+
152
+
153
+ def test_cli_parses_clean_numeric_command_with_relaxed_mode():
154
+ parser = build_parser()
155
+ args = parser.parse_args(["clean-numeric", "not-a-number", "--relaxed"])
156
+
147
157
  assert args.command == "clean-numeric"
148
158
  assert args.value == "not-a-number"
149
159
  assert args.relaxed is True
@@ -178,3 +188,12 @@ def test_cli_parses_find_matches_command():
178
188
  assert args.ignore_case is True
179
189
  assert args.ignore_whitespace is True
180
190
  assert args.no_nfkc is True
191
+
192
+
193
+ def test_cli_parses_strict_concat_command():
194
+ parser = build_parser()
195
+ args = parser.parse_args(["strict-concat", "input_dir", "-o", "output.csv"])
196
+
197
+ assert args.command == "strict-concat"
198
+ assert args.input_dir == "input_dir"
199
+ assert args.output == "output.csv"
@@ -0,0 +1,102 @@
1
+ import csv
2
+ from pathlib import Path
3
+
4
+ import pytest
5
+
6
+ from csvsmith.tools.strict_concat import strict_concat_rows
7
+
8
+
9
+ def write_csv(path: Path, rows: list[list[str]]):
10
+ with path.open("w", newline="", encoding="utf-8") as f:
11
+ writer = csv.writer(f)
12
+ writer.writerows(rows)
13
+
14
+
15
+ def test_strict_concat_happy(tmp_path):
16
+ f1 = tmp_path / "a.csv"
17
+ f2 = tmp_path / "b.csv"
18
+
19
+ write_csv(f1, [["id", "name"], ["1", "Alice"]])
20
+ write_csv(f2, [["id", "name"], ["2", "Bob"]])
21
+
22
+ rows = strict_concat_rows(tmp_path)
23
+
24
+ assert rows == [
25
+ ["file_stem", "id", "name"],
26
+ ["a", "1", "Alice"],
27
+ ["b", "2", "Bob"],
28
+ ]
29
+
30
+
31
+ def test_strict_concat_header_mismatch(tmp_path):
32
+ write_csv(tmp_path / "a.csv", [["id", "name"], ["1", "Alice"]])
33
+ write_csv(tmp_path / "b.csv", [["id", "age"], ["2", "30"]])
34
+
35
+ with pytest.raises(ValueError, match="Header mismatch"):
36
+ strict_concat_rows(tmp_path)
37
+
38
+
39
+ def test_strict_concat_empty_dir(tmp_path):
40
+ with pytest.raises(FileNotFoundError):
41
+ strict_concat_rows(tmp_path)
42
+
43
+ def test_strict_concat_with_non_csv_files(tmp_path):
44
+ # Create sample CSV files.
45
+ write_csv(tmp_path / "a.csv", [["id", "name"], ["1", "Alice"]])
46
+ write_csv(tmp_path / "b.csv", [["id", "name"], ["2", "Bob"]])
47
+
48
+ # Create some non-CSV files in the same directory.
49
+ (tmp_path / "file1.txt").write_text("This is a text file.", encoding="utf-8")
50
+ (tmp_path / "file2.doc").write_text("This is a Word document.", encoding="utf-8")
51
+
52
+ # Call function and check results ignore non-CSV files.
53
+ rows = strict_concat_rows(tmp_path)
54
+ assert rows == [
55
+ ["file_stem", "id", "name"],
56
+ ["a", "1", "Alice"],
57
+ ["b", "2", "Bob"],
58
+ ]
59
+
60
+
61
+ def test_strict_concat_empty_csv(tmp_path):
62
+ (tmp_path / "a.csv").write_text("", encoding="utf-8")
63
+
64
+ with pytest.raises(ValueError, match="Empty CSV"):
65
+ strict_concat_rows(tmp_path)
66
+
67
+
68
+ def test_header_only_files(tmp_path):
69
+ write_csv(tmp_path / "a.csv", [["id", "name"]])
70
+ write_csv(tmp_path / "b.csv", [["id", "name"]])
71
+
72
+ rows = strict_concat_rows(tmp_path)
73
+
74
+ assert rows == [["file_stem", "id", "name"]]
75
+
76
+
77
+ def test_strict_concat_cli(tmp_path, capsys):
78
+ from csvsmith.cli import main
79
+ f1 = tmp_path / "a.csv"
80
+ f2 = tmp_path / "b.csv"
81
+ out = tmp_path / "out.csv"
82
+
83
+ write_csv(f1, [["id", "name"], ["1", "Alice"]])
84
+ write_csv(f2, [["id", "name"], ["2", "Bob"]])
85
+
86
+ exit_code = main(["strict-concat", str(tmp_path), "-o", str(out)])
87
+
88
+ assert exit_code == 0
89
+ assert out.exists()
90
+
91
+ with out.open(encoding="utf-8") as f:
92
+ reader = csv.reader(f)
93
+ rows = list(reader)
94
+
95
+ assert rows == [
96
+ ["file_stem", "id", "name"],
97
+ ["a", "1", "Alice"],
98
+ ["b", "2", "Bob"],
99
+ ]
100
+
101
+ captured = capsys.readouterr()
102
+ assert "Wrote concatenated CSV to:" in captured.out
@@ -1,124 +0,0 @@
1
- import re
2
- from typing import Any
3
-
4
- NON_BREAKING_SPACE = "\xa0"
5
- SEPARATOR_PATTERN = re.compile(r"[ _\xa0]")
6
- NUMBER_PATTERN = re.compile(r"^-?(?:\d+|\d*\.\d+)$")
7
-
8
-
9
- def _normalize_numeric_text(value: Any, *, sep: str, decimal: str) -> str:
10
- """
11
- Normalize a numeric text string for consistent formatting.
12
-
13
- Converts a given value to a string representation and ensures normalization of numeric formatting,
14
- such as removing group separators, converting localized decimal separators, and handling negative
15
- values enclosed in parentheses.
16
-
17
- :param value: The value to be normalized, which may be of any type.
18
- :type value: Any
19
- :param sep: The character used as a group separator in the input value, which will be removed
20
- during normalization.
21
- :type sep: str
22
- :param decimal: The character used as the decimal separator in the input value, which will
23
- be replaced with a standard period ('.') during normalization.
24
- :type decimal: str
25
- :return: A normalized numeric string with consistent formatting.
26
- :rtype: str
27
- """
28
- numeric_text = str(value).strip()
29
-
30
- if numeric_text.startswith("(") and numeric_text.endswith(")"):
31
- numeric_text = f"-{numeric_text[1:-1]}"
32
-
33
- if sep:
34
- numeric_text = numeric_text.replace(sep, "")
35
-
36
- if decimal != ".":
37
- numeric_text = numeric_text.replace(decimal, ".")
38
-
39
- return numeric_text
40
-
41
-
42
- def _has_valid_grouping(numeric_text: str, *, decimal: str) -> bool:
43
- """
44
- Checks whether a numeric text string has valid grouping based on a specified decimal character.
45
-
46
- This function validates the structure of the given numeric text to determine if it adheres to allowed
47
- grouping conventions. It ensures the string does not contain invalid or misplaced group separators,
48
- decimal points, or spacing characters.
49
-
50
- :param numeric_text: A string representing the numeric text to be validated.
51
- :type numeric_text: str
52
- :param decimal: A string representing the character used as the decimal point.
53
- :type decimal: str
54
- :return: True if the numeric text satisfies the grouping rules; otherwise, False.
55
- :rtype: bool
56
- """
57
- if not numeric_text:
58
- return False
59
-
60
- unsigned_text = numeric_text[1:] if numeric_text.startswith("-") else numeric_text
61
-
62
- if unsigned_text.count(decimal) > 1:
63
- return False
64
-
65
- integer_text, _, fraction_text = unsigned_text.partition(decimal)
66
-
67
- if not integer_text and not fraction_text:
68
- return False
69
-
70
- for part in (integer_text, fraction_text):
71
- if not part:
72
- continue
73
- if part.startswith("_") or part.endswith("_"):
74
- return False
75
- if part.startswith(" ") or part.endswith(" "):
76
- return False
77
- if part.startswith(NON_BREAKING_SPACE) or part.endswith(NON_BREAKING_SPACE):
78
- return False
79
- if "__" in part or " " in part or NON_BREAKING_SPACE * 2 in part:
80
- return False
81
-
82
- stripped_text = SEPARATOR_PATTERN.sub("", numeric_text)
83
- return bool(NUMBER_PATTERN.fullmatch(stripped_text))
84
-
85
-
86
- def clean_numeric(
87
- value: Any, *, sep: str = ",", decimal: str = ".", relaxed: bool = False
88
- ) -> float | Any:
89
- """
90
- Cleans and converts a given input to a float by normalizing its numeric representation.
91
- Handles separators and decimal points based on the provided arguments. If the input
92
- value is invalid or cannot be converted, a ValueError is raised unless relaxed mode
93
- is enabled.
94
-
95
- :param value: The input value to be cleaned and converted.
96
- :type value: Any
97
- :param sep: The character used as a thousands separator in the input value. Default is ",".
98
- :type sep: str
99
- :param decimal: The character used as a decimal point in the input value. Default is ".".
100
- :type decimal: str
101
- :param relaxed: If True, return the original input when it is not numeric.
102
- :type relaxed: bool
103
- :return: The cleaned and converted numeric value as a float, or the original value in relaxed mode.
104
- :rtype: float | Any
105
- :raises ValueError: If the input value cannot be converted to a valid number and relaxed is False.
106
- """
107
- if value is None:
108
- return 0.0
109
-
110
- normalized_number_text = _normalize_numeric_text(value, sep=sep, decimal=decimal)
111
-
112
- if not _has_valid_grouping(normalized_number_text, decimal=decimal):
113
- if relaxed:
114
- return value
115
- raise ValueError(f"Could not convert {value!r} to a valid number.")
116
-
117
- numeric_text = SEPARATOR_PATTERN.sub("", normalized_number_text)
118
-
119
- try:
120
- return float(numeric_text)
121
- except ValueError as exc:
122
- if relaxed:
123
- return value
124
- raise ValueError(f"Could not convert {value!r} to a valid number.") from exc
@@ -1 +0,0 @@
1
- openpyxl>=3.1
File without changes
File without changes