csvsmith 0.7.2__tar.gz → 0.7.3__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {csvsmith-0.7.2/src/csvsmith.egg-info → csvsmith-0.7.3}/PKG-INFO +18 -10
- {csvsmith-0.7.2 → csvsmith-0.7.3}/README.rst +14 -9
- {csvsmith-0.7.2 → csvsmith-0.7.3}/pyproject.toml +7 -1
- {csvsmith-0.7.2 → csvsmith-0.7.3}/src/csvsmith/__init__.py +1 -1
- {csvsmith-0.7.2 → csvsmith-0.7.3}/src/csvsmith/cli.py +33 -1
- csvsmith-0.7.3/src/csvsmith/utils/clean_numeric.py +124 -0
- {csvsmith-0.7.2 → csvsmith-0.7.3/src/csvsmith.egg-info}/PKG-INFO +18 -10
- csvsmith-0.7.3/src/csvsmith.egg-info/requires.txt +5 -0
- {csvsmith-0.7.2 → csvsmith-0.7.3}/tests/test_clean_numeric.py +33 -1
- {csvsmith-0.7.2 → csvsmith-0.7.3}/tests/test_cli.py +13 -3
- csvsmith-0.7.2/src/csvsmith/utils/clean_numeric.py +0 -124
- csvsmith-0.7.2/src/csvsmith.egg-info/requires.txt +0 -1
- {csvsmith-0.7.2 → csvsmith-0.7.3}/LICENSE +0 -0
- {csvsmith-0.7.2 → csvsmith-0.7.3}/setup.cfg +0 -0
- {csvsmith-0.7.2 → csvsmith-0.7.3}/src/csvsmith/tools/__init__.py +0 -0
- {csvsmith-0.7.2 → csvsmith-0.7.3}/src/csvsmith/tools/classify.py +0 -0
- {csvsmith-0.7.2 → csvsmith-0.7.3}/src/csvsmith/tools/excel2csv.py +0 -0
- {csvsmith-0.7.2 → csvsmith-0.7.3}/src/csvsmith/tools/filter_rows.py +0 -0
- {csvsmith-0.7.2 → csvsmith-0.7.3}/src/csvsmith/tools/find_matches_in_csv.py +0 -0
- {csvsmith-0.7.2 → csvsmith-0.7.3}/src/csvsmith/tools/move_files.py +0 -0
- {csvsmith-0.7.2 → csvsmith-0.7.3}/src/csvsmith/tools/row_dedup.py +0 -0
- {csvsmith-0.7.2 → csvsmith-0.7.3}/src/csvsmith/utils/__init__.py +0 -0
- {csvsmith-0.7.2 → csvsmith-0.7.3}/src/csvsmith/utils/distance.py +0 -0
- {csvsmith-0.7.2 → csvsmith-0.7.3}/src/csvsmith/utils/io.py +0 -0
- {csvsmith-0.7.2 → csvsmith-0.7.3}/src/csvsmith/utils/normalize.py +0 -0
- {csvsmith-0.7.2 → csvsmith-0.7.3}/src/csvsmith.egg-info/SOURCES.txt +0 -0
- {csvsmith-0.7.2 → csvsmith-0.7.3}/src/csvsmith.egg-info/dependency_links.txt +0 -0
- {csvsmith-0.7.2 → csvsmith-0.7.3}/src/csvsmith.egg-info/entry_points.txt +0 -0
- {csvsmith-0.7.2 → csvsmith-0.7.3}/src/csvsmith.egg-info/top_level.txt +0 -0
- {csvsmith-0.7.2 → csvsmith-0.7.3}/tests/test_classify.py +0 -0
- {csvsmith-0.7.2 → csvsmith-0.7.3}/tests/test_excel2csv.py +0 -0
- {csvsmith-0.7.2 → csvsmith-0.7.3}/tests/test_filter_rows.py +0 -0
- {csvsmith-0.7.2 → csvsmith-0.7.3}/tests/test_find_matches_in_csv.py +0 -0
- {csvsmith-0.7.2 → csvsmith-0.7.3}/tests/test_move_files.py +0 -0
- {csvsmith-0.7.2 → csvsmith-0.7.3}/tests/test_normalize.py +0 -0
- {csvsmith-0.7.2 → csvsmith-0.7.3}/tests/test_row_dedup.py +0 -0
- {csvsmith-0.7.2 → csvsmith-0.7.3}/tests/test_string_distance.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: csvsmith
|
|
3
|
-
Version: 0.7.
|
|
3
|
+
Version: 0.7.3
|
|
4
4
|
Summary: Small CSV utilities: row deduplication, classification, row filtering, and CLI helpers.
|
|
5
5
|
Author-email: Eiichi YAMAMOTO <info@yeiichi.com>
|
|
6
6
|
License: MIT License
|
|
@@ -39,6 +39,9 @@ Requires-Python: >=3.10
|
|
|
39
39
|
Description-Content-Type: text/x-rst
|
|
40
40
|
License-File: LICENSE
|
|
41
41
|
Requires-Dist: openpyxl>=3.1
|
|
42
|
+
Provides-Extra: docs
|
|
43
|
+
Requires-Dist: sphinx<9,>=8; extra == "docs"
|
|
44
|
+
Requires-Dist: furo>=2024.8.6; extra == "docs"
|
|
42
45
|
Dynamic: license-file
|
|
43
46
|
|
|
44
47
|
csvsmith
|
|
@@ -111,17 +114,11 @@ You can use the library from Python:
|
|
|
111
114
|
|
|
112
115
|
.. code-block:: python
|
|
113
116
|
|
|
114
|
-
from csvsmith import
|
|
115
|
-
clean_numeric,
|
|
116
|
-
dedupe_with_report,
|
|
117
|
-
excel_to_csv,
|
|
118
|
-
find_matches_in_csv,
|
|
119
|
-
move_by_suffix,
|
|
120
|
-
)
|
|
117
|
+
from csvsmith.utils.clean_numeric import clean_currency_numeric
|
|
121
118
|
|
|
122
|
-
print(
|
|
119
|
+
print(clean_currency_numeric("$1,234.56"))
|
|
123
120
|
|
|
124
|
-
|
|
121
|
+
For command-line usage, use single quotes around values containing ``$``:
|
|
125
122
|
|
|
126
123
|
.. code-block:: console
|
|
127
124
|
|
|
@@ -138,6 +135,17 @@ Clean numeric values:
|
|
|
138
135
|
|
|
139
136
|
csvsmith clean-numeric "1,234.56" --sep "," --decimal "."
|
|
140
137
|
|
|
138
|
+
Clean currency-prefixed numeric values:
|
|
139
|
+
|
|
140
|
+
.. code-block:: console
|
|
141
|
+
|
|
142
|
+
csvsmith clean-currency-numeric '$1,234.56' --sep "," --decimal "."
|
|
143
|
+
|
|
144
|
+
.. note::
|
|
145
|
+
|
|
146
|
+
Use single quotes for values containing ``$``. Double quotes may trigger
|
|
147
|
+
shell expansion and change the input unexpectedly.
|
|
148
|
+
|
|
141
149
|
Filter rows in a CSV:
|
|
142
150
|
|
|
143
151
|
.. code-block:: console
|
|
@@ -68,17 +68,11 @@ You can use the library from Python:
|
|
|
68
68
|
|
|
69
69
|
.. code-block:: python
|
|
70
70
|
|
|
71
|
-
from csvsmith import
|
|
72
|
-
clean_numeric,
|
|
73
|
-
dedupe_with_report,
|
|
74
|
-
excel_to_csv,
|
|
75
|
-
find_matches_in_csv,
|
|
76
|
-
move_by_suffix,
|
|
77
|
-
)
|
|
71
|
+
from csvsmith.utils.clean_numeric import clean_currency_numeric
|
|
78
72
|
|
|
79
|
-
print(
|
|
73
|
+
print(clean_currency_numeric("$1,234.56"))
|
|
80
74
|
|
|
81
|
-
|
|
75
|
+
For command-line usage, use single quotes around values containing ``$``:
|
|
82
76
|
|
|
83
77
|
.. code-block:: console
|
|
84
78
|
|
|
@@ -95,6 +89,17 @@ Clean numeric values:
|
|
|
95
89
|
|
|
96
90
|
csvsmith clean-numeric "1,234.56" --sep "," --decimal "."
|
|
97
91
|
|
|
92
|
+
Clean currency-prefixed numeric values:
|
|
93
|
+
|
|
94
|
+
.. code-block:: console
|
|
95
|
+
|
|
96
|
+
csvsmith clean-currency-numeric '$1,234.56' --sep "," --decimal "."
|
|
97
|
+
|
|
98
|
+
.. note::
|
|
99
|
+
|
|
100
|
+
Use single quotes for values containing ``$``. Double quotes may trigger
|
|
101
|
+
shell expansion and change the input unexpectedly.
|
|
102
|
+
|
|
98
103
|
Filter rows in a CSV:
|
|
99
104
|
|
|
100
105
|
.. code-block:: console
|
|
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
|
|
|
4
4
|
|
|
5
5
|
[project]
|
|
6
6
|
name = "csvsmith"
|
|
7
|
-
version = "0.7.
|
|
7
|
+
version = "0.7.3"
|
|
8
8
|
description = "Small CSV utilities: row deduplication, classification, row filtering, and CLI helpers."
|
|
9
9
|
readme = "README.rst"
|
|
10
10
|
requires-python = ">=3.10"
|
|
@@ -35,6 +35,12 @@ dependencies = [
|
|
|
35
35
|
"openpyxl>=3.1",
|
|
36
36
|
]
|
|
37
37
|
|
|
38
|
+
[project.optional-dependencies]
|
|
39
|
+
docs = [
|
|
40
|
+
"sphinx>=8,<9",
|
|
41
|
+
"furo>=2024.8.6",
|
|
42
|
+
]
|
|
43
|
+
|
|
38
44
|
[project.urls]
|
|
39
45
|
Homepage = "https://github.com/yeiichi/csvsmith"
|
|
40
46
|
Repository = "https://github.com/yeiichi/csvsmith"
|
|
@@ -9,7 +9,7 @@ from typing import Optional, Sequence
|
|
|
9
9
|
from . import __version__
|
|
10
10
|
from .tools.classify import CSVClassifier
|
|
11
11
|
from .tools.excel2csv import excel_to_csv
|
|
12
|
-
from .utils.clean_numeric import clean_numeric
|
|
12
|
+
from .utils.clean_numeric import clean_currency_numeric, clean_numeric
|
|
13
13
|
from .tools.filter_rows import DropRowsBySubstring
|
|
14
14
|
from .tools.move_files import move_by_suffix, normalize_suffixes
|
|
15
15
|
from .tools.row_dedup import (
|
|
@@ -149,6 +149,21 @@ def cmd_clean_numeric(args: argparse.Namespace) -> int:
|
|
|
149
149
|
return 0
|
|
150
150
|
|
|
151
151
|
|
|
152
|
+
def cmd_clean_currency_numeric(args: argparse.Namespace) -> int:
|
|
153
|
+
try:
|
|
154
|
+
cleaned = clean_currency_numeric(
|
|
155
|
+
args.value,
|
|
156
|
+
sep=args.sep,
|
|
157
|
+
decimal=args.decimal,
|
|
158
|
+
relaxed=args.relaxed,
|
|
159
|
+
)
|
|
160
|
+
print(cleaned)
|
|
161
|
+
except ValueError as e:
|
|
162
|
+
print(f"Error: {e}", file=sys.stderr)
|
|
163
|
+
return 1
|
|
164
|
+
return 0
|
|
165
|
+
|
|
166
|
+
|
|
152
167
|
def cmd_find_matches(args: argparse.Namespace) -> int:
|
|
153
168
|
results = find_matches_in_csv(
|
|
154
169
|
args.input,
|
|
@@ -272,6 +287,22 @@ def _add_clean_numeric_parser(subparsers) -> None:
|
|
|
272
287
|
parser.set_defaults(func=cmd_clean_numeric)
|
|
273
288
|
|
|
274
289
|
|
|
290
|
+
def _add_clean_currency_numeric_parser(subparsers) -> None:
|
|
291
|
+
parser = subparsers.add_parser(
|
|
292
|
+
"clean-currency-numeric",
|
|
293
|
+
help="Clean and convert a currency-prefixed numeric string to float.",
|
|
294
|
+
)
|
|
295
|
+
parser.add_argument("value", help="Numeric value to clean.")
|
|
296
|
+
parser.add_argument("--sep", default=",", help="Group separator (default: ,).")
|
|
297
|
+
parser.add_argument("--decimal", default=".", help="Decimal separator (default: .).")
|
|
298
|
+
parser.add_argument(
|
|
299
|
+
"--relaxed",
|
|
300
|
+
action="store_true",
|
|
301
|
+
help="Return the original input when it is not numeric.",
|
|
302
|
+
)
|
|
303
|
+
parser.set_defaults(func=cmd_clean_currency_numeric)
|
|
304
|
+
|
|
305
|
+
|
|
275
306
|
def build_parser() -> argparse.ArgumentParser:
|
|
276
307
|
parser = argparse.ArgumentParser(
|
|
277
308
|
prog="csvsmith",
|
|
@@ -293,6 +324,7 @@ def build_parser() -> argparse.ArgumentParser:
|
|
|
293
324
|
_add_drop_rows_parser(subparsers)
|
|
294
325
|
_add_string_distance_parser(subparsers)
|
|
295
326
|
_add_clean_numeric_parser(subparsers)
|
|
327
|
+
_add_clean_currency_numeric_parser(subparsers)
|
|
296
328
|
|
|
297
329
|
return parser
|
|
298
330
|
|
|
@@ -0,0 +1,124 @@
|
|
|
1
|
+
import re
|
|
2
|
+
from typing import Any
|
|
3
|
+
|
|
4
|
+
NON_BREAKING_SPACE = "\xa0"
|
|
5
|
+
SEPARATOR_PATTERN = re.compile(r"[ _\xa0]")
|
|
6
|
+
NUMBER_PATTERN = re.compile(r"^-?(?:\d+|\d*\.\d+)$")
|
|
7
|
+
INVALID_NUMBER_MESSAGE = "Could not convert {value!r} to a valid number."
|
|
8
|
+
CURRENCY_PREFIX_PATTERN = re.compile(r"^[\$€£¥]")
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
def strip_currency_prefix(value: Any) -> Any:
|
|
12
|
+
"""
|
|
13
|
+
Remove a single common currency symbol from the start of a value.
|
|
14
|
+
"""
|
|
15
|
+
text = str(value).strip()
|
|
16
|
+
if text and CURRENCY_PREFIX_PATTERN.match(text):
|
|
17
|
+
return text[1:].strip()
|
|
18
|
+
return value
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
def _normalize_numeric_text(value: Any, *, sep: str, decimal: str) -> str:
|
|
22
|
+
"""
|
|
23
|
+
Normalize a numeric text string for consistent formatting.
|
|
24
|
+
|
|
25
|
+
Converts a given value to a string representation and ensures normalization of numeric formatting,
|
|
26
|
+
such as removing group separators, converting localized decimal separators, and handling negative
|
|
27
|
+
values enclosed in parentheses.
|
|
28
|
+
"""
|
|
29
|
+
numeric_text = str(value).strip()
|
|
30
|
+
|
|
31
|
+
if numeric_text.startswith("(") and numeric_text.endswith(")"):
|
|
32
|
+
numeric_text = f"-{numeric_text[1:-1]}"
|
|
33
|
+
|
|
34
|
+
if sep:
|
|
35
|
+
numeric_text = numeric_text.replace(sep, "")
|
|
36
|
+
|
|
37
|
+
if decimal != ".":
|
|
38
|
+
numeric_text = numeric_text.replace(decimal, ".")
|
|
39
|
+
|
|
40
|
+
return numeric_text
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
def _has_valid_grouping(numeric_text: str, *, decimal: str) -> bool:
|
|
44
|
+
"""
|
|
45
|
+
Checks whether a numeric text string has valid grouping based on a specified decimal character.
|
|
46
|
+
"""
|
|
47
|
+
if not numeric_text:
|
|
48
|
+
return False
|
|
49
|
+
|
|
50
|
+
unsigned_text = numeric_text[1:] if numeric_text.startswith("-") else numeric_text
|
|
51
|
+
|
|
52
|
+
if unsigned_text.count(decimal) > 1:
|
|
53
|
+
return False
|
|
54
|
+
|
|
55
|
+
integer_text, _, fraction_text = unsigned_text.partition(decimal)
|
|
56
|
+
|
|
57
|
+
if not integer_text and not fraction_text:
|
|
58
|
+
return False
|
|
59
|
+
|
|
60
|
+
for part in (integer_text, fraction_text):
|
|
61
|
+
if not part:
|
|
62
|
+
continue
|
|
63
|
+
if part.startswith("_") or part.endswith("_"):
|
|
64
|
+
return False
|
|
65
|
+
if part.startswith(" ") or part.endswith(" "):
|
|
66
|
+
return False
|
|
67
|
+
if part.startswith(NON_BREAKING_SPACE) or part.endswith(NON_BREAKING_SPACE):
|
|
68
|
+
return False
|
|
69
|
+
if "__" in part or " " in part or NON_BREAKING_SPACE * 2 in part:
|
|
70
|
+
return False
|
|
71
|
+
|
|
72
|
+
stripped_text = SEPARATOR_PATTERN.sub("", numeric_text)
|
|
73
|
+
return bool(NUMBER_PATTERN.fullmatch(stripped_text))
|
|
74
|
+
|
|
75
|
+
|
|
76
|
+
def _strip_group_separators(numeric_text: str) -> str:
|
|
77
|
+
return SEPARATOR_PATTERN.sub("", numeric_text)
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
def _invalid_number_error(value: Any) -> ValueError:
|
|
81
|
+
return ValueError(INVALID_NUMBER_MESSAGE.format(value=value))
|
|
82
|
+
|
|
83
|
+
|
|
84
|
+
def clean_numeric(
|
|
85
|
+
value: Any, *, sep: str = ",", decimal: str = ".", relaxed: bool = False
|
|
86
|
+
) -> float | Any:
|
|
87
|
+
"""
|
|
88
|
+
Cleans and converts a given input to a float by normalizing its numeric representation.
|
|
89
|
+
"""
|
|
90
|
+
if value is None:
|
|
91
|
+
return 0.0
|
|
92
|
+
|
|
93
|
+
normalized_text = _normalize_numeric_text(value, sep=sep, decimal=decimal)
|
|
94
|
+
|
|
95
|
+
if not _has_valid_grouping(normalized_text, decimal=decimal):
|
|
96
|
+
if relaxed:
|
|
97
|
+
return value
|
|
98
|
+
raise _invalid_number_error(value)
|
|
99
|
+
|
|
100
|
+
candidate_text = _strip_group_separators(normalized_text)
|
|
101
|
+
|
|
102
|
+
try:
|
|
103
|
+
return float(candidate_text)
|
|
104
|
+
except ValueError as exc:
|
|
105
|
+
if relaxed:
|
|
106
|
+
return value
|
|
107
|
+
raise _invalid_number_error(value) from exc
|
|
108
|
+
|
|
109
|
+
|
|
110
|
+
def clean_currency_numeric(
|
|
111
|
+
value: Any, *, sep: str = ",", decimal: str = ".", relaxed: bool = False
|
|
112
|
+
) -> float | Any:
|
|
113
|
+
"""
|
|
114
|
+
Cleans and converts a currency-prefixed numeric string to a float.
|
|
115
|
+
"""
|
|
116
|
+
if value is None:
|
|
117
|
+
return 0.0
|
|
118
|
+
|
|
119
|
+
return clean_numeric(
|
|
120
|
+
strip_currency_prefix(value),
|
|
121
|
+
sep=sep,
|
|
122
|
+
decimal=decimal,
|
|
123
|
+
relaxed=relaxed,
|
|
124
|
+
)
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: csvsmith
|
|
3
|
-
Version: 0.7.
|
|
3
|
+
Version: 0.7.3
|
|
4
4
|
Summary: Small CSV utilities: row deduplication, classification, row filtering, and CLI helpers.
|
|
5
5
|
Author-email: Eiichi YAMAMOTO <info@yeiichi.com>
|
|
6
6
|
License: MIT License
|
|
@@ -39,6 +39,9 @@ Requires-Python: >=3.10
|
|
|
39
39
|
Description-Content-Type: text/x-rst
|
|
40
40
|
License-File: LICENSE
|
|
41
41
|
Requires-Dist: openpyxl>=3.1
|
|
42
|
+
Provides-Extra: docs
|
|
43
|
+
Requires-Dist: sphinx<9,>=8; extra == "docs"
|
|
44
|
+
Requires-Dist: furo>=2024.8.6; extra == "docs"
|
|
42
45
|
Dynamic: license-file
|
|
43
46
|
|
|
44
47
|
csvsmith
|
|
@@ -111,17 +114,11 @@ You can use the library from Python:
|
|
|
111
114
|
|
|
112
115
|
.. code-block:: python
|
|
113
116
|
|
|
114
|
-
from csvsmith import
|
|
115
|
-
clean_numeric,
|
|
116
|
-
dedupe_with_report,
|
|
117
|
-
excel_to_csv,
|
|
118
|
-
find_matches_in_csv,
|
|
119
|
-
move_by_suffix,
|
|
120
|
-
)
|
|
117
|
+
from csvsmith.utils.clean_numeric import clean_currency_numeric
|
|
121
118
|
|
|
122
|
-
print(
|
|
119
|
+
print(clean_currency_numeric("$1,234.56"))
|
|
123
120
|
|
|
124
|
-
|
|
121
|
+
For command-line usage, use single quotes around values containing ``$``:
|
|
125
122
|
|
|
126
123
|
.. code-block:: console
|
|
127
124
|
|
|
@@ -138,6 +135,17 @@ Clean numeric values:
|
|
|
138
135
|
|
|
139
136
|
csvsmith clean-numeric "1,234.56" --sep "," --decimal "."
|
|
140
137
|
|
|
138
|
+
Clean currency-prefixed numeric values:
|
|
139
|
+
|
|
140
|
+
.. code-block:: console
|
|
141
|
+
|
|
142
|
+
csvsmith clean-currency-numeric '$1,234.56' --sep "," --decimal "."
|
|
143
|
+
|
|
144
|
+
.. note::
|
|
145
|
+
|
|
146
|
+
Use single quotes for values containing ``$``. Double quotes may trigger
|
|
147
|
+
shell expansion and change the input unexpectedly.
|
|
148
|
+
|
|
141
149
|
Filter rows in a CSV:
|
|
142
150
|
|
|
143
151
|
.. code-block:: console
|
|
@@ -1,5 +1,12 @@
|
|
|
1
1
|
import pytest
|
|
2
|
-
from csvsmith.
|
|
2
|
+
from csvsmith.cli import build_parser, main
|
|
3
|
+
from csvsmith.utils.clean_numeric import clean_currency_numeric, clean_numeric
|
|
4
|
+
|
|
5
|
+
|
|
6
|
+
def test_main_help():
|
|
7
|
+
with pytest.raises(SystemExit) as excinfo:
|
|
8
|
+
main(["--help"])
|
|
9
|
+
assert excinfo.value.code == 0
|
|
3
10
|
|
|
4
11
|
|
|
5
12
|
def test_clean_numeric_with_valid_integer_string() -> None:
|
|
@@ -34,3 +41,28 @@ def test_clean_numeric_with_multiple_decimal_points_raises_valueerror() -> None:
|
|
|
34
41
|
|
|
35
42
|
def test_clean_numeric_with_none_value() -> None:
|
|
36
43
|
assert clean_numeric(None) == 0.0
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
def test_clean_currency_numeric_with_currency_prefix() -> None:
|
|
47
|
+
assert clean_currency_numeric("$1,000") == 1000.0
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
def test_clean_currency_numeric_with_euro_prefix() -> None:
|
|
51
|
+
assert clean_currency_numeric("€1,000.50") == 1000.5
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
def test_clean_numeric_still_rejects_currency_prefix() -> None:
|
|
55
|
+
with pytest.raises(ValueError, match=r"Could not convert '\$1,000'.*"):
|
|
56
|
+
clean_numeric("$1,000")
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
def test_cli_parses_clean_currency_numeric_command():
|
|
60
|
+
parser = build_parser()
|
|
61
|
+
args = parser.parse_args(
|
|
62
|
+
["clean-currency-numeric", "$1,234.56", "--sep", ",", "--decimal", "."]
|
|
63
|
+
)
|
|
64
|
+
|
|
65
|
+
assert args.command == "clean-currency-numeric"
|
|
66
|
+
assert args.value == "$1,234.56"
|
|
67
|
+
assert args.sep == ","
|
|
68
|
+
assert args.decimal == "."
|
|
@@ -138,12 +138,22 @@ def test_cli_parses_clean_numeric_command():
|
|
|
138
138
|
assert args.decimal == "."
|
|
139
139
|
|
|
140
140
|
|
|
141
|
-
def
|
|
141
|
+
def test_cli_parses_clean_currency_numeric_command():
|
|
142
142
|
parser = build_parser()
|
|
143
143
|
args = parser.parse_args(
|
|
144
|
-
["clean-numeric", "
|
|
144
|
+
["clean-currency-numeric", "$1,234.56", "--sep", ",", "--decimal", "."]
|
|
145
145
|
)
|
|
146
146
|
|
|
147
|
+
assert args.command == "clean-currency-numeric"
|
|
148
|
+
assert args.value == "$1,234.56"
|
|
149
|
+
assert args.sep == ","
|
|
150
|
+
assert args.decimal == "."
|
|
151
|
+
|
|
152
|
+
|
|
153
|
+
def test_cli_parses_clean_numeric_command_with_relaxed_mode():
|
|
154
|
+
parser = build_parser()
|
|
155
|
+
args = parser.parse_args(["clean-numeric", "not-a-number", "--relaxed"])
|
|
156
|
+
|
|
147
157
|
assert args.command == "clean-numeric"
|
|
148
158
|
assert args.value == "not-a-number"
|
|
149
159
|
assert args.relaxed is True
|
|
@@ -177,4 +187,4 @@ def test_cli_parses_find_matches_command():
|
|
|
177
187
|
assert args.target == "target"
|
|
178
188
|
assert args.ignore_case is True
|
|
179
189
|
assert args.ignore_whitespace is True
|
|
180
|
-
assert args.no_nfkc is True
|
|
190
|
+
assert args.no_nfkc is True
|
|
@@ -1,124 +0,0 @@
|
|
|
1
|
-
import re
|
|
2
|
-
from typing import Any
|
|
3
|
-
|
|
4
|
-
NON_BREAKING_SPACE = "\xa0"
|
|
5
|
-
SEPARATOR_PATTERN = re.compile(r"[ _\xa0]")
|
|
6
|
-
NUMBER_PATTERN = re.compile(r"^-?(?:\d+|\d*\.\d+)$")
|
|
7
|
-
|
|
8
|
-
|
|
9
|
-
def _normalize_numeric_text(value: Any, *, sep: str, decimal: str) -> str:
|
|
10
|
-
"""
|
|
11
|
-
Normalize a numeric text string for consistent formatting.
|
|
12
|
-
|
|
13
|
-
Converts a given value to a string representation and ensures normalization of numeric formatting,
|
|
14
|
-
such as removing group separators, converting localized decimal separators, and handling negative
|
|
15
|
-
values enclosed in parentheses.
|
|
16
|
-
|
|
17
|
-
:param value: The value to be normalized, which may be of any type.
|
|
18
|
-
:type value: Any
|
|
19
|
-
:param sep: The character used as a group separator in the input value, which will be removed
|
|
20
|
-
during normalization.
|
|
21
|
-
:type sep: str
|
|
22
|
-
:param decimal: The character used as the decimal separator in the input value, which will
|
|
23
|
-
be replaced with a standard period ('.') during normalization.
|
|
24
|
-
:type decimal: str
|
|
25
|
-
:return: A normalized numeric string with consistent formatting.
|
|
26
|
-
:rtype: str
|
|
27
|
-
"""
|
|
28
|
-
numeric_text = str(value).strip()
|
|
29
|
-
|
|
30
|
-
if numeric_text.startswith("(") and numeric_text.endswith(")"):
|
|
31
|
-
numeric_text = f"-{numeric_text[1:-1]}"
|
|
32
|
-
|
|
33
|
-
if sep:
|
|
34
|
-
numeric_text = numeric_text.replace(sep, "")
|
|
35
|
-
|
|
36
|
-
if decimal != ".":
|
|
37
|
-
numeric_text = numeric_text.replace(decimal, ".")
|
|
38
|
-
|
|
39
|
-
return numeric_text
|
|
40
|
-
|
|
41
|
-
|
|
42
|
-
def _has_valid_grouping(numeric_text: str, *, decimal: str) -> bool:
|
|
43
|
-
"""
|
|
44
|
-
Checks whether a numeric text string has valid grouping based on a specified decimal character.
|
|
45
|
-
|
|
46
|
-
This function validates the structure of the given numeric text to determine if it adheres to allowed
|
|
47
|
-
grouping conventions. It ensures the string does not contain invalid or misplaced group separators,
|
|
48
|
-
decimal points, or spacing characters.
|
|
49
|
-
|
|
50
|
-
:param numeric_text: A string representing the numeric text to be validated.
|
|
51
|
-
:type numeric_text: str
|
|
52
|
-
:param decimal: A string representing the character used as the decimal point.
|
|
53
|
-
:type decimal: str
|
|
54
|
-
:return: True if the numeric text satisfies the grouping rules; otherwise, False.
|
|
55
|
-
:rtype: bool
|
|
56
|
-
"""
|
|
57
|
-
if not numeric_text:
|
|
58
|
-
return False
|
|
59
|
-
|
|
60
|
-
unsigned_text = numeric_text[1:] if numeric_text.startswith("-") else numeric_text
|
|
61
|
-
|
|
62
|
-
if unsigned_text.count(decimal) > 1:
|
|
63
|
-
return False
|
|
64
|
-
|
|
65
|
-
integer_text, _, fraction_text = unsigned_text.partition(decimal)
|
|
66
|
-
|
|
67
|
-
if not integer_text and not fraction_text:
|
|
68
|
-
return False
|
|
69
|
-
|
|
70
|
-
for part in (integer_text, fraction_text):
|
|
71
|
-
if not part:
|
|
72
|
-
continue
|
|
73
|
-
if part.startswith("_") or part.endswith("_"):
|
|
74
|
-
return False
|
|
75
|
-
if part.startswith(" ") or part.endswith(" "):
|
|
76
|
-
return False
|
|
77
|
-
if part.startswith(NON_BREAKING_SPACE) or part.endswith(NON_BREAKING_SPACE):
|
|
78
|
-
return False
|
|
79
|
-
if "__" in part or " " in part or NON_BREAKING_SPACE * 2 in part:
|
|
80
|
-
return False
|
|
81
|
-
|
|
82
|
-
stripped_text = SEPARATOR_PATTERN.sub("", numeric_text)
|
|
83
|
-
return bool(NUMBER_PATTERN.fullmatch(stripped_text))
|
|
84
|
-
|
|
85
|
-
|
|
86
|
-
def clean_numeric(
|
|
87
|
-
value: Any, *, sep: str = ",", decimal: str = ".", relaxed: bool = False
|
|
88
|
-
) -> float | Any:
|
|
89
|
-
"""
|
|
90
|
-
Cleans and converts a given input to a float by normalizing its numeric representation.
|
|
91
|
-
Handles separators and decimal points based on the provided arguments. If the input
|
|
92
|
-
value is invalid or cannot be converted, a ValueError is raised unless relaxed mode
|
|
93
|
-
is enabled.
|
|
94
|
-
|
|
95
|
-
:param value: The input value to be cleaned and converted.
|
|
96
|
-
:type value: Any
|
|
97
|
-
:param sep: The character used as a thousands separator in the input value. Default is ",".
|
|
98
|
-
:type sep: str
|
|
99
|
-
:param decimal: The character used as a decimal point in the input value. Default is ".".
|
|
100
|
-
:type decimal: str
|
|
101
|
-
:param relaxed: If True, return the original input when it is not numeric.
|
|
102
|
-
:type relaxed: bool
|
|
103
|
-
:return: The cleaned and converted numeric value as a float, or the original value in relaxed mode.
|
|
104
|
-
:rtype: float | Any
|
|
105
|
-
:raises ValueError: If the input value cannot be converted to a valid number and relaxed is False.
|
|
106
|
-
"""
|
|
107
|
-
if value is None:
|
|
108
|
-
return 0.0
|
|
109
|
-
|
|
110
|
-
normalized_number_text = _normalize_numeric_text(value, sep=sep, decimal=decimal)
|
|
111
|
-
|
|
112
|
-
if not _has_valid_grouping(normalized_number_text, decimal=decimal):
|
|
113
|
-
if relaxed:
|
|
114
|
-
return value
|
|
115
|
-
raise ValueError(f"Could not convert {value!r} to a valid number.")
|
|
116
|
-
|
|
117
|
-
numeric_text = SEPARATOR_PATTERN.sub("", normalized_number_text)
|
|
118
|
-
|
|
119
|
-
try:
|
|
120
|
-
return float(numeric_text)
|
|
121
|
-
except ValueError as exc:
|
|
122
|
-
if relaxed:
|
|
123
|
-
return value
|
|
124
|
-
raise ValueError(f"Could not convert {value!r} to a valid number.") from exc
|
|
@@ -1 +0,0 @@
|
|
|
1
|
-
openpyxl>=3.1
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|