normalize-uk 0.4.2__cp315-cp315-win_amd64.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- normalize_uk/__init__.py +273 -0
- normalize_uk/__init__.pyi +135 -0
- normalize_uk/_normalize_uk.cp315-win_amd64.pyd +0 -0
- normalize_uk/_normalize_uk.pyi +181 -0
- normalize_uk/py.typed +1 -0
- normalize_uk-0.4.2.dist-info/DELVEWHEEL +2 -0
- normalize_uk-0.4.2.dist-info/METADATA +171 -0
- normalize_uk-0.4.2.dist-info/RECORD +11 -0
- normalize_uk-0.4.2.dist-info/WHEEL +5 -0
- normalize_uk-0.4.2.dist-info/licenses/LICENSE +21 -0
- normalize_uk.libs/msvcp140-a4c2229bdc2a2a630acdc095b4d86008.dll +0 -0
normalize_uk/__init__.py
ADDED
|
@@ -0,0 +1,273 @@
|
|
|
1
|
+
"""Ukrainian text normalization and tokenization."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
|
|
6
|
+
# start delvewheel patch
|
|
7
|
+
def _delvewheel_patch_1_13_1():
|
|
8
|
+
import os
|
|
9
|
+
if os.path.isdir(libs_dir := os.path.abspath(os.path.join(os.path.dirname(__file__), os.pardir, 'normalize_uk.libs'))):
|
|
10
|
+
os.add_dll_directory(libs_dir)
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
_delvewheel_patch_1_13_1()
|
|
14
|
+
del _delvewheel_patch_1_13_1
|
|
15
|
+
# end delvewheel patch
|
|
16
|
+
|
|
17
|
+
import warnings as _warnings
|
|
18
|
+
from collections.abc import Iterable
|
|
19
|
+
|
|
20
|
+
from ._normalize_uk import (
|
|
21
|
+
ColonStyle,
|
|
22
|
+
CurrencySymbolPolicy,
|
|
23
|
+
DateStyle,
|
|
24
|
+
NormalizeOptions,
|
|
25
|
+
NormalizePreset,
|
|
26
|
+
NumericDateOrder,
|
|
27
|
+
PhoneStyle,
|
|
28
|
+
QuoteStyle,
|
|
29
|
+
RangeStyle,
|
|
30
|
+
Substring,
|
|
31
|
+
SymbolStyle,
|
|
32
|
+
UncertainSpan,
|
|
33
|
+
UncertaintyCategory,
|
|
34
|
+
UncertaintySeverity,
|
|
35
|
+
expand_abbreviations,
|
|
36
|
+
normalize_abbreviations,
|
|
37
|
+
options_for_preset,
|
|
38
|
+
split_sentences,
|
|
39
|
+
tokenize,
|
|
40
|
+
transliterate_to_cyrillic,
|
|
41
|
+
)
|
|
42
|
+
from ._normalize_uk import (
|
|
43
|
+
cyrilize as _native_cyrilize,
|
|
44
|
+
)
|
|
45
|
+
from ._normalize_uk import (
|
|
46
|
+
cyrrilize as _native_cyrrilize,
|
|
47
|
+
)
|
|
48
|
+
from ._normalize_uk import (
|
|
49
|
+
flag_uncertain as _native_flag_uncertain,
|
|
50
|
+
)
|
|
51
|
+
from ._normalize_uk import (
|
|
52
|
+
normalize_ukrainian as _native_normalize_ukrainian,
|
|
53
|
+
)
|
|
54
|
+
from ._normalize_uk import (
|
|
55
|
+
normalize_ukrainian_many as _native_normalize_ukrainian_many,
|
|
56
|
+
)
|
|
57
|
+
from ._normalize_uk import (
|
|
58
|
+
number_to_ordinal_words as _native_number_to_ordinal_words,
|
|
59
|
+
)
|
|
60
|
+
from ._normalize_uk import (
|
|
61
|
+
number_to_words as _native_number_to_words,
|
|
62
|
+
)
|
|
63
|
+
from ._normalize_uk import (
|
|
64
|
+
number_to_words_case as _native_number_to_words_case,
|
|
65
|
+
)
|
|
66
|
+
from ._normalize_uk import (
|
|
67
|
+
number_to_words_digit_by_digit as _native_number_to_words_digit_by_digit,
|
|
68
|
+
)
|
|
69
|
+
from ._normalize_uk import (
|
|
70
|
+
sentenize as _native_sentenize,
|
|
71
|
+
)
|
|
72
|
+
|
|
73
|
+
__all__ = (
|
|
74
|
+
"ColonStyle",
|
|
75
|
+
"CurrencySymbolPolicy",
|
|
76
|
+
"DateStyle",
|
|
77
|
+
"NormalizeOptions",
|
|
78
|
+
"NormalizePreset",
|
|
79
|
+
"NumericDateOrder",
|
|
80
|
+
"PhoneStyle",
|
|
81
|
+
"QuoteStyle",
|
|
82
|
+
"RangeStyle",
|
|
83
|
+
"Substring",
|
|
84
|
+
"SymbolStyle",
|
|
85
|
+
"UncertainSpan",
|
|
86
|
+
"UncertaintyCategory",
|
|
87
|
+
"UncertaintySeverity",
|
|
88
|
+
"cyrilize",
|
|
89
|
+
"cyrrilize",
|
|
90
|
+
"expand_abbreviations",
|
|
91
|
+
"flag_uncertain",
|
|
92
|
+
"normalize_abbreviations",
|
|
93
|
+
"normalize_ukrainian",
|
|
94
|
+
"normalize_ukrainian_many",
|
|
95
|
+
"normalize_ukrainian_with_preset",
|
|
96
|
+
"number_to_ordinal_words",
|
|
97
|
+
"number_to_words",
|
|
98
|
+
"number_to_words_case",
|
|
99
|
+
"number_to_words_digit_by_digit",
|
|
100
|
+
"options_for_preset",
|
|
101
|
+
"sentenize",
|
|
102
|
+
"split_sentences",
|
|
103
|
+
"tokenize",
|
|
104
|
+
"transliterate_to_cyrillic",
|
|
105
|
+
)
|
|
106
|
+
|
|
107
|
+
_MAX_SPOKEN_NUMBER = 10**18 - 1
|
|
108
|
+
_ORDINAL_FORMS = frozenset(
|
|
109
|
+
(
|
|
110
|
+
"nom_m",
|
|
111
|
+
"nom_n",
|
|
112
|
+
"nom_f",
|
|
113
|
+
"nom_pl",
|
|
114
|
+
"gen",
|
|
115
|
+
"dat",
|
|
116
|
+
"prep",
|
|
117
|
+
"loc",
|
|
118
|
+
"pl",
|
|
119
|
+
"loc_pl",
|
|
120
|
+
"acc_f",
|
|
121
|
+
"gen_f",
|
|
122
|
+
"ins",
|
|
123
|
+
"ins_f",
|
|
124
|
+
"ins_pl",
|
|
125
|
+
"loc_f",
|
|
126
|
+
)
|
|
127
|
+
)
|
|
128
|
+
_CARDINAL_CASES = frozenset(("gen", "dat", "instr", "prep"))
|
|
129
|
+
|
|
130
|
+
|
|
131
|
+
def _spoken_number(n: int) -> int:
|
|
132
|
+
if isinstance(n, bool) or not isinstance(n, int):
|
|
133
|
+
raise TypeError("n must be an integer")
|
|
134
|
+
if not 0 <= n <= _MAX_SPOKEN_NUMBER:
|
|
135
|
+
raise ValueError(f"n must be between 0 and {_MAX_SPOKEN_NUMBER}")
|
|
136
|
+
return n
|
|
137
|
+
|
|
138
|
+
|
|
139
|
+
def number_to_words(n: int) -> str:
|
|
140
|
+
"""Spell a nonnegative integer smaller than 10**18 in Ukrainian."""
|
|
141
|
+
return _native_number_to_words(_spoken_number(n))
|
|
142
|
+
|
|
143
|
+
|
|
144
|
+
def number_to_words_digit_by_digit(digits: str) -> str:
|
|
145
|
+
"""Spell each ASCII digit, preserving leading zeroes."""
|
|
146
|
+
if not isinstance(digits, str):
|
|
147
|
+
raise TypeError("digits must be a string")
|
|
148
|
+
if not digits or any(ch < "0" or ch > "9" for ch in digits):
|
|
149
|
+
raise ValueError("digits must contain one or more ASCII digits only")
|
|
150
|
+
return _native_number_to_words_digit_by_digit(digits)
|
|
151
|
+
|
|
152
|
+
|
|
153
|
+
def number_to_ordinal_words(n: int, form: str = "nom_m") -> str:
|
|
154
|
+
"""Spell an ordinal using one of the supported grammatical forms."""
|
|
155
|
+
if not isinstance(form, str):
|
|
156
|
+
raise TypeError("form must be a string")
|
|
157
|
+
if form not in _ORDINAL_FORMS:
|
|
158
|
+
raise ValueError(
|
|
159
|
+
f"unknown ordinal form {form!r}; expected one of {', '.join(sorted(_ORDINAL_FORMS))}"
|
|
160
|
+
)
|
|
161
|
+
return _native_number_to_ordinal_words(_spoken_number(n), form)
|
|
162
|
+
|
|
163
|
+
|
|
164
|
+
def number_to_words_case(n: int, grammatical_case: str) -> str:
|
|
165
|
+
"""Spell a cardinal in the genitive, dative, instrumental, or prepositional case."""
|
|
166
|
+
if not isinstance(grammatical_case, str):
|
|
167
|
+
raise TypeError("grammatical_case must be a string")
|
|
168
|
+
if grammatical_case not in _CARDINAL_CASES:
|
|
169
|
+
raise ValueError(
|
|
170
|
+
f"unknown grammatical case {grammatical_case!r}; expected one of "
|
|
171
|
+
f"{', '.join(sorted(_CARDINAL_CASES))}"
|
|
172
|
+
)
|
|
173
|
+
return _native_number_to_words_case(_spoken_number(n), grammatical_case)
|
|
174
|
+
|
|
175
|
+
|
|
176
|
+
def _selection(
|
|
177
|
+
options: NormalizeOptions | NormalizePreset | None,
|
|
178
|
+
preset: NormalizePreset | None,
|
|
179
|
+
) -> NormalizeOptions | NormalizePreset | None:
|
|
180
|
+
if isinstance(options, NormalizePreset):
|
|
181
|
+
if preset is not None:
|
|
182
|
+
raise ValueError("provide either options or preset, not both")
|
|
183
|
+
preset, options = options, None
|
|
184
|
+
if options is not None and preset is not None:
|
|
185
|
+
raise ValueError("provide either options or preset, not both")
|
|
186
|
+
if options is not None:
|
|
187
|
+
if not isinstance(options, NormalizeOptions):
|
|
188
|
+
raise TypeError("options must be a NormalizeOptions instance")
|
|
189
|
+
return options
|
|
190
|
+
if preset is not None:
|
|
191
|
+
if not isinstance(preset, NormalizePreset):
|
|
192
|
+
raise TypeError("preset must be a NormalizePreset value")
|
|
193
|
+
return preset
|
|
194
|
+
return None
|
|
195
|
+
|
|
196
|
+
|
|
197
|
+
def normalize_ukrainian(
|
|
198
|
+
text: str,
|
|
199
|
+
options: NormalizeOptions | NormalizePreset | None = None,
|
|
200
|
+
*,
|
|
201
|
+
preset: NormalizePreset | None = None,
|
|
202
|
+
) -> str:
|
|
203
|
+
"""Normalize text with either an options object or a preset."""
|
|
204
|
+
selected = _selection(options, preset)
|
|
205
|
+
if selected is None:
|
|
206
|
+
return _native_normalize_ukrainian(text)
|
|
207
|
+
if isinstance(selected, NormalizePreset):
|
|
208
|
+
return _native_normalize_ukrainian(text, selected)
|
|
209
|
+
return _native_normalize_ukrainian(text, selected)
|
|
210
|
+
|
|
211
|
+
|
|
212
|
+
def normalize_ukrainian_many(
|
|
213
|
+
texts: Iterable[str],
|
|
214
|
+
options: NormalizeOptions | NormalizePreset | None = None,
|
|
215
|
+
*,
|
|
216
|
+
preset: NormalizePreset | None = None,
|
|
217
|
+
) -> list[str]:
|
|
218
|
+
"""Normalize an iterable of strings using one options snapshot for the batch."""
|
|
219
|
+
selected = _selection(options, preset)
|
|
220
|
+
if selected is None:
|
|
221
|
+
return _native_normalize_ukrainian_many(texts)
|
|
222
|
+
if isinstance(selected, NormalizePreset):
|
|
223
|
+
return _native_normalize_ukrainian_many(texts, selected)
|
|
224
|
+
return _native_normalize_ukrainian_many(texts, selected)
|
|
225
|
+
|
|
226
|
+
|
|
227
|
+
def flag_uncertain(
|
|
228
|
+
text: str,
|
|
229
|
+
options: NormalizeOptions | NormalizePreset | None = None,
|
|
230
|
+
*,
|
|
231
|
+
preset: NormalizePreset | None = None,
|
|
232
|
+
) -> list[UncertainSpan]:
|
|
233
|
+
"""Find uncertain spans; explicit options suppress resolved ambiguity warnings."""
|
|
234
|
+
selected = _selection(options, preset)
|
|
235
|
+
if selected is None:
|
|
236
|
+
return _native_flag_uncertain(text)
|
|
237
|
+
if isinstance(selected, NormalizePreset):
|
|
238
|
+
return _native_flag_uncertain(text, selected)
|
|
239
|
+
return _native_flag_uncertain(text, selected)
|
|
240
|
+
|
|
241
|
+
|
|
242
|
+
def normalize_ukrainian_with_preset(
|
|
243
|
+
text: str, preset: NormalizePreset = NormalizePreset.Default
|
|
244
|
+
) -> str:
|
|
245
|
+
"""Compatibility alias for normalize_ukrainian(text, preset=preset)."""
|
|
246
|
+
return normalize_ukrainian(text, preset=preset)
|
|
247
|
+
|
|
248
|
+
|
|
249
|
+
def sentenize(text: str) -> list[Substring]:
|
|
250
|
+
_warnings.warn(
|
|
251
|
+
"sentenize() is deprecated; use split_sentences()",
|
|
252
|
+
DeprecationWarning,
|
|
253
|
+
stacklevel=2,
|
|
254
|
+
)
|
|
255
|
+
return _native_sentenize(text)
|
|
256
|
+
|
|
257
|
+
|
|
258
|
+
def cyrilize(text: str) -> str:
|
|
259
|
+
_warnings.warn(
|
|
260
|
+
"cyrilize() is deprecated; use transliterate_to_cyrillic()",
|
|
261
|
+
DeprecationWarning,
|
|
262
|
+
stacklevel=2,
|
|
263
|
+
)
|
|
264
|
+
return _native_cyrilize(text)
|
|
265
|
+
|
|
266
|
+
|
|
267
|
+
def cyrrilize(text: str) -> str:
|
|
268
|
+
_warnings.warn(
|
|
269
|
+
"cyrrilize() is deprecated; use transliterate_to_cyrillic()",
|
|
270
|
+
DeprecationWarning,
|
|
271
|
+
stacklevel=2,
|
|
272
|
+
)
|
|
273
|
+
return _native_cyrrilize(text)
|
|
@@ -0,0 +1,135 @@
|
|
|
1
|
+
from collections.abc import Iterable
|
|
2
|
+
from typing import Literal, overload
|
|
3
|
+
|
|
4
|
+
from ._normalize_uk import (
|
|
5
|
+
ColonStyle as ColonStyle,
|
|
6
|
+
)
|
|
7
|
+
from ._normalize_uk import (
|
|
8
|
+
CurrencySymbolPolicy as CurrencySymbolPolicy,
|
|
9
|
+
)
|
|
10
|
+
from ._normalize_uk import (
|
|
11
|
+
DateStyle as DateStyle,
|
|
12
|
+
)
|
|
13
|
+
from ._normalize_uk import (
|
|
14
|
+
NormalizeOptions as NormalizeOptions,
|
|
15
|
+
)
|
|
16
|
+
from ._normalize_uk import (
|
|
17
|
+
NormalizePreset as NormalizePreset,
|
|
18
|
+
)
|
|
19
|
+
from ._normalize_uk import (
|
|
20
|
+
NumericDateOrder as NumericDateOrder,
|
|
21
|
+
)
|
|
22
|
+
from ._normalize_uk import (
|
|
23
|
+
PhoneStyle as PhoneStyle,
|
|
24
|
+
)
|
|
25
|
+
from ._normalize_uk import (
|
|
26
|
+
QuoteStyle as QuoteStyle,
|
|
27
|
+
)
|
|
28
|
+
from ._normalize_uk import (
|
|
29
|
+
RangeStyle as RangeStyle,
|
|
30
|
+
)
|
|
31
|
+
from ._normalize_uk import (
|
|
32
|
+
Substring as Substring,
|
|
33
|
+
)
|
|
34
|
+
from ._normalize_uk import (
|
|
35
|
+
SymbolStyle as SymbolStyle,
|
|
36
|
+
)
|
|
37
|
+
from ._normalize_uk import (
|
|
38
|
+
UncertainSpan as UncertainSpan,
|
|
39
|
+
)
|
|
40
|
+
from ._normalize_uk import (
|
|
41
|
+
UncertaintyCategory as UncertaintyCategory,
|
|
42
|
+
)
|
|
43
|
+
from ._normalize_uk import (
|
|
44
|
+
UncertaintySeverity as UncertaintySeverity,
|
|
45
|
+
)
|
|
46
|
+
from ._normalize_uk import (
|
|
47
|
+
expand_abbreviations as expand_abbreviations,
|
|
48
|
+
)
|
|
49
|
+
from ._normalize_uk import (
|
|
50
|
+
normalize_abbreviations as normalize_abbreviations,
|
|
51
|
+
)
|
|
52
|
+
from ._normalize_uk import (
|
|
53
|
+
options_for_preset as options_for_preset,
|
|
54
|
+
)
|
|
55
|
+
from ._normalize_uk import (
|
|
56
|
+
split_sentences as split_sentences,
|
|
57
|
+
)
|
|
58
|
+
from ._normalize_uk import (
|
|
59
|
+
tokenize as tokenize,
|
|
60
|
+
)
|
|
61
|
+
from ._normalize_uk import (
|
|
62
|
+
transliterate_to_cyrillic as transliterate_to_cyrillic,
|
|
63
|
+
)
|
|
64
|
+
|
|
65
|
+
def number_to_words(n: int) -> str: ...
|
|
66
|
+
def number_to_words_digit_by_digit(digits: str) -> str: ...
|
|
67
|
+
def number_to_ordinal_words(
|
|
68
|
+
n: int,
|
|
69
|
+
form: Literal[
|
|
70
|
+
"nom_m",
|
|
71
|
+
"nom_n",
|
|
72
|
+
"nom_f",
|
|
73
|
+
"nom_pl",
|
|
74
|
+
"gen",
|
|
75
|
+
"dat",
|
|
76
|
+
"prep",
|
|
77
|
+
"loc",
|
|
78
|
+
"pl",
|
|
79
|
+
"loc_pl",
|
|
80
|
+
"acc_f",
|
|
81
|
+
"gen_f",
|
|
82
|
+
"ins",
|
|
83
|
+
"ins_f",
|
|
84
|
+
"ins_pl",
|
|
85
|
+
"loc_f",
|
|
86
|
+
] = "nom_m",
|
|
87
|
+
) -> str: ...
|
|
88
|
+
def number_to_words_case(
|
|
89
|
+
n: int, grammatical_case: Literal["gen", "dat", "instr", "prep"]
|
|
90
|
+
) -> str: ...
|
|
91
|
+
def cyrilize(text: str) -> str: ...
|
|
92
|
+
def cyrrilize(text: str) -> str: ...
|
|
93
|
+
@overload
|
|
94
|
+
def normalize_ukrainian(
|
|
95
|
+
text: str, options: NormalizeOptions | None = None, *, preset: None = None
|
|
96
|
+
) -> str: ...
|
|
97
|
+
@overload
|
|
98
|
+
def normalize_ukrainian(
|
|
99
|
+
text: str, options: NormalizePreset, *, preset: None = None
|
|
100
|
+
) -> str: ...
|
|
101
|
+
@overload
|
|
102
|
+
def normalize_ukrainian(
|
|
103
|
+
text: str, options: None = None, *, preset: NormalizePreset
|
|
104
|
+
) -> str: ...
|
|
105
|
+
def normalize_ukrainian_with_preset(
|
|
106
|
+
text: str, preset: NormalizePreset = NormalizePreset.Default
|
|
107
|
+
) -> str: ...
|
|
108
|
+
@overload
|
|
109
|
+
def normalize_ukrainian_many(
|
|
110
|
+
texts: Iterable[str],
|
|
111
|
+
options: NormalizeOptions | None = None,
|
|
112
|
+
*,
|
|
113
|
+
preset: None = None,
|
|
114
|
+
) -> list[str]: ...
|
|
115
|
+
@overload
|
|
116
|
+
def normalize_ukrainian_many(
|
|
117
|
+
texts: Iterable[str], options: NormalizePreset, *, preset: None = None
|
|
118
|
+
) -> list[str]: ...
|
|
119
|
+
@overload
|
|
120
|
+
def normalize_ukrainian_many(
|
|
121
|
+
texts: Iterable[str], options: None = None, *, preset: NormalizePreset
|
|
122
|
+
) -> list[str]: ...
|
|
123
|
+
@overload
|
|
124
|
+
def flag_uncertain(
|
|
125
|
+
text: str, options: NormalizeOptions | None = None, *, preset: None = None
|
|
126
|
+
) -> list[UncertainSpan]: ...
|
|
127
|
+
@overload
|
|
128
|
+
def flag_uncertain(
|
|
129
|
+
text: str, options: NormalizePreset, *, preset: None = None
|
|
130
|
+
) -> list[UncertainSpan]: ...
|
|
131
|
+
@overload
|
|
132
|
+
def flag_uncertain(
|
|
133
|
+
text: str, options: None = None, *, preset: NormalizePreset
|
|
134
|
+
) -> list[UncertainSpan]: ...
|
|
135
|
+
def sentenize(text: str) -> list[Substring]: ...
|
|
Binary file
|
|
@@ -0,0 +1,181 @@
|
|
|
1
|
+
from collections.abc import Iterable
|
|
2
|
+
from enum import Enum
|
|
3
|
+
from typing import overload
|
|
4
|
+
|
|
5
|
+
class UncertaintyCategory(Enum):
|
|
6
|
+
AmbiguousAbbreviation: UncertaintyCategory
|
|
7
|
+
BareNumber: UncertaintyCategory
|
|
8
|
+
Currency: UncertaintyCategory
|
|
9
|
+
Date: UncertaintyCategory
|
|
10
|
+
Identifier: UncertaintyCategory
|
|
11
|
+
ForeignWord: UncertaintyCategory
|
|
12
|
+
MixedScript: UncertaintyCategory
|
|
13
|
+
RomanNumeral: UncertaintyCategory
|
|
14
|
+
Unit: UncertaintyCategory
|
|
15
|
+
Web: UncertaintyCategory
|
|
16
|
+
InvalidDate: UncertaintyCategory
|
|
17
|
+
AmbiguousNumberGrouping: UncertaintyCategory
|
|
18
|
+
Agreement: UncertaintyCategory
|
|
19
|
+
Time: UncertaintyCategory
|
|
20
|
+
Fraction: UncertaintyCategory
|
|
21
|
+
Network: UncertaintyCategory
|
|
22
|
+
Scientific: UncertaintyCategory
|
|
23
|
+
Coordinate: UncertaintyCategory
|
|
24
|
+
|
|
25
|
+
class UncertaintySeverity(Enum):
|
|
26
|
+
Info: UncertaintySeverity
|
|
27
|
+
Warning: UncertaintySeverity
|
|
28
|
+
Error: UncertaintySeverity
|
|
29
|
+
|
|
30
|
+
class RangeStyle(Enum):
|
|
31
|
+
Compact: RangeStyle
|
|
32
|
+
FromTo: RangeStyle
|
|
33
|
+
|
|
34
|
+
class PhoneStyle(Enum):
|
|
35
|
+
Grouped: PhoneStyle
|
|
36
|
+
DigitByDigit: PhoneStyle
|
|
37
|
+
|
|
38
|
+
class SymbolStyle(Enum):
|
|
39
|
+
Expand: SymbolStyle
|
|
40
|
+
Preserve: SymbolStyle
|
|
41
|
+
|
|
42
|
+
class DateStyle(Enum):
|
|
43
|
+
Formal: DateStyle
|
|
44
|
+
Spoken: DateStyle
|
|
45
|
+
|
|
46
|
+
class ColonStyle(Enum):
|
|
47
|
+
Contextual: ColonStyle
|
|
48
|
+
Clock: ColonStyle
|
|
49
|
+
Ratio: ColonStyle
|
|
50
|
+
|
|
51
|
+
class NumericDateOrder(Enum):
|
|
52
|
+
DayMonthYear: NumericDateOrder
|
|
53
|
+
MonthDayYear: NumericDateOrder
|
|
54
|
+
PreserveAmbiguous: NumericDateOrder
|
|
55
|
+
|
|
56
|
+
class CurrencySymbolPolicy(Enum):
|
|
57
|
+
AssumeCommon: CurrencySymbolPolicy
|
|
58
|
+
PreserveAmbiguous: CurrencySymbolPolicy
|
|
59
|
+
|
|
60
|
+
class QuoteStyle(Enum):
|
|
61
|
+
Keep: QuoteStyle
|
|
62
|
+
Guillemets: QuoteStyle
|
|
63
|
+
Straight: QuoteStyle
|
|
64
|
+
Strip: QuoteStyle
|
|
65
|
+
|
|
66
|
+
class NormalizePreset(Enum):
|
|
67
|
+
Default: NormalizePreset
|
|
68
|
+
TtsFriendly: NormalizePreset
|
|
69
|
+
Conservative: NormalizePreset
|
|
70
|
+
SearchIndexing: NormalizePreset
|
|
71
|
+
|
|
72
|
+
class UncertainSpan:
|
|
73
|
+
@property
|
|
74
|
+
def start(self) -> int: ...
|
|
75
|
+
@property
|
|
76
|
+
def stop(self) -> int: ...
|
|
77
|
+
@property
|
|
78
|
+
def text(self) -> str: ...
|
|
79
|
+
@property
|
|
80
|
+
def reason(self) -> str: ...
|
|
81
|
+
@property
|
|
82
|
+
def category(self) -> UncertaintyCategory: ...
|
|
83
|
+
@property
|
|
84
|
+
def severity(self) -> UncertaintySeverity: ...
|
|
85
|
+
def __eq__(self, other: object) -> bool: ...
|
|
86
|
+
def __repr__(self) -> str: ...
|
|
87
|
+
def __copy__(self) -> UncertainSpan: ...
|
|
88
|
+
def __deepcopy__(self, memo: dict[object, object]) -> UncertainSpan: ...
|
|
89
|
+
|
|
90
|
+
class NormalizeOptions:
|
|
91
|
+
expand_known_acronyms: bool
|
|
92
|
+
spell_unknown_acronyms: bool
|
|
93
|
+
normalize_english_words: bool
|
|
94
|
+
transliterate_latin: bool
|
|
95
|
+
range_style: RangeStyle
|
|
96
|
+
phone_style: PhoneStyle
|
|
97
|
+
symbol_style: SymbolStyle
|
|
98
|
+
date_style: DateStyle
|
|
99
|
+
colon_style: ColonStyle
|
|
100
|
+
numeric_date_order: NumericDateOrder
|
|
101
|
+
currency_symbol_policy: CurrencySymbolPolicy
|
|
102
|
+
repair_homoglyphs: bool
|
|
103
|
+
validate_dates: bool
|
|
104
|
+
parse_thousand_separators: bool
|
|
105
|
+
normalize_network_addresses: bool
|
|
106
|
+
quote_style: QuoteStyle
|
|
107
|
+
|
|
108
|
+
def __init__(
|
|
109
|
+
self,
|
|
110
|
+
preset: NormalizePreset = NormalizePreset.Default,
|
|
111
|
+
*,
|
|
112
|
+
expand_known_acronyms: bool = ...,
|
|
113
|
+
spell_unknown_acronyms: bool = ...,
|
|
114
|
+
normalize_english_words: bool = ...,
|
|
115
|
+
transliterate_latin: bool = ...,
|
|
116
|
+
repair_homoglyphs: bool = ...,
|
|
117
|
+
validate_dates: bool = ...,
|
|
118
|
+
parse_thousand_separators: bool = ...,
|
|
119
|
+
normalize_network_addresses: bool = ...,
|
|
120
|
+
range_style: RangeStyle = ...,
|
|
121
|
+
phone_style: PhoneStyle = ...,
|
|
122
|
+
symbol_style: SymbolStyle = ...,
|
|
123
|
+
date_style: DateStyle = ...,
|
|
124
|
+
colon_style: ColonStyle = ...,
|
|
125
|
+
numeric_date_order: NumericDateOrder = ...,
|
|
126
|
+
currency_symbol_policy: CurrencySymbolPolicy = ...,
|
|
127
|
+
quote_style: QuoteStyle = ...,
|
|
128
|
+
) -> None: ...
|
|
129
|
+
def __copy__(self) -> NormalizeOptions: ...
|
|
130
|
+
def __deepcopy__(self, memo: dict[object, object]) -> NormalizeOptions: ...
|
|
131
|
+
|
|
132
|
+
class Substring:
|
|
133
|
+
@property
|
|
134
|
+
def start(self) -> int: ...
|
|
135
|
+
@property
|
|
136
|
+
def stop(self) -> int: ...
|
|
137
|
+
@property
|
|
138
|
+
def text(self) -> str: ...
|
|
139
|
+
def __eq__(self, other: object) -> bool: ...
|
|
140
|
+
def __repr__(self) -> str: ...
|
|
141
|
+
def __copy__(self) -> Substring: ...
|
|
142
|
+
def __deepcopy__(self, memo: dict[object, object]) -> Substring: ...
|
|
143
|
+
|
|
144
|
+
def options_for_preset(preset: NormalizePreset) -> NormalizeOptions: ...
|
|
145
|
+
def number_to_words(n: int) -> str: ...
|
|
146
|
+
def number_to_words_digit_by_digit(digits: str) -> str: ...
|
|
147
|
+
def number_to_ordinal_words(n: int, form: str = "nom_m") -> str: ...
|
|
148
|
+
def number_to_words_case(n: int, grammatical_case: str) -> str: ...
|
|
149
|
+
def normalize_abbreviations(text: str) -> str: ...
|
|
150
|
+
def expand_abbreviations(text: str) -> str: ...
|
|
151
|
+
def transliterate_to_cyrillic(text: str) -> str: ...
|
|
152
|
+
def cyrilize(text: str) -> str: ...
|
|
153
|
+
def cyrrilize(text: str) -> str: ...
|
|
154
|
+
@overload
|
|
155
|
+
def normalize_ukrainian(text: str) -> str: ...
|
|
156
|
+
@overload
|
|
157
|
+
def normalize_ukrainian(text: str, options: NormalizeOptions) -> str: ...
|
|
158
|
+
@overload
|
|
159
|
+
def normalize_ukrainian(text: str, preset: NormalizePreset) -> str: ...
|
|
160
|
+
@overload
|
|
161
|
+
def normalize_ukrainian_many(texts: Iterable[str]) -> list[str]: ...
|
|
162
|
+
@overload
|
|
163
|
+
def normalize_ukrainian_many(
|
|
164
|
+
texts: Iterable[str], options: NormalizeOptions
|
|
165
|
+
) -> list[str]: ...
|
|
166
|
+
@overload
|
|
167
|
+
def normalize_ukrainian_many(
|
|
168
|
+
texts: Iterable[str], preset: NormalizePreset
|
|
169
|
+
) -> list[str]: ...
|
|
170
|
+
def normalize_ukrainian_with_preset(
|
|
171
|
+
text: str, preset: NormalizePreset = NormalizePreset.Default
|
|
172
|
+
) -> str: ...
|
|
173
|
+
@overload
|
|
174
|
+
def flag_uncertain(text: str) -> list[UncertainSpan]: ...
|
|
175
|
+
@overload
|
|
176
|
+
def flag_uncertain(text: str, options: NormalizeOptions) -> list[UncertainSpan]: ...
|
|
177
|
+
@overload
|
|
178
|
+
def flag_uncertain(text: str, preset: NormalizePreset) -> list[UncertainSpan]: ...
|
|
179
|
+
def split_sentences(text: str) -> list[Substring]: ...
|
|
180
|
+
def sentenize(text: str) -> list[Substring]: ...
|
|
181
|
+
def tokenize(text: str) -> list[Substring]: ...
|
normalize_uk/py.typed
ADDED
|
@@ -0,0 +1 @@
|
|
|
1
|
+
|
|
@@ -0,0 +1,2 @@
|
|
|
1
|
+
Version: 1.13.1
|
|
2
|
+
Arguments: ['C:\\Users\\runneradmin\\AppData\\Local\\Temp\\cibw-run-3cq3h79w\\cp315-win_amd64\\build\\venv\\Scripts\\delvewheel', 'repair', '-w', 'C:\\Users\\runneradmin\\AppData\\Local\\Temp\\cibw-run-3cq3h79w\\cp315-win_amd64\\repaired_wheel', '-v', 'C:\\Users\\runneradmin\\AppData\\Local\\Temp\\cibw-run-3cq3h79w\\cp315-win_amd64\\built_wheel\\normalize_uk-0.4.2-cp315-cp315-win_amd64.whl']
|
|
@@ -0,0 +1,171 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: normalize-uk
|
|
3
|
+
Version: 0.4.2
|
|
4
|
+
Summary: Python bindings for Ukrainian text normalization and tokenization utilities
|
|
5
|
+
License-Expression: MIT
|
|
6
|
+
Classifier: Programming Language :: Python :: 3
|
|
7
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
8
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
9
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
10
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
11
|
+
Classifier: Programming Language :: Python :: 3.14
|
|
12
|
+
Classifier: Programming Language :: Python :: 3.15
|
|
13
|
+
Classifier: Programming Language :: C++
|
|
14
|
+
Requires-Python: >=3.10
|
|
15
|
+
Description-Content-Type: text/markdown
|
|
16
|
+
|
|
17
|
+
# normalize-uk-cpp
|
|
18
|
+
|
|
19
|
+
[](https://github.com/ThirdLetterC/normalize_uk-cpp/actions/workflows/ci.yml)
|
|
20
|
+
[](https://github.com/ThirdLetterC/normalize_uk-cpp/actions/workflows/release.yml)
|
|
21
|
+
|
|
22
|
+
C++23 Ukrainian text normalization and tokenization utilities with optional Python 3.10+ bindings.
|
|
23
|
+
|
|
24
|
+
## CMake
|
|
25
|
+
|
|
26
|
+
```sh
|
|
27
|
+
cmake -S . -B build
|
|
28
|
+
cmake --build build
|
|
29
|
+
ctest --test-dir build
|
|
30
|
+
```
|
|
31
|
+
|
|
32
|
+
Enable Python bindings explicitly when building with CMake:
|
|
33
|
+
|
|
34
|
+
```sh
|
|
35
|
+
cmake -S . -B build-python -DNORMALIZE_UK_CPP_BUILD_PYTHON=ON
|
|
36
|
+
cmake --build build-python
|
|
37
|
+
```
|
|
38
|
+
|
|
39
|
+
A regular CMake install includes the C++ library, headers, and CMake package.
|
|
40
|
+
Python wheels contain only the Python package and compiled extension.
|
|
41
|
+
|
|
42
|
+
## Python
|
|
43
|
+
|
|
44
|
+
```sh
|
|
45
|
+
python -m pip install .
|
|
46
|
+
```
|
|
47
|
+
|
|
48
|
+
```python
|
|
49
|
+
import normalize_uk as nuk
|
|
50
|
+
|
|
51
|
+
print(nuk.number_to_words(123))
|
|
52
|
+
print(nuk.normalize_ukrainian("01.05.2024"))
|
|
53
|
+
print(nuk.normalize_ukrainian("01.05.2024", preset=nuk.NormalizePreset.TtsFriendly))
|
|
54
|
+
print(nuk.normalize_ukrainian_many(["01.05.2024", "5 кг"]))
|
|
55
|
+
print([sentence.text for sentence in nuk.split_sentences("П'ять зв'язків. Два.")])
|
|
56
|
+
print([token.text for token in nuk.tokenize("П'ять зв'язків.")])
|
|
57
|
+
```
|
|
58
|
+
|
|
59
|
+
More examples live in `examples/python/`.
|
|
60
|
+
|
|
61
|
+
Tags matching the version in `pyproject.toml` (for example, `v0.4.2`) trigger
|
|
62
|
+
wheel builds for supported Python versions. The workflow uploads the wheels to
|
|
63
|
+
GitHub Release Assets, then downloads those Assets and publishes them to PyPI.
|
|
64
|
+
To enable PyPI Trusted Publishing, register `ThirdLetterC/normalize_uk-cpp` as
|
|
65
|
+
a publisher for `normalize-uk` with workflow `release.yml` and environment
|
|
66
|
+
`pypi`. For a new PyPI project, register a
|
|
67
|
+
[pending publisher](https://docs.pypi.org/trusted-publishers/creating-a-project-through-oidc/)
|
|
68
|
+
first. No PyPI API token is needed.
|
|
69
|
+
|
|
70
|
+
`NormalizeOptions` accepts a preset and named overrides at construction time:
|
|
71
|
+
|
|
72
|
+
```python
|
|
73
|
+
options = nuk.NormalizeOptions(
|
|
74
|
+
preset=nuk.NormalizePreset.TtsFriendly,
|
|
75
|
+
range_style=nuk.RangeStyle.Compact,
|
|
76
|
+
numeric_date_order=nuk.NumericDateOrder.DayMonthYear,
|
|
77
|
+
)
|
|
78
|
+
result = nuk.normalize_ukrainian("5–7 кг", options=options)
|
|
79
|
+
spans = nuk.flag_uncertain("10:30, $12", options=options)
|
|
80
|
+
```
|
|
81
|
+
|
|
82
|
+
Pass either `options=` or `preset=` to `normalize_ukrainian` and `flag_uncertain`.
|
|
83
|
+
`normalize_ukrainian_many()` accepts any iterable of Python strings and applies one
|
|
84
|
+
options snapshot to the entire batch. It returns a list in input order and raises
|
|
85
|
+
`TypeError` if an item is not a string. It accepts the same `options=` and `preset=`
|
|
86
|
+
selection as `normalize_ukrainian`. Identical strings within one batch are
|
|
87
|
+
normalized once and their result is reused.
|
|
88
|
+
The older positional options/preset calls and `normalize_ukrainian_with_preset()` remain available.
|
|
89
|
+
Without options, `flag_uncertain()` reports all ambiguity candidates. With explicit options
|
|
90
|
+
or a preset, it omits warnings for ambiguous dates, colon pairs, and currency symbols
|
|
91
|
+
when the selected policy resolves them; invalid-value diagnostics remain.
|
|
92
|
+
|
|
93
|
+
`Substring.start`/`stop` and `UncertainSpan.start`/`stop` are Python `str` indexes,
|
|
94
|
+
with `stop` exclusive: `span.text == source[span.start:span.stop]`. They count Unicode
|
|
95
|
+
code points, matching Python slicing, rather than UTF-8 bytes.
|
|
96
|
+
|
|
97
|
+
`number_to_words()`, `number_to_ordinal_words()`, and `number_to_words_case()` accept
|
|
98
|
+
integers from 0 through `999999999999999999`. Values outside that range raise
|
|
99
|
+
`ValueError`; non-integers raise `TypeError`. `number_to_words_digit_by_digit()` accepts
|
|
100
|
+
a nonempty string of ASCII digits only and preserves leading zeroes. Its invalid input
|
|
101
|
+
raises `ValueError`. Ordinal forms are `nom_m`, `nom_n`, `nom_f`, `nom_pl`, `gen`, `dat`,
|
|
102
|
+
`prep`, `loc`, `pl`, `loc_pl`, `acc_f`, `gen_f`, `ins`, `ins_f`, `ins_pl`, and `loc_f`.
|
|
103
|
+
Cardinal cases are `gen`, `dat`, `instr`, and `prep`. Unknown forms raise `ValueError`.
|
|
104
|
+
|
|
105
|
+
The legacy spellings `sentenize()`, `cyrilize()`, and `cyrrilize()` remain available
|
|
106
|
+
but issue `DeprecationWarning`; use `split_sentences()` and
|
|
107
|
+
`transliterate_to_cyrillic()` in new code.
|
|
108
|
+
|
|
109
|
+
`NormalizeOptions`, `Substring`, and `UncertainSpan` support `copy.copy()`,
|
|
110
|
+
`copy.deepcopy()`, and `pickle` serialization. Copies are independent value objects.
|
|
111
|
+
|
|
112
|
+
## Currency and cryptocurrency coverage
|
|
113
|
+
|
|
114
|
+
Normalization covers all 178 active ISO 4217 List One codes, including their
|
|
115
|
+
0-, 2-, 3-, or 4-digit minor-unit rules. More than 70 common cryptocurrency and
|
|
116
|
+
finance tickers have natural Ukrainian readings. Other 2–10 character uppercase
|
|
117
|
+
alphanumeric tickers are spelled out after amounts and when paired with a known
|
|
118
|
+
asset, so newly introduced assets do not require an immediate library release.
|
|
119
|
+
Prefix and suffix
|
|
120
|
+
amounts, localized thousands separators, signs, decimals, and the `₿` symbol
|
|
121
|
+
are supported.
|
|
122
|
+
|
|
123
|
+
## Ambiguity controls
|
|
124
|
+
|
|
125
|
+
`NormalizeOptions` keeps backward-compatible defaults while allowing callers to resolve ambiguous input explicitly:
|
|
126
|
+
|
|
127
|
+
- `colon_style`: contextual clock/ratio detection, forced clock, or forced ratio.
|
|
128
|
+
- `numeric_date_order`: day-month-year, month-day-year, or preservation of dates where both fields are at most 12.
|
|
129
|
+
- `currency_symbol_policy`: assume the common currency for `$` and `¥`, or preserve those ambiguous symbols.
|
|
130
|
+
|
|
131
|
+
The CLI exposes the same controls through `--colon-style`, `--date-order`, and
|
|
132
|
+
`--preserve-ambiguous-currency`.
|
|
133
|
+
|
|
134
|
+
## Benchmarks and fuzzing
|
|
135
|
+
|
|
136
|
+
Build and run the benchmark explicitly:
|
|
137
|
+
|
|
138
|
+
```sh
|
|
139
|
+
cmake --build build --target uktextnorm_benchmark
|
|
140
|
+
./build/uktextnorm_benchmark .
|
|
141
|
+
```
|
|
142
|
+
|
|
143
|
+
For Python binding timings, install the package and run
|
|
144
|
+
`python benchmarks/python_binding_benchmark.py`. It compares scalar and batched
|
|
145
|
+
normalization, including unique inputs, and span-returning calls.
|
|
146
|
+
|
|
147
|
+
With Clang and libFuzzer support, build the normalization harness with sanitizers:
|
|
148
|
+
|
|
149
|
+
```sh
|
|
150
|
+
cmake -S . -B build-fuzz -DCMAKE_CXX_COMPILER=clang++ -DNORMALIZE_UK_CPP_BUILD_FUZZER=ON
|
|
151
|
+
cmake --build build-fuzz --target uktextnorm_fuzzer
|
|
152
|
+
./build-fuzz/uktextnorm_fuzzer -max_total_time=60 tests/data
|
|
153
|
+
```
|
|
154
|
+
|
|
155
|
+
## Development
|
|
156
|
+
|
|
157
|
+
Install `uv` and `just` for development. Run `just` to see all recipes. `just format` applies Ruff fixes and Python formatting; `just lint` runs the Python static checks; `just check` also runs Python and C++ tests.
|
|
158
|
+
|
|
159
|
+
```sh
|
|
160
|
+
just setup
|
|
161
|
+
just format
|
|
162
|
+
just lint
|
|
163
|
+
just check
|
|
164
|
+
just wheel 3.15
|
|
165
|
+
```
|
|
166
|
+
|
|
167
|
+
The project includes a `.clang-format` file and a CMake formatting target. Install `clang-format`, then run:
|
|
168
|
+
|
|
169
|
+
```sh
|
|
170
|
+
cmake --build build --target format
|
|
171
|
+
```
|
|
@@ -0,0 +1,11 @@
|
|
|
1
|
+
normalize_uk/py.typed,sha256=frcCV1k9oG9oKj3dpUqdJg1PxRT2RSN_XKdLCPjaYaY,2
|
|
2
|
+
normalize_uk/_normalize_uk.cp315-win_amd64.pyd,sha256=qv34QHoSAaISipbxyrrlUIilM_ZaRF5CyzxEBlE8tPo,3089408
|
|
3
|
+
normalize_uk/_normalize_uk.pyi,sha256=5h1sHW7-FJeskuuant4AoMfJSNAJxvgaGvunyw6pXzY,5961
|
|
4
|
+
normalize_uk/__init__.py,sha256=-zN5qc6V28A7mZCKcsqWWHfV48a4fS0J344J9x8oCbQ,8367
|
|
5
|
+
normalize_uk/__init__.pyi,sha256=UrTlJAckVhsfHykAJbCoi_BKarv2FC0o9Bu0m6qUmQA,3598
|
|
6
|
+
normalize_uk-0.4.2.dist-info/DELVEWHEEL,sha256=IK6a-eKDrniSWv60rLdIEX6_LO6e6-4E6Mt-RA57Xi4,411
|
|
7
|
+
normalize_uk-0.4.2.dist-info/METADATA,sha256=TCRiWCKAJmJuCcg_81Trh5U7lrpJV8eKmk388Ivd47o,7208
|
|
8
|
+
normalize_uk-0.4.2.dist-info/RECORD,,
|
|
9
|
+
normalize_uk-0.4.2.dist-info/WHEEL,sha256=zRSxk0lFSz9yrcoNCqCgDrmWhvhsSyignczP3ORAtmA,105
|
|
10
|
+
normalize_uk-0.4.2.dist-info/licenses/LICENSE,sha256=H74gUlraCIIvf60AlpOT_K4KB_BN_sH9_zhIwYkT4VU,1093
|
|
11
|
+
normalize_uk.libs/msvcp140-a4c2229bdc2a2a630acdc095b4d86008.dll,sha256=pMIim9wqKmMKzcCVtNhgCOXD47x3cxdDVPPaT1vrnN4,575056
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Yehor Smoliakov
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
Binary file
|