normalize-uk 0.4.2__cp315-cp315-win_amd64.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,273 @@
1
+ """Ukrainian text normalization and tokenization."""
2
+
3
+ from __future__ import annotations
4
+
5
+
6
+ # start delvewheel patch
7
+ def _delvewheel_patch_1_13_1():
8
+ import os
9
+ if os.path.isdir(libs_dir := os.path.abspath(os.path.join(os.path.dirname(__file__), os.pardir, 'normalize_uk.libs'))):
10
+ os.add_dll_directory(libs_dir)
11
+
12
+
13
+ _delvewheel_patch_1_13_1()
14
+ del _delvewheel_patch_1_13_1
15
+ # end delvewheel patch
16
+
17
+ import warnings as _warnings
18
+ from collections.abc import Iterable
19
+
20
+ from ._normalize_uk import (
21
+ ColonStyle,
22
+ CurrencySymbolPolicy,
23
+ DateStyle,
24
+ NormalizeOptions,
25
+ NormalizePreset,
26
+ NumericDateOrder,
27
+ PhoneStyle,
28
+ QuoteStyle,
29
+ RangeStyle,
30
+ Substring,
31
+ SymbolStyle,
32
+ UncertainSpan,
33
+ UncertaintyCategory,
34
+ UncertaintySeverity,
35
+ expand_abbreviations,
36
+ normalize_abbreviations,
37
+ options_for_preset,
38
+ split_sentences,
39
+ tokenize,
40
+ transliterate_to_cyrillic,
41
+ )
42
+ from ._normalize_uk import (
43
+ cyrilize as _native_cyrilize,
44
+ )
45
+ from ._normalize_uk import (
46
+ cyrrilize as _native_cyrrilize,
47
+ )
48
+ from ._normalize_uk import (
49
+ flag_uncertain as _native_flag_uncertain,
50
+ )
51
+ from ._normalize_uk import (
52
+ normalize_ukrainian as _native_normalize_ukrainian,
53
+ )
54
+ from ._normalize_uk import (
55
+ normalize_ukrainian_many as _native_normalize_ukrainian_many,
56
+ )
57
+ from ._normalize_uk import (
58
+ number_to_ordinal_words as _native_number_to_ordinal_words,
59
+ )
60
+ from ._normalize_uk import (
61
+ number_to_words as _native_number_to_words,
62
+ )
63
+ from ._normalize_uk import (
64
+ number_to_words_case as _native_number_to_words_case,
65
+ )
66
+ from ._normalize_uk import (
67
+ number_to_words_digit_by_digit as _native_number_to_words_digit_by_digit,
68
+ )
69
+ from ._normalize_uk import (
70
+ sentenize as _native_sentenize,
71
+ )
72
+
73
+ __all__ = (
74
+ "ColonStyle",
75
+ "CurrencySymbolPolicy",
76
+ "DateStyle",
77
+ "NormalizeOptions",
78
+ "NormalizePreset",
79
+ "NumericDateOrder",
80
+ "PhoneStyle",
81
+ "QuoteStyle",
82
+ "RangeStyle",
83
+ "Substring",
84
+ "SymbolStyle",
85
+ "UncertainSpan",
86
+ "UncertaintyCategory",
87
+ "UncertaintySeverity",
88
+ "cyrilize",
89
+ "cyrrilize",
90
+ "expand_abbreviations",
91
+ "flag_uncertain",
92
+ "normalize_abbreviations",
93
+ "normalize_ukrainian",
94
+ "normalize_ukrainian_many",
95
+ "normalize_ukrainian_with_preset",
96
+ "number_to_ordinal_words",
97
+ "number_to_words",
98
+ "number_to_words_case",
99
+ "number_to_words_digit_by_digit",
100
+ "options_for_preset",
101
+ "sentenize",
102
+ "split_sentences",
103
+ "tokenize",
104
+ "transliterate_to_cyrillic",
105
+ )
106
+
107
+ _MAX_SPOKEN_NUMBER = 10**18 - 1
108
+ _ORDINAL_FORMS = frozenset(
109
+ (
110
+ "nom_m",
111
+ "nom_n",
112
+ "nom_f",
113
+ "nom_pl",
114
+ "gen",
115
+ "dat",
116
+ "prep",
117
+ "loc",
118
+ "pl",
119
+ "loc_pl",
120
+ "acc_f",
121
+ "gen_f",
122
+ "ins",
123
+ "ins_f",
124
+ "ins_pl",
125
+ "loc_f",
126
+ )
127
+ )
128
+ _CARDINAL_CASES = frozenset(("gen", "dat", "instr", "prep"))
129
+
130
+
131
+ def _spoken_number(n: int) -> int:
132
+ if isinstance(n, bool) or not isinstance(n, int):
133
+ raise TypeError("n must be an integer")
134
+ if not 0 <= n <= _MAX_SPOKEN_NUMBER:
135
+ raise ValueError(f"n must be between 0 and {_MAX_SPOKEN_NUMBER}")
136
+ return n
137
+
138
+
139
+ def number_to_words(n: int) -> str:
140
+ """Spell a nonnegative integer smaller than 10**18 in Ukrainian."""
141
+ return _native_number_to_words(_spoken_number(n))
142
+
143
+
144
+ def number_to_words_digit_by_digit(digits: str) -> str:
145
+ """Spell each ASCII digit, preserving leading zeroes."""
146
+ if not isinstance(digits, str):
147
+ raise TypeError("digits must be a string")
148
+ if not digits or any(ch < "0" or ch > "9" for ch in digits):
149
+ raise ValueError("digits must contain one or more ASCII digits only")
150
+ return _native_number_to_words_digit_by_digit(digits)
151
+
152
+
153
+ def number_to_ordinal_words(n: int, form: str = "nom_m") -> str:
154
+ """Spell an ordinal using one of the supported grammatical forms."""
155
+ if not isinstance(form, str):
156
+ raise TypeError("form must be a string")
157
+ if form not in _ORDINAL_FORMS:
158
+ raise ValueError(
159
+ f"unknown ordinal form {form!r}; expected one of {', '.join(sorted(_ORDINAL_FORMS))}"
160
+ )
161
+ return _native_number_to_ordinal_words(_spoken_number(n), form)
162
+
163
+
164
+ def number_to_words_case(n: int, grammatical_case: str) -> str:
165
+ """Spell a cardinal in the genitive, dative, instrumental, or prepositional case."""
166
+ if not isinstance(grammatical_case, str):
167
+ raise TypeError("grammatical_case must be a string")
168
+ if grammatical_case not in _CARDINAL_CASES:
169
+ raise ValueError(
170
+ f"unknown grammatical case {grammatical_case!r}; expected one of "
171
+ f"{', '.join(sorted(_CARDINAL_CASES))}"
172
+ )
173
+ return _native_number_to_words_case(_spoken_number(n), grammatical_case)
174
+
175
+
176
+ def _selection(
177
+ options: NormalizeOptions | NormalizePreset | None,
178
+ preset: NormalizePreset | None,
179
+ ) -> NormalizeOptions | NormalizePreset | None:
180
+ if isinstance(options, NormalizePreset):
181
+ if preset is not None:
182
+ raise ValueError("provide either options or preset, not both")
183
+ preset, options = options, None
184
+ if options is not None and preset is not None:
185
+ raise ValueError("provide either options or preset, not both")
186
+ if options is not None:
187
+ if not isinstance(options, NormalizeOptions):
188
+ raise TypeError("options must be a NormalizeOptions instance")
189
+ return options
190
+ if preset is not None:
191
+ if not isinstance(preset, NormalizePreset):
192
+ raise TypeError("preset must be a NormalizePreset value")
193
+ return preset
194
+ return None
195
+
196
+
197
+ def normalize_ukrainian(
198
+ text: str,
199
+ options: NormalizeOptions | NormalizePreset | None = None,
200
+ *,
201
+ preset: NormalizePreset | None = None,
202
+ ) -> str:
203
+ """Normalize text with either an options object or a preset."""
204
+ selected = _selection(options, preset)
205
+ if selected is None:
206
+ return _native_normalize_ukrainian(text)
207
+ if isinstance(selected, NormalizePreset):
208
+ return _native_normalize_ukrainian(text, selected)
209
+ return _native_normalize_ukrainian(text, selected)
210
+
211
+
212
+ def normalize_ukrainian_many(
213
+ texts: Iterable[str],
214
+ options: NormalizeOptions | NormalizePreset | None = None,
215
+ *,
216
+ preset: NormalizePreset | None = None,
217
+ ) -> list[str]:
218
+ """Normalize an iterable of strings using one options snapshot for the batch."""
219
+ selected = _selection(options, preset)
220
+ if selected is None:
221
+ return _native_normalize_ukrainian_many(texts)
222
+ if isinstance(selected, NormalizePreset):
223
+ return _native_normalize_ukrainian_many(texts, selected)
224
+ return _native_normalize_ukrainian_many(texts, selected)
225
+
226
+
227
+ def flag_uncertain(
228
+ text: str,
229
+ options: NormalizeOptions | NormalizePreset | None = None,
230
+ *,
231
+ preset: NormalizePreset | None = None,
232
+ ) -> list[UncertainSpan]:
233
+ """Find uncertain spans; explicit options suppress resolved ambiguity warnings."""
234
+ selected = _selection(options, preset)
235
+ if selected is None:
236
+ return _native_flag_uncertain(text)
237
+ if isinstance(selected, NormalizePreset):
238
+ return _native_flag_uncertain(text, selected)
239
+ return _native_flag_uncertain(text, selected)
240
+
241
+
242
+ def normalize_ukrainian_with_preset(
243
+ text: str, preset: NormalizePreset = NormalizePreset.Default
244
+ ) -> str:
245
+ """Compatibility alias for normalize_ukrainian(text, preset=preset)."""
246
+ return normalize_ukrainian(text, preset=preset)
247
+
248
+
249
+ def sentenize(text: str) -> list[Substring]:
250
+ _warnings.warn(
251
+ "sentenize() is deprecated; use split_sentences()",
252
+ DeprecationWarning,
253
+ stacklevel=2,
254
+ )
255
+ return _native_sentenize(text)
256
+
257
+
258
+ def cyrilize(text: str) -> str:
259
+ _warnings.warn(
260
+ "cyrilize() is deprecated; use transliterate_to_cyrillic()",
261
+ DeprecationWarning,
262
+ stacklevel=2,
263
+ )
264
+ return _native_cyrilize(text)
265
+
266
+
267
+ def cyrrilize(text: str) -> str:
268
+ _warnings.warn(
269
+ "cyrrilize() is deprecated; use transliterate_to_cyrillic()",
270
+ DeprecationWarning,
271
+ stacklevel=2,
272
+ )
273
+ return _native_cyrrilize(text)
@@ -0,0 +1,135 @@
1
+ from collections.abc import Iterable
2
+ from typing import Literal, overload
3
+
4
+ from ._normalize_uk import (
5
+ ColonStyle as ColonStyle,
6
+ )
7
+ from ._normalize_uk import (
8
+ CurrencySymbolPolicy as CurrencySymbolPolicy,
9
+ )
10
+ from ._normalize_uk import (
11
+ DateStyle as DateStyle,
12
+ )
13
+ from ._normalize_uk import (
14
+ NormalizeOptions as NormalizeOptions,
15
+ )
16
+ from ._normalize_uk import (
17
+ NormalizePreset as NormalizePreset,
18
+ )
19
+ from ._normalize_uk import (
20
+ NumericDateOrder as NumericDateOrder,
21
+ )
22
+ from ._normalize_uk import (
23
+ PhoneStyle as PhoneStyle,
24
+ )
25
+ from ._normalize_uk import (
26
+ QuoteStyle as QuoteStyle,
27
+ )
28
+ from ._normalize_uk import (
29
+ RangeStyle as RangeStyle,
30
+ )
31
+ from ._normalize_uk import (
32
+ Substring as Substring,
33
+ )
34
+ from ._normalize_uk import (
35
+ SymbolStyle as SymbolStyle,
36
+ )
37
+ from ._normalize_uk import (
38
+ UncertainSpan as UncertainSpan,
39
+ )
40
+ from ._normalize_uk import (
41
+ UncertaintyCategory as UncertaintyCategory,
42
+ )
43
+ from ._normalize_uk import (
44
+ UncertaintySeverity as UncertaintySeverity,
45
+ )
46
+ from ._normalize_uk import (
47
+ expand_abbreviations as expand_abbreviations,
48
+ )
49
+ from ._normalize_uk import (
50
+ normalize_abbreviations as normalize_abbreviations,
51
+ )
52
+ from ._normalize_uk import (
53
+ options_for_preset as options_for_preset,
54
+ )
55
+ from ._normalize_uk import (
56
+ split_sentences as split_sentences,
57
+ )
58
+ from ._normalize_uk import (
59
+ tokenize as tokenize,
60
+ )
61
+ from ._normalize_uk import (
62
+ transliterate_to_cyrillic as transliterate_to_cyrillic,
63
+ )
64
+
65
+ def number_to_words(n: int) -> str: ...
66
+ def number_to_words_digit_by_digit(digits: str) -> str: ...
67
+ def number_to_ordinal_words(
68
+ n: int,
69
+ form: Literal[
70
+ "nom_m",
71
+ "nom_n",
72
+ "nom_f",
73
+ "nom_pl",
74
+ "gen",
75
+ "dat",
76
+ "prep",
77
+ "loc",
78
+ "pl",
79
+ "loc_pl",
80
+ "acc_f",
81
+ "gen_f",
82
+ "ins",
83
+ "ins_f",
84
+ "ins_pl",
85
+ "loc_f",
86
+ ] = "nom_m",
87
+ ) -> str: ...
88
+ def number_to_words_case(
89
+ n: int, grammatical_case: Literal["gen", "dat", "instr", "prep"]
90
+ ) -> str: ...
91
+ def cyrilize(text: str) -> str: ...
92
+ def cyrrilize(text: str) -> str: ...
93
+ @overload
94
+ def normalize_ukrainian(
95
+ text: str, options: NormalizeOptions | None = None, *, preset: None = None
96
+ ) -> str: ...
97
+ @overload
98
+ def normalize_ukrainian(
99
+ text: str, options: NormalizePreset, *, preset: None = None
100
+ ) -> str: ...
101
+ @overload
102
+ def normalize_ukrainian(
103
+ text: str, options: None = None, *, preset: NormalizePreset
104
+ ) -> str: ...
105
+ def normalize_ukrainian_with_preset(
106
+ text: str, preset: NormalizePreset = NormalizePreset.Default
107
+ ) -> str: ...
108
+ @overload
109
+ def normalize_ukrainian_many(
110
+ texts: Iterable[str],
111
+ options: NormalizeOptions | None = None,
112
+ *,
113
+ preset: None = None,
114
+ ) -> list[str]: ...
115
+ @overload
116
+ def normalize_ukrainian_many(
117
+ texts: Iterable[str], options: NormalizePreset, *, preset: None = None
118
+ ) -> list[str]: ...
119
+ @overload
120
+ def normalize_ukrainian_many(
121
+ texts: Iterable[str], options: None = None, *, preset: NormalizePreset
122
+ ) -> list[str]: ...
123
+ @overload
124
+ def flag_uncertain(
125
+ text: str, options: NormalizeOptions | None = None, *, preset: None = None
126
+ ) -> list[UncertainSpan]: ...
127
+ @overload
128
+ def flag_uncertain(
129
+ text: str, options: NormalizePreset, *, preset: None = None
130
+ ) -> list[UncertainSpan]: ...
131
+ @overload
132
+ def flag_uncertain(
133
+ text: str, options: None = None, *, preset: NormalizePreset
134
+ ) -> list[UncertainSpan]: ...
135
+ def sentenize(text: str) -> list[Substring]: ...
@@ -0,0 +1,181 @@
1
+ from collections.abc import Iterable
2
+ from enum import Enum
3
+ from typing import overload
4
+
5
+ class UncertaintyCategory(Enum):
6
+ AmbiguousAbbreviation: UncertaintyCategory
7
+ BareNumber: UncertaintyCategory
8
+ Currency: UncertaintyCategory
9
+ Date: UncertaintyCategory
10
+ Identifier: UncertaintyCategory
11
+ ForeignWord: UncertaintyCategory
12
+ MixedScript: UncertaintyCategory
13
+ RomanNumeral: UncertaintyCategory
14
+ Unit: UncertaintyCategory
15
+ Web: UncertaintyCategory
16
+ InvalidDate: UncertaintyCategory
17
+ AmbiguousNumberGrouping: UncertaintyCategory
18
+ Agreement: UncertaintyCategory
19
+ Time: UncertaintyCategory
20
+ Fraction: UncertaintyCategory
21
+ Network: UncertaintyCategory
22
+ Scientific: UncertaintyCategory
23
+ Coordinate: UncertaintyCategory
24
+
25
+ class UncertaintySeverity(Enum):
26
+ Info: UncertaintySeverity
27
+ Warning: UncertaintySeverity
28
+ Error: UncertaintySeverity
29
+
30
+ class RangeStyle(Enum):
31
+ Compact: RangeStyle
32
+ FromTo: RangeStyle
33
+
34
+ class PhoneStyle(Enum):
35
+ Grouped: PhoneStyle
36
+ DigitByDigit: PhoneStyle
37
+
38
+ class SymbolStyle(Enum):
39
+ Expand: SymbolStyle
40
+ Preserve: SymbolStyle
41
+
42
+ class DateStyle(Enum):
43
+ Formal: DateStyle
44
+ Spoken: DateStyle
45
+
46
+ class ColonStyle(Enum):
47
+ Contextual: ColonStyle
48
+ Clock: ColonStyle
49
+ Ratio: ColonStyle
50
+
51
+ class NumericDateOrder(Enum):
52
+ DayMonthYear: NumericDateOrder
53
+ MonthDayYear: NumericDateOrder
54
+ PreserveAmbiguous: NumericDateOrder
55
+
56
+ class CurrencySymbolPolicy(Enum):
57
+ AssumeCommon: CurrencySymbolPolicy
58
+ PreserveAmbiguous: CurrencySymbolPolicy
59
+
60
+ class QuoteStyle(Enum):
61
+ Keep: QuoteStyle
62
+ Guillemets: QuoteStyle
63
+ Straight: QuoteStyle
64
+ Strip: QuoteStyle
65
+
66
+ class NormalizePreset(Enum):
67
+ Default: NormalizePreset
68
+ TtsFriendly: NormalizePreset
69
+ Conservative: NormalizePreset
70
+ SearchIndexing: NormalizePreset
71
+
72
+ class UncertainSpan:
73
+ @property
74
+ def start(self) -> int: ...
75
+ @property
76
+ def stop(self) -> int: ...
77
+ @property
78
+ def text(self) -> str: ...
79
+ @property
80
+ def reason(self) -> str: ...
81
+ @property
82
+ def category(self) -> UncertaintyCategory: ...
83
+ @property
84
+ def severity(self) -> UncertaintySeverity: ...
85
+ def __eq__(self, other: object) -> bool: ...
86
+ def __repr__(self) -> str: ...
87
+ def __copy__(self) -> UncertainSpan: ...
88
+ def __deepcopy__(self, memo: dict[object, object]) -> UncertainSpan: ...
89
+
90
+ class NormalizeOptions:
91
+ expand_known_acronyms: bool
92
+ spell_unknown_acronyms: bool
93
+ normalize_english_words: bool
94
+ transliterate_latin: bool
95
+ range_style: RangeStyle
96
+ phone_style: PhoneStyle
97
+ symbol_style: SymbolStyle
98
+ date_style: DateStyle
99
+ colon_style: ColonStyle
100
+ numeric_date_order: NumericDateOrder
101
+ currency_symbol_policy: CurrencySymbolPolicy
102
+ repair_homoglyphs: bool
103
+ validate_dates: bool
104
+ parse_thousand_separators: bool
105
+ normalize_network_addresses: bool
106
+ quote_style: QuoteStyle
107
+
108
+ def __init__(
109
+ self,
110
+ preset: NormalizePreset = NormalizePreset.Default,
111
+ *,
112
+ expand_known_acronyms: bool = ...,
113
+ spell_unknown_acronyms: bool = ...,
114
+ normalize_english_words: bool = ...,
115
+ transliterate_latin: bool = ...,
116
+ repair_homoglyphs: bool = ...,
117
+ validate_dates: bool = ...,
118
+ parse_thousand_separators: bool = ...,
119
+ normalize_network_addresses: bool = ...,
120
+ range_style: RangeStyle = ...,
121
+ phone_style: PhoneStyle = ...,
122
+ symbol_style: SymbolStyle = ...,
123
+ date_style: DateStyle = ...,
124
+ colon_style: ColonStyle = ...,
125
+ numeric_date_order: NumericDateOrder = ...,
126
+ currency_symbol_policy: CurrencySymbolPolicy = ...,
127
+ quote_style: QuoteStyle = ...,
128
+ ) -> None: ...
129
+ def __copy__(self) -> NormalizeOptions: ...
130
+ def __deepcopy__(self, memo: dict[object, object]) -> NormalizeOptions: ...
131
+
132
+ class Substring:
133
+ @property
134
+ def start(self) -> int: ...
135
+ @property
136
+ def stop(self) -> int: ...
137
+ @property
138
+ def text(self) -> str: ...
139
+ def __eq__(self, other: object) -> bool: ...
140
+ def __repr__(self) -> str: ...
141
+ def __copy__(self) -> Substring: ...
142
+ def __deepcopy__(self, memo: dict[object, object]) -> Substring: ...
143
+
144
+ def options_for_preset(preset: NormalizePreset) -> NormalizeOptions: ...
145
+ def number_to_words(n: int) -> str: ...
146
+ def number_to_words_digit_by_digit(digits: str) -> str: ...
147
+ def number_to_ordinal_words(n: int, form: str = "nom_m") -> str: ...
148
+ def number_to_words_case(n: int, grammatical_case: str) -> str: ...
149
+ def normalize_abbreviations(text: str) -> str: ...
150
+ def expand_abbreviations(text: str) -> str: ...
151
+ def transliterate_to_cyrillic(text: str) -> str: ...
152
+ def cyrilize(text: str) -> str: ...
153
+ def cyrrilize(text: str) -> str: ...
154
+ @overload
155
+ def normalize_ukrainian(text: str) -> str: ...
156
+ @overload
157
+ def normalize_ukrainian(text: str, options: NormalizeOptions) -> str: ...
158
+ @overload
159
+ def normalize_ukrainian(text: str, preset: NormalizePreset) -> str: ...
160
+ @overload
161
+ def normalize_ukrainian_many(texts: Iterable[str]) -> list[str]: ...
162
+ @overload
163
+ def normalize_ukrainian_many(
164
+ texts: Iterable[str], options: NormalizeOptions
165
+ ) -> list[str]: ...
166
+ @overload
167
+ def normalize_ukrainian_many(
168
+ texts: Iterable[str], preset: NormalizePreset
169
+ ) -> list[str]: ...
170
+ def normalize_ukrainian_with_preset(
171
+ text: str, preset: NormalizePreset = NormalizePreset.Default
172
+ ) -> str: ...
173
+ @overload
174
+ def flag_uncertain(text: str) -> list[UncertainSpan]: ...
175
+ @overload
176
+ def flag_uncertain(text: str, options: NormalizeOptions) -> list[UncertainSpan]: ...
177
+ @overload
178
+ def flag_uncertain(text: str, preset: NormalizePreset) -> list[UncertainSpan]: ...
179
+ def split_sentences(text: str) -> list[Substring]: ...
180
+ def sentenize(text: str) -> list[Substring]: ...
181
+ def tokenize(text: str) -> list[Substring]: ...
normalize_uk/py.typed ADDED
@@ -0,0 +1 @@
1
+
@@ -0,0 +1,2 @@
1
+ Version: 1.13.1
2
+ Arguments: ['C:\\Users\\runneradmin\\AppData\\Local\\Temp\\cibw-run-3cq3h79w\\cp315-win_amd64\\build\\venv\\Scripts\\delvewheel', 'repair', '-w', 'C:\\Users\\runneradmin\\AppData\\Local\\Temp\\cibw-run-3cq3h79w\\cp315-win_amd64\\repaired_wheel', '-v', 'C:\\Users\\runneradmin\\AppData\\Local\\Temp\\cibw-run-3cq3h79w\\cp315-win_amd64\\built_wheel\\normalize_uk-0.4.2-cp315-cp315-win_amd64.whl']
@@ -0,0 +1,171 @@
1
+ Metadata-Version: 2.4
2
+ Name: normalize-uk
3
+ Version: 0.4.2
4
+ Summary: Python bindings for Ukrainian text normalization and tokenization utilities
5
+ License-Expression: MIT
6
+ Classifier: Programming Language :: Python :: 3
7
+ Classifier: Programming Language :: Python :: 3.10
8
+ Classifier: Programming Language :: Python :: 3.11
9
+ Classifier: Programming Language :: Python :: 3.12
10
+ Classifier: Programming Language :: Python :: 3.13
11
+ Classifier: Programming Language :: Python :: 3.14
12
+ Classifier: Programming Language :: Python :: 3.15
13
+ Classifier: Programming Language :: C++
14
+ Requires-Python: >=3.10
15
+ Description-Content-Type: text/markdown
16
+
17
+ # normalize-uk-cpp
18
+
19
+ [![CI](https://github.com/ThirdLetterC/normalize_uk-cpp/actions/workflows/ci.yml/badge.svg)](https://github.com/ThirdLetterC/normalize_uk-cpp/actions/workflows/ci.yml)
20
+ [![Release](https://github.com/ThirdLetterC/normalize_uk-cpp/actions/workflows/release.yml/badge.svg)](https://github.com/ThirdLetterC/normalize_uk-cpp/actions/workflows/release.yml)
21
+
22
+ C++23 Ukrainian text normalization and tokenization utilities with optional Python 3.10+ bindings.
23
+
24
+ ## CMake
25
+
26
+ ```sh
27
+ cmake -S . -B build
28
+ cmake --build build
29
+ ctest --test-dir build
30
+ ```
31
+
32
+ Enable Python bindings explicitly when building with CMake:
33
+
34
+ ```sh
35
+ cmake -S . -B build-python -DNORMALIZE_UK_CPP_BUILD_PYTHON=ON
36
+ cmake --build build-python
37
+ ```
38
+
39
+ A regular CMake install includes the C++ library, headers, and CMake package.
40
+ Python wheels contain only the Python package and compiled extension.
41
+
42
+ ## Python
43
+
44
+ ```sh
45
+ python -m pip install .
46
+ ```
47
+
48
+ ```python
49
+ import normalize_uk as nuk
50
+
51
+ print(nuk.number_to_words(123))
52
+ print(nuk.normalize_ukrainian("01.05.2024"))
53
+ print(nuk.normalize_ukrainian("01.05.2024", preset=nuk.NormalizePreset.TtsFriendly))
54
+ print(nuk.normalize_ukrainian_many(["01.05.2024", "5 кг"]))
55
+ print([sentence.text for sentence in nuk.split_sentences("П'ять зв'язків. Два.")])
56
+ print([token.text for token in nuk.tokenize("П'ять зв'язків.")])
57
+ ```
58
+
59
+ More examples live in `examples/python/`.
60
+
61
+ Tags matching the version in `pyproject.toml` (for example, `v0.4.2`) trigger
62
+ wheel builds for supported Python versions. The workflow uploads the wheels to
63
+ GitHub Release Assets, then downloads those Assets and publishes them to PyPI.
64
+ To enable PyPI Trusted Publishing, register `ThirdLetterC/normalize_uk-cpp` as
65
+ a publisher for `normalize-uk` with workflow `release.yml` and environment
66
+ `pypi`. For a new PyPI project, register a
67
+ [pending publisher](https://docs.pypi.org/trusted-publishers/creating-a-project-through-oidc/)
68
+ first. No PyPI API token is needed.
69
+
70
+ `NormalizeOptions` accepts a preset and named overrides at construction time:
71
+
72
+ ```python
73
+ options = nuk.NormalizeOptions(
74
+ preset=nuk.NormalizePreset.TtsFriendly,
75
+ range_style=nuk.RangeStyle.Compact,
76
+ numeric_date_order=nuk.NumericDateOrder.DayMonthYear,
77
+ )
78
+ result = nuk.normalize_ukrainian("5–7 кг", options=options)
79
+ spans = nuk.flag_uncertain("10:30, $12", options=options)
80
+ ```
81
+
82
+ Pass either `options=` or `preset=` to `normalize_ukrainian` and `flag_uncertain`.
83
+ `normalize_ukrainian_many()` accepts any iterable of Python strings and applies one
84
+ options snapshot to the entire batch. It returns a list in input order and raises
85
+ `TypeError` if an item is not a string. It accepts the same `options=` and `preset=`
86
+ selection as `normalize_ukrainian`. Identical strings within one batch are
87
+ normalized once and their result is reused.
88
+ The older positional options/preset calls and `normalize_ukrainian_with_preset()` remain available.
89
+ Without options, `flag_uncertain()` reports all ambiguity candidates. With explicit options
90
+ or a preset, it omits warnings for ambiguous dates, colon pairs, and currency symbols
91
+ when the selected policy resolves them; invalid-value diagnostics remain.
92
+
93
+ `Substring.start`/`stop` and `UncertainSpan.start`/`stop` are Python `str` indexes,
94
+ with `stop` exclusive: `span.text == source[span.start:span.stop]`. They count Unicode
95
+ code points, matching Python slicing, rather than UTF-8 bytes.
96
+
97
+ `number_to_words()`, `number_to_ordinal_words()`, and `number_to_words_case()` accept
98
+ integers from 0 through `999999999999999999`. Values outside that range raise
99
+ `ValueError`; non-integers raise `TypeError`. `number_to_words_digit_by_digit()` accepts
100
+ a nonempty string of ASCII digits only and preserves leading zeroes. Its invalid input
101
+ raises `ValueError`. Ordinal forms are `nom_m`, `nom_n`, `nom_f`, `nom_pl`, `gen`, `dat`,
102
+ `prep`, `loc`, `pl`, `loc_pl`, `acc_f`, `gen_f`, `ins`, `ins_f`, `ins_pl`, and `loc_f`.
103
+ Cardinal cases are `gen`, `dat`, `instr`, and `prep`. Unknown forms raise `ValueError`.
104
+
105
+ The legacy spellings `sentenize()`, `cyrilize()`, and `cyrrilize()` remain available
106
+ but issue `DeprecationWarning`; use `split_sentences()` and
107
+ `transliterate_to_cyrillic()` in new code.
108
+
109
+ `NormalizeOptions`, `Substring`, and `UncertainSpan` support `copy.copy()`,
110
+ `copy.deepcopy()`, and `pickle` serialization. Copies are independent value objects.
111
+
112
+ ## Currency and cryptocurrency coverage
113
+
114
+ Normalization covers all 178 active ISO 4217 List One codes, including their
115
+ 0-, 2-, 3-, or 4-digit minor-unit rules. More than 70 common cryptocurrency and
116
+ finance tickers have natural Ukrainian readings. Other 2–10 character uppercase
117
+ alphanumeric tickers are spelled out after amounts and when paired with a known
118
+ asset, so newly introduced assets do not require an immediate library release.
119
+ Prefix and suffix
120
+ amounts, localized thousands separators, signs, decimals, and the `₿` symbol
121
+ are supported.
122
+
123
+ ## Ambiguity controls
124
+
125
+ `NormalizeOptions` keeps backward-compatible defaults while allowing callers to resolve ambiguous input explicitly:
126
+
127
+ - `colon_style`: contextual clock/ratio detection, forced clock, or forced ratio.
128
+ - `numeric_date_order`: day-month-year, month-day-year, or preservation of dates where both fields are at most 12.
129
+ - `currency_symbol_policy`: assume the common currency for `$` and `¥`, or preserve those ambiguous symbols.
130
+
131
+ The CLI exposes the same controls through `--colon-style`, `--date-order`, and
132
+ `--preserve-ambiguous-currency`.
133
+
134
+ ## Benchmarks and fuzzing
135
+
136
+ Build and run the benchmark explicitly:
137
+
138
+ ```sh
139
+ cmake --build build --target uktextnorm_benchmark
140
+ ./build/uktextnorm_benchmark .
141
+ ```
142
+
143
+ For Python binding timings, install the package and run
144
+ `python benchmarks/python_binding_benchmark.py`. It compares scalar and batched
145
+ normalization, including unique inputs, and span-returning calls.
146
+
147
+ With Clang and libFuzzer support, build the normalization harness with sanitizers:
148
+
149
+ ```sh
150
+ cmake -S . -B build-fuzz -DCMAKE_CXX_COMPILER=clang++ -DNORMALIZE_UK_CPP_BUILD_FUZZER=ON
151
+ cmake --build build-fuzz --target uktextnorm_fuzzer
152
+ ./build-fuzz/uktextnorm_fuzzer -max_total_time=60 tests/data
153
+ ```
154
+
155
+ ## Development
156
+
157
+ Install `uv` and `just` for development. Run `just` to see all recipes. `just format` applies Ruff fixes and Python formatting; `just lint` runs the Python static checks; `just check` also runs Python and C++ tests.
158
+
159
+ ```sh
160
+ just setup
161
+ just format
162
+ just lint
163
+ just check
164
+ just wheel 3.15
165
+ ```
166
+
167
+ The project includes a `.clang-format` file and a CMake formatting target. Install `clang-format`, then run:
168
+
169
+ ```sh
170
+ cmake --build build --target format
171
+ ```
@@ -0,0 +1,11 @@
1
+ normalize_uk/py.typed,sha256=frcCV1k9oG9oKj3dpUqdJg1PxRT2RSN_XKdLCPjaYaY,2
2
+ normalize_uk/_normalize_uk.cp315-win_amd64.pyd,sha256=qv34QHoSAaISipbxyrrlUIilM_ZaRF5CyzxEBlE8tPo,3089408
3
+ normalize_uk/_normalize_uk.pyi,sha256=5h1sHW7-FJeskuuant4AoMfJSNAJxvgaGvunyw6pXzY,5961
4
+ normalize_uk/__init__.py,sha256=-zN5qc6V28A7mZCKcsqWWHfV48a4fS0J344J9x8oCbQ,8367
5
+ normalize_uk/__init__.pyi,sha256=UrTlJAckVhsfHykAJbCoi_BKarv2FC0o9Bu0m6qUmQA,3598
6
+ normalize_uk-0.4.2.dist-info/DELVEWHEEL,sha256=IK6a-eKDrniSWv60rLdIEX6_LO6e6-4E6Mt-RA57Xi4,411
7
+ normalize_uk-0.4.2.dist-info/METADATA,sha256=TCRiWCKAJmJuCcg_81Trh5U7lrpJV8eKmk388Ivd47o,7208
8
+ normalize_uk-0.4.2.dist-info/RECORD,,
9
+ normalize_uk-0.4.2.dist-info/WHEEL,sha256=zRSxk0lFSz9yrcoNCqCgDrmWhvhsSyignczP3ORAtmA,105
10
+ normalize_uk-0.4.2.dist-info/licenses/LICENSE,sha256=H74gUlraCIIvf60AlpOT_K4KB_BN_sH9_zhIwYkT4VU,1093
11
+ normalize_uk.libs/msvcp140-a4c2229bdc2a2a630acdc095b4d86008.dll,sha256=pMIim9wqKmMKzcCVtNhgCOXD47x3cxdDVPPaT1vrnN4,575056
@@ -0,0 +1,5 @@
1
+ Wheel-Version: 1.0
2
+ Generator: scikit-build-core 1.0.3
3
+ Root-Is-Purelib: false
4
+ Tag: cp315-cp315-win_amd64
5
+
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Yehor Smoliakov
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.